diff --git a/.github/workflows/rust.yml b/.github/workflows/rust.yml index c01593da..ccd9d23d 100644 --- a/.github/workflows/rust.yml +++ b/.github/workflows/rust.yml @@ -83,6 +83,8 @@ jobs: --locked -- --ignored --exact --list | grep -Fx "$smoke_test: test" cargo test -p cellule-app --test integration "$smoke_test" \ --locked -- --ignored --exact --nocapture + - name: Test fleet journal and public controller models + run: cargo test -p cellule-host --example fleet_operations --all-features --locked - name: Test local LTX without replica run: cargo test -p cellule-ltx --no-default-features --locked - name: Test replica feature diff --git a/Cargo.lock b/Cargo.lock index b512761a..a8363e9c 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -243,9 +243,15 @@ dependencies = [ name = "cellule-host" version = "0.1.0" dependencies = [ + "blake3", + "bytes", "cellule-app", "cellule-runtime", + "cellule-store", + "ed25519-dalek", "futures-util", + "object_store", + "rusqlite", "tempfile", "tokio", "tokio-util", diff --git a/crates/cellule-app/performance/2026-09-29-write-capacity.md b/crates/cellule-app/performance/2026-09-29-write-capacity.md index 51be4633..9c641d25 100644 --- a/crates/cellule-app/performance/2026-09-29-write-capacity.md +++ b/crates/cellule-app/performance/2026-09-29-write-capacity.md @@ -45,6 +45,49 @@ volume, object prefix, and evidence directory. The workflow uploads raw samples, logs, binary digests, and provider details even if a repeat fails. Its output requires review before a capacity claim. +## Arrival evidence collection + +The driver retains each scheduled outcome in its existing bounded sample +vector during the arrival window and accepted-work drain. It writes and flushes +the sample TSV after capturing the original end clocks. Synchronous evidence +storage therefore cannot delay the next scheduled arrival or extend the measured +drain. The offered rates, concurrency, ten-second arrival window, two-second +drain allowance, and classification of late arrivals remain unchanged. + +PR #37's follower-proof run `37060783753`, repeat 2, recorded 58/60 successful +actions at the first uniform point. Arrivals 38 and 39 were `scheduler_late` +and were never dispatched; every dispatched action succeeded. The driver had +no CPU throttling. The previous loop synchronously flushed completed samples +before scheduling each next arrival, exposing the arrival clock to evidence +storage stalls. The traces do not identify the individual stalled syscall; +deferring these writes removes that known blocking path. This change alone +does not establish a qualified capacity result: fresh provider repeats remain +required. + +## Canonical root capture after client work + +Follower-proof responses can precede object publication. After joining client +work and checking every receipt ledger, the driver observes canonical roots +under one two-second deadline for the complete original serving roster. It +pins owner session and endpoint, epoch, incarnation, code and schema. Changed, +missing, unreadable or expired authority cannot pass. The barrier only reads +authority; it does not rotate epochs or force publication. + +`capacity-root-barrier-3.tsv` records actual duration, complete read passes and +Linux boot-clock bounds. The duration includes clock reads, which are measured +separately for the centisecond clock comparison. `capacity-roots-3.tsv` retains +actual canonical roots and adds each Cell's minimum acknowledged sequence. +The independent verifier derives these minima from all arrival records and +retains the existing root coverage and original owner assertions. Scaling +stages emit the corresponding `entity-root-barrier-N.tsv` evidence. + +This post-load observation is separate from response and arrival latency. +The ten-second arrival windows, two-second accepted-work drain allowance, +offered rates, concurrency, readback, overload and follower-proof gates remain. +Each barrier must finish within two seconds; a stalled publisher still fails. +Earlier artifacts retain their historical verifier contract and cannot provide +the new barrier evidence. Later telemetry cannot repair a stale root capture. + ## First isolated object-proof result [CI run 36649534205](https://github.com/crabbuild/cellule/actions/runs/36649534205) diff --git a/crates/cellule-app/qualification/entities.py b/crates/cellule-app/qualification/entities.py index ccec9052..578a3f3b 100644 --- a/crates/cellule-app/qualification/entities.py +++ b/crates/cellule-app/qualification/entities.py @@ -330,6 +330,32 @@ def verify_root_coverage(roots: list[dict], positions: dict[int, list[int]], assert int(row["root_sequence"]) >= max(positions[entity]), "published root does not cover writes" +def verify_root_barrier(control: Path, prefix: str, nodes: int, roots: list[dict], + positions: dict[int, list[int]], windows: list[dict]) -> dict: + metadata, = rows(control / f"{prefix}-root-barrier-{nodes}.tsv") + cells = nodes * CELLS_PER_NODE + assert int(metadata["nodes"]) == nodes and int(metadata["cells"]) == cells, "root barrier roster changed" + assert int(metadata["limit_us"]) == CAPACITY_DRAIN_GRACE_US, "root barrier budget changed" + elapsed_us, reads = int(metadata["elapsed_us"]), int(metadata["reads"]) + assert 0 <= elapsed_us < CAPACITY_DRAIN_GRACE_US, "root barrier exceeded original budget" + assert reads >= cells and reads % cells == 0, "root barrier did not traverse the complete roster" + started, ended = int(metadata["started_boot_ms"]), int(metadata["ended_boot_ms"]) + last_window = max(window["ended_boot_ms"] for window in windows if window["nodes"] == nodes) + assert started >= last_window and ended >= started, "root barrier preceded accepted client work" + # /proc/uptime has 10ms resolution. Bound the difference by that bucket + # plus the measured clock reads enclosed by the driver's elapsed timer. + clock_read_us = int(metadata["clock_read_us"]) + assert 0 <= clock_read_us <= elapsed_us, "invalid root barrier clock-read duration" + assert abs((ended - started) * 1000 - elapsed_us) <= 10_000 + clock_read_us, "root barrier clocks disagree" + assert [int(row["entity"]) for row in roots] == list(range(cells)) + for row in roots: + entity = int(row["entity"]) + assert int(row["minimum_sequence"]) == max(positions[entity]), "root barrier omitted an acknowledged sequence" + return dict(nodes=nodes, cells=cells, limit_us=CAPACITY_DRAIN_GRACE_US, + elapsed_us=elapsed_us, reads=reads, started_boot_ms=started, ended_boot_ms=ended, + clock_read_us=clock_read_us) + + def verify_entities(control: Path, capacity: bool = False, follower: bool = False) -> dict: assert not follower or capacity stages = (3,) if capacity else STAGES @@ -347,7 +373,7 @@ def verify_entities(control: Path, capacity: bool = False, follower: bool = Fals ingress = list(map(int, (control / f"{evidence_prefix}-ingress-{stage}.txt").read_text().split())) assert len(ingress) == stage and min(ingress) > 0 and max(ingress) - min(ingress) <= 1 assert len({value[0] for value in identity.values()}) == stages[-1] * CELLS_PER_NODE, "entity targets collapsed" - windows, positions = [], {} + windows, positions, root_barriers = [], {}, [] for nodes in stages: if capacity: windows.extend(verify_capacity_windows(control, positions)) @@ -357,6 +383,7 @@ def verify_entities(control: Path, capacity: bool = False, follower: bool = Fals windows.append(verify_window(control, nodes, shape, rate, concurrency, len(windows), positions)) roots = rows(control / f"{evidence_prefix}-roots-{nodes}.tsv") verify_root_coverage(roots, positions, identity, nodes * CELLS_PER_NODE) + root_barriers.append(verify_root_barrier(control, evidence_prefix, nodes, roots, positions, windows)) for node in range(nodes): assert sum(window["acknowledged_writes_by_node"][node] for window in windows if window["nodes"] == nodes) > 0 resources = {} @@ -441,7 +468,7 @@ def verify_entities(control: Path, capacity: bool = False, follower: bool = Fals node["published_roots_per_second"] for node in fully_served["node_durability"].values()), ) extra["capacity_curves"] = capacity_curves - return dict(integrity_verified=True, windows=windows, resources=resources, + return dict(integrity_verified=True, windows=windows, resources=resources, root_barriers=root_barriers, verified_cells=len(positions), acknowledged_writes=sum(map(len, positions.values())), raw_sha256={path.name: hashlib.sha256(path.read_bytes()).hexdigest() for path in sorted(control.glob("*.tsv"))}, **extra) diff --git a/crates/cellule-app/qualification/test_entities.py b/crates/cellule-app/qualification/test_entities.py index f9b4290c..3c9c50c8 100644 --- a/crates/cellule-app/qualification/test_entities.py +++ b/crates/cellule-app/qualification/test_entities.py @@ -6,7 +6,7 @@ import unittest from unittest.mock import patch -from entities import destination, verify_capacity_windows, verify_follower_proof, verify_object_operations, verify_root_coverage, verify_timing_evidence, verify_window +from entities import destination, verify_capacity_windows, verify_follower_proof, verify_object_operations, verify_root_barrier, verify_root_coverage, verify_timing_evidence, verify_window class EntityWindowEvidence(unittest.TestCase): @@ -228,6 +228,79 @@ def test_rate_mislabeled_as_fully_served_is_rejected(self): verify_capacity_windows(self.root, {}) +class RootBarrierEvidence(unittest.TestCase): + def setUp(self): + temporary = tempfile.TemporaryDirectory() + self.addCleanup(temporary.cleanup) + self.control = Path(temporary.name) + self.path = self.control / "capacity-root-barrier-3.tsv" + self.metadata = dict(nodes="3", cells="12", limit_us="2000000", elapsed_us="10000", + reads="24", started_boot_ms="110000", ended_boot_ms="110010", clock_read_us="0") + self.roots = [dict(entity=str(entity), minimum_sequence="5") for entity in range(12)] + self.positions = {entity: [2, 3, 5] for entity in range(12)} + self.windows = [dict(nodes=3, ended_boot_ms=110000)] + + def verify(self): + self.path.write_text("\t".join(self.metadata) + "\n" + "\t".join(self.metadata.values()) + "\n") + return verify_root_barrier(self.control, "capacity", 3, self.roots, self.positions, self.windows) + + def test_complete_original_roster_and_budget_are_required(self): + self.assertEqual(self.verify()["reads"], 24) + + def test_missing_barrier_is_rejected(self): + with self.assertRaises(FileNotFoundError): + verify_root_barrier(self.control, "capacity", 3, self.roots, self.positions, self.windows) + + def test_late_and_extended_barriers_are_rejected(self): + for field, value, message in [("elapsed_us", "2000000", "exceeded original budget"), + ("elapsed_us", "2000001", "exceeded original budget"), + ("elapsed_us", "-1", "exceeded original budget"), + ("limit_us", "2000001", "budget changed")]: + with self.subTest(field=field, value=value): + original = self.metadata[field] + self.metadata[field] = value + with self.assertRaisesRegex(AssertionError, message): + self.verify() + self.metadata[field] = original + + def test_incomplete_roster_and_read_passes_are_rejected(self): + for field, value, message in [("cells", "11", "roster changed"), + ("nodes", "2", "roster changed"), + ("reads", "11", "complete roster"), + ("reads", "13", "complete roster")]: + with self.subTest(field=field, value=value): + original = self.metadata[field] + self.metadata[field] = value + with self.assertRaisesRegex(AssertionError, message): + self.verify() + self.metadata[field] = original + + def test_barrier_cannot_precede_client_work_or_forge_elapsed_time(self): + for field, value, message in [("started_boot_ms", "109999", "preceded accepted"), + ("ended_boot_ms", "109999", "preceded accepted"), + ("elapsed_us", "21001", "clocks disagree")]: + with self.subTest(field=field, value=value): + original = self.metadata[field] + self.metadata[field] = value + with self.assertRaisesRegex(AssertionError, message): + self.verify() + self.metadata[field] = original + + def test_each_minimum_is_derived_from_all_independent_acknowledgements(self): + for minimum in ("3", "6"): + self.roots[-1]["minimum_sequence"] = minimum + with self.assertRaisesRegex(AssertionError, "omitted an acknowledged sequence"): + self.verify() + + def test_clock_read_duration_is_measured_inside_the_original_budget(self): + for duration in ("-1", "10001"): + self.metadata["clock_read_us"] = duration + with self.assertRaisesRegex(AssertionError, "clock-read duration"): + self.verify() + self.metadata.update(clock_read_us="100", elapsed_us="20100") + self.assertEqual(self.verify()["clock_read_us"], 100) + + class ObjectOperationEvidence(unittest.TestCase): def test_missing_provider_operation_is_rejected(self): with tempfile.TemporaryDirectory() as root: diff --git a/crates/cellule-app/tests/entities/process.rs b/crates/cellule-app/tests/entities/process.rs index b3d1745f..dc6ddbed 100644 --- a/crates/cellule-app/tests/entities/process.rs +++ b/crates/cellule-app/tests/entities/process.rs @@ -19,6 +19,7 @@ use tokio::net::TcpListener; mod driver; mod observation; +mod root_capture; async fn endpoints(sync: &Path, count: usize) -> Vec { let mut endpoints = Vec::new(); diff --git a/crates/cellule-app/tests/entities/process/driver.rs b/crates/cellule-app/tests/entities/process/driver/mod.rs similarity index 84% rename from crates/cellule-app/tests/entities/process/driver.rs rename to crates/cellule-app/tests/entities/process/driver/mod.rs index aa84a7ab..966510f6 100644 --- a/crates/cellule-app/tests/entities/process/driver.rs +++ b/crates/cellule-app/tests/entities/process/driver/mod.rs @@ -1,5 +1,6 @@ //! Scheduled arrivals retain overload and verify every resulting Cell ledger. +use super::root_capture::{DRAIN_GRACE_US, capture}; use super::*; use crate::fleet::{balancer_round_trip, start_balancer}; use crate::process_performance::Controller; @@ -13,8 +14,9 @@ use std::{ }; use tokio::task::JoinSet; +mod tests; + const WINDOW_SECONDS: usize = 10; -const CAPACITY_DRAIN_GRACE_US: u64 = 2_000_000; struct Window { id: usize, @@ -83,6 +85,7 @@ async fn run_entity_process(capacity: bool, follower_enabled: bool) { BufWriter::new(File::create(sync.join(format!("{evidence_prefix}-owners.tsv"))).unwrap()); writeln!(owners, "stage\tentity\tcell\towner\tepoch\tincarnation").unwrap(); let mut expected = Vec::new(); + let mut latest_sequences = Vec::new(); let mut window_id = 0; let mut capacity_windows = capacity.then(|| { let mut output = BufWriter::new(File::create(sync.join("capacity-windows.tsv")).unwrap()); @@ -127,6 +130,8 @@ async fn run_entity_process(capacity: bool, follower_enabled: bool) { .unwrap(); let client = Arc::new(EntityReferenceClient::new(handle).unwrap()); expected.resize(nodes * ENTITIES_PER_NODE, 0_u64); + latest_sequences.resize(expected.len(), 0_u64); + let mut original_controls = Vec::with_capacity(expected.len()); for (entity, count) in expected.iter().enumerate() { let target = entity_target(&application, entity); let control = authority.load(target.cell_id()).await.unwrap().unwrap(); @@ -134,6 +139,7 @@ async fn run_entity_process(capacity: bool, follower_enabled: bool) { let node = entity / ENTITIES_PER_NODE; assert_eq!(owner.session, node_session(node)); assert_eq!(owner.endpoint, format!("https://{}", addresses[node])); + original_controls.push(control.value().clone()); writeln!( owners, "{nodes}\t{entity}\t{:?}\t{node}\t{}\t{:?}", @@ -177,7 +183,14 @@ async fn run_entity_process(capacity: bool, follower_enabled: bool) { rate_per_node, concurrency, }; - let fully_served = run_window(sync, &window, client.clone(), &mut expected).await; + let fully_served = run_window( + sync, + &window, + client.clone(), + &mut expected, + &mut latest_sequences, + ) + .await; if let Some(output) = capacity_windows.as_mut() { writeln!( output, @@ -198,29 +211,48 @@ async fn run_entity_process(capacity: bool, follower_enabled: bool) { assert!(overloaded, "{shape}: rate ramp did not reach overload"); } } + let barrier_started = Instant::now(); + let started_boot_ms = boot_ms(); + let start_clock_us = barrier_started.elapsed().as_micros() as u64; + let captured = capture(&authority, &original_controls, &latest_sequences) + .await + .unwrap(); + let end_clock_started = Instant::now(); + let ended_boot_ms = boot_ms(); + let clock_read_us = start_clock_us + end_clock_started.elapsed().as_micros() as u64; + let elapsed_us = barrier_started.elapsed().as_micros() as u64; + assert!(elapsed_us >= captured.elapsed_us && elapsed_us < DRAIN_GRACE_US); + publish_marker( + &sync.join(format!("{evidence_prefix}-root-barrier-{nodes}.tsv")), + format!( + "nodes\tcells\tlimit_us\telapsed_us\treads\tstarted_boot_ms\tended_boot_ms\tclock_read_us\n{nodes}\t{}\t{DRAIN_GRACE_US}\t{}\t{}\t{started_boot_ms}\t{ended_boot_ms}\t{clock_read_us}\n", + captured.values.len(), + elapsed_us, + captured.reads, + ), + ); let mut roots = BufWriter::new( File::create(sync.join(format!("{evidence_prefix}-roots-{nodes}.tsv"))).unwrap(), ); writeln!( roots, - "entity\tcell\towner\tepoch\tincarnation\troot_sequence\troot_digest" + "entity\tcell\towner\tepoch\tincarnation\troot_sequence\troot_digest\tminimum_sequence" ) .unwrap(); - for entity in 0..expected.len() { - let target = entity_target(&application, entity); - let control = authority.load(target.cell_id()).await.unwrap().unwrap(); - let owner = control.value().owner.as_ref().unwrap(); + for (entity, control) in captured.values.iter().enumerate() { + let owner = control.owner.as_ref().unwrap(); let node = entity / ENTITIES_PER_NODE; assert_eq!(owner.session, node_session(node)); - let root = control.value().root.as_ref().unwrap(); + let root = control.root.as_ref().unwrap(); writeln!( roots, - "{entity}\t{:?}\t{node}\t{}\t{:?}\t{}\t{:?}", - target.cell_id(), - control.value().epoch, - control.value().incarnation, + "{entity}\t{:?}\t{node}\t{}\t{:?}\t{}\t{:?}\t{}", + control.cell, + control.epoch, + control.incarnation, root.commit_sequence, - root.digest + root.digest, + latest_sequences[entity], ) .unwrap(); } @@ -283,7 +315,9 @@ async fn run_window( window: &Window, client: Arc, expected: &mut [u64], + latest_sequences: &mut [u64], ) -> bool { + assert_eq!(expected.len(), latest_sequences.len()); let rate = window.nodes * window.rate_per_node; let planned = rate * WINDOW_SECONDS; let label = window.label(); @@ -305,18 +339,21 @@ async fn run_window( |sample, request, admitted| execute(Arc::clone(&client), sample, request, admitted), ) .await; - // Evidence I/O must not block scheduled arrivals. Keep its cost in the - // measured window, but flush only after the offered work has drained. - write_samples(&mut output, &samples); tokio::time::sleep_until((started + Duration::from_secs(WINDOW_SECONDS as u64)).into()).await; let elapsed_us = started.elapsed().as_micros() as u64; let ended_ms = now_ms(); let ended_boot_ms = boot_ms(); + // Keep synchronous evidence I/O outside the arrival and drain clocks. A + // slow bind mount must not turn a completed request into missed arrivals. + // The already bounded sample vector retains every outcome, including real + // scheduler lateness and concurrency refusals, without a catch-up burst. + write_samples(&mut output, &samples).unwrap(); samples.sort_by_key(|sample| sample.arrival); assert_eq!(samples.len(), planned); for sample in &samples { if sample.write && sample.sequence > 0 { expected[sample.entity] += 1; + latest_sequences[sample.entity] = latest_sequences[sample.entity].max(sample.sequence); } } let mut checks = @@ -355,9 +392,7 @@ async fn run_window( println!( "ENTITY_WINDOW label={label} planned={planned} complete={complete} elapsed_us={elapsed_us}" ); - complete == planned - && (window.prefix != "capacity" - || elapsed_us <= WINDOW_SECONDS as u64 * 1_000_000 + CAPACITY_DRAIN_GRACE_US) + fully_served(&samples, elapsed_us, window.prefix == "capacity") } async fn collect_arrivals( @@ -411,7 +446,14 @@ where samples } -fn write_samples(output: &mut impl Write, samples: &[Sample]) { +fn fully_served(samples: &[Sample], elapsed_us: u64, capacity: bool) -> bool { + samples + .iter() + .all(|sample| sample.outcome == "ok" || sample.outcome == "resolved") + && (!capacity || elapsed_us <= WINDOW_SECONDS as u64 * 1_000_000 + DRAIN_GRACE_US) +} + +fn write_samples(output: &mut impl Write, samples: &[Sample]) -> std::io::Result<()> { for sample in samples { writeln!( output, @@ -426,10 +468,9 @@ fn write_samples(output: &mut impl Write, samples: &[Sample]) { sample.sequence, sample.read_sequence, sample.count - ) - .unwrap(); + )?; } - output.flush().unwrap(); + output.flush() } async fn execute( @@ -519,77 +560,3 @@ async fn execute( sample.elapsed_us = started.elapsed().as_micros() as u64; sample } - -#[tokio::test] -async fn slow_evidence_flush_does_not_drop_scheduled_arrivals() { - struct SlowFlush(Vec); - impl Write for SlowFlush { - fn write(&mut self, bytes: &[u8]) -> std::io::Result { - self.0.extend_from_slice(bytes); - Ok(bytes.len()) - } - fn flush(&mut self) -> std::io::Result<()> { - std::thread::sleep(Duration::from_millis(750)); - Ok(()) - } - } - let window = Window { - id: 0, - prefix: "capacity", - nodes: 1, - shape: "uniform", - rate_per_node: 2, - concurrency: 8, - }; - let mut output = SlowFlush(Vec::new()); - let samples = collect_arrivals( - &window, - 3, - Instant::now(), - 12, - |mut sample, _, _| async move { - sample.outcome = "ok"; - sample - }, - ) - .await; - write_samples(&mut output, &samples); - assert_eq!(samples.len(), 3); - assert!( - samples.iter().all(|sample| sample.outcome == "ok"), - "evidence I/O under-offered the workload: {:?}", - samples - .iter() - .map(|s| (s.arrival, s.started_us, s.outcome)) - .collect::>() - ); - assert_eq!(String::from_utf8(output.0).unwrap().lines().count(), 3); -} - -#[tokio::test] -async fn missed_arrivals_remain_visible_without_dispatch() { - let window = Window { - id: 0, - prefix: "capacity", - nodes: 1, - shape: "uniform", - rate_per_node: 2, - concurrency: 8, - }; - let mut output = Vec::new(); - let samples = collect_arrivals( - &window, - 2, - Instant::now() - Duration::from_secs(2), - 12, - |_, _, _| async { panic!("missed arrival was dispatched") }, - ) - .await; - write_samples(&mut output, &samples); - assert!( - samples - .iter() - .all(|s| s.outcome == "scheduler_late" && s.elapsed_us == 0) - ); - assert_eq!(String::from_utf8(output).unwrap().lines().count(), 2); -} diff --git a/crates/cellule-app/tests/entities/process/driver/tests.rs b/crates/cellule-app/tests/entities/process/driver/tests.rs new file mode 100644 index 00000000..e2c69b4e --- /dev/null +++ b/crates/cellule-app/tests/entities/process/driver/tests.rs @@ -0,0 +1,147 @@ +use super::*; + +fn sample(arrival: usize, outcome: &'static str) -> Sample { + Sample { + arrival, + scheduled_us: 1_000, + started_us: 2_000, + elapsed_us: 3_000, + entity: 4, + write: true, + outcome, + sequence: 5, + read_sequence: 6, + count: 7, + } +} + +#[test] +fn deferred_evidence_retains_every_outcome_and_original_timing() { + let outcomes = [ + "ok", + "resolved", + "scheduler_late", + "client_full", + "not_started", + ]; + let samples: Vec<_> = outcomes + .iter() + .enumerate() + .map(|(arrival, outcome)| sample(arrival, outcome)) + .collect(); + let mut output = Vec::new(); + write_samples(&mut output, &samples).unwrap(); + let output = String::from_utf8(output).unwrap(); + let expected = outcomes + .iter() + .enumerate() + .map(|(arrival, outcome)| { + format!("{arrival}\t1000\t2000\t3000\t4\twrite\t{outcome}\t5\t6\t7\n") + }) + .collect::(); + assert_eq!(output, expected); +} + +struct FailedEvidence; + +impl Write for FailedEvidence { + fn write(&mut self, _: &[u8]) -> std::io::Result { + Err(std::io::Error::other("evidence storage unavailable")) + } + + fn flush(&mut self) -> std::io::Result<()> { + Ok(()) + } +} + +#[test] +fn deferred_evidence_failure_is_reported() { + let error = write_samples(&mut FailedEvidence, &[sample(0, "ok")]).unwrap_err(); + assert_eq!(error.kind(), std::io::ErrorKind::Other); + assert_eq!(error.to_string(), "evidence storage unavailable"); +} + +#[test] +fn capacity_still_rejects_missed_arrivals_and_slow_drain() { + let limit = WINDOW_SECONDS as u64 * 1_000_000 + DRAIN_GRACE_US; + let samples = [sample(0, "ok"), sample(1, "resolved")]; + assert!(fully_served(&samples, limit, true)); + assert!(!fully_served(&samples, limit + 1, true)); + for outcome in ["scheduler_late", "client_full", "not_started", "write_only"] { + assert!(!fully_served(&[sample(0, outcome)], 1, true)); + } +} + +#[tokio::test] +async fn slow_evidence_flush_does_not_drop_scheduled_arrivals() { + struct SlowFlush(Vec); + impl Write for SlowFlush { + fn write(&mut self, bytes: &[u8]) -> std::io::Result { + self.0.extend_from_slice(bytes); + Ok(bytes.len()) + } + fn flush(&mut self) -> std::io::Result<()> { + std::thread::sleep(Duration::from_millis(750)); + Ok(()) + } + } + let window = Window { + id: 0, + prefix: "capacity", + nodes: 1, + shape: "uniform", + rate_per_node: 2, + concurrency: 8, + }; + let mut output = SlowFlush(Vec::new()); + let samples = collect_arrivals( + &window, + 3, + Instant::now(), + 12, + |mut sample, _, _| async move { + sample.outcome = "ok"; + sample + }, + ) + .await; + write_samples(&mut output, &samples).unwrap(); + assert_eq!(samples.len(), 3); + assert!( + samples.iter().all(|sample| sample.outcome == "ok"), + "evidence I/O under-offered the workload: {:?}", + samples + .iter() + .map(|s| (s.arrival, s.started_us, s.outcome)) + .collect::>() + ); + assert_eq!(String::from_utf8(output.0).unwrap().lines().count(), 3); +} + +#[tokio::test] +async fn missed_arrivals_remain_visible_without_dispatch() { + let window = Window { + id: 0, + prefix: "capacity", + nodes: 1, + shape: "uniform", + rate_per_node: 2, + concurrency: 8, + }; + let mut output = Vec::new(); + let samples = collect_arrivals( + &window, + 2, + Instant::now() - Duration::from_secs(2), + 12, + |_, _, _| async { panic!("missed arrival was dispatched") }, + ) + .await; + write_samples(&mut output, &samples).unwrap(); + assert!( + samples + .iter() + .all(|s| s.outcome == "scheduler_late" && s.elapsed_us == 0) + ); + assert_eq!(String::from_utf8(output).unwrap().lines().count(), 2); +} diff --git a/crates/cellule-app/tests/entities/process/root_capture/mod.rs b/crates/cellule-app/tests/entities/process/root_capture/mod.rs new file mode 100644 index 00000000..038c4eb3 --- /dev/null +++ b/crates/cellule-app/tests/entities/process/root_capture/mod.rs @@ -0,0 +1,93 @@ +//! Bounded canonical publication observation after the offered-load window. + +use cellule_runtime::control::{Control, ControlState, authority::CellAuthority}; +use cellule_runtime::{Error, Result}; +use std::{ + collections::HashSet, + time::{Duration, Instant}, +}; + +pub(super) const DRAIN_GRACE_US: u64 = 2_000_000; + +pub(super) struct CapturedRoots { + pub(super) values: Vec, + pub(super) elapsed_us: u64, + pub(super) reads: usize, +} + +pub(super) async fn capture( + authority: &CellAuthority, + expected: &[Control], + minimum: &[u64], +) -> Result { + let started = Instant::now(); + let deadline = started + Duration::from_micros(DRAIN_GRACE_US); + let mut cells = HashSet::new(); + if expected.is_empty() + || expected.len() > 80 + || expected.len() != minimum.len() + || expected.iter().any(|original| { + original.state != ControlState::Serving + || original.owner.is_none() + || original.recovery.is_some() + || original.root.is_none() + || !cells.insert(original.cell) + }) + { + return Err(Error::Control("invalid qualified root capture inputs")); + } + let mut reads = 0; + loop { + let mut values = Vec::with_capacity(expected.len()); + let mut covered = true; + for (original, minimum) in expected.iter().zip(minimum) { + let observed = tokio::time::timeout_at(deadline.into(), authority.load(original.cell)) + .await + .map_err(|_| Error::Deadline)?? + .ok_or(Error::CellNotActive)?; + let current = observed.value(); + if Instant::now() >= deadline { + return Err(Error::Deadline); + } + if current.cell != original.cell + || current.incarnation != original.incarnation + || current.epoch != original.epoch + || current.owner != original.owner + || current.code != original.code + || current.schema != original.schema + || current.state != ControlState::Serving + || current.recovery.is_some() + || current.revision < original.revision + || current.progress < original.progress + { + return Err(Error::Fenced); + } + let root = current.root.as_ref().ok_or(Error::Fenced)?; + covered &= root.commit_sequence >= *minimum; + reads += 1; + values.push(current.clone()); + } + if covered { + let elapsed_us = started.elapsed().as_micros() as u64; + if elapsed_us >= DRAIN_GRACE_US { + return Err(Error::Deadline); + } + return Ok(CapturedRoots { + reads, + values, + elapsed_us, + }); + } + // A follower proof can return before object publication. Observe the + // canonical publisher within one deadline for the entire original roster; + // do not rotate authority or charge this readback to arrival latency. + tokio::time::timeout_at( + deadline.into(), + tokio::time::sleep(Duration::from_millis(10)), + ) + .await + .map_err(|_| Error::Deadline)?; + } +} + +mod tests; diff --git a/crates/cellule-app/tests/entities/process/root_capture/tests.rs b/crates/cellule-app/tests/entities/process/root_capture/tests.rs new file mode 100644 index 00000000..35ec48ff --- /dev/null +++ b/crates/cellule-app/tests/entities/process/root_capture/tests.rs @@ -0,0 +1,483 @@ +use super::*; +use crate::entities::{compiled_entities, entity_target, provision_entity}; +use crate::performance_fixture::{node_session, now_ms}; +use cellule_host::{CellNode, CellNodeBuilder}; +use cellule_runtime::cell::actor::CellHandle; +use cellule_runtime::cell::executor::HandlerOutcome; +use cellule_runtime::cell::worker::SqlWorkerPool; +use cellule_runtime::node::lease::NodeLeaseGuard; +use cellule_runtime::{ + Digest, + ltx::{CellStorageLayout, DiskBudget, Host}, +}; +use cellule_store::{StorageObservation, StorageObserver, StorageOperation, Store}; +use object_store::memory::InMemory; +use std::sync::{ + Arc, + atomic::{AtomicBool, AtomicUsize, Ordering}, +}; +use tokio_util::sync::CancellationToken; + +#[derive(Default)] +struct FirstRead { + armed: AtomicBool, + observed: tokio::sync::Notify, +} +impl StorageObserver for FirstRead { + fn started(&self, _: StorageOperation) {} + + fn finished(&self, observation: StorageObservation) { + if observation.operation == StorageOperation::Get + && self.armed.swap(false, Ordering::SeqCst) + { + self.observed.notify_one(); + } + } +} +struct Fixture { + _root: tempfile::TempDir, + node: CellNode, + handle: CellHandle, + authority: CellAuthority, + original: Control, + observer: Arc, + store: Store, + layout: CellStorageLayout, +} +impl Fixture { + async fn new() -> Self { + let application = compiled_entities(); + let observer = Arc::new(FirstRead::default()); + let store = Store::new(Arc::new(InMemory::new())).with_storage_observer(observer.clone()); + let layout = CellStorageLayout::new(store.clone(), "root-capture".into(), [82; 16]); + let root = tempfile::tempdir().unwrap(); + let node = CellNodeBuilder::new(application.clone()) + .with_runtime(SqlWorkerPool::new(1, 8).unwrap(), 64 << 20) + .with_session(node_session(0)) + .with_replica_host(Host::default().with_local_disk_budget(DiskBudget::new(8 << 30))) + .build() + .unwrap(); + node.install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + let now = now_ms(); + node.install_node_lease_for_startup(NodeLeaseGuard::new(now, now + 60_000).unwrap()) + .unwrap(); + node.start().unwrap(); + let handle = + provision_entity(&node, &layout, root.path(), 0, 0, "https://node-0".into()).await; + let authority = CellAuthority::new(layout.clone()); + let original = authority + .load(entity_target(&application, 0).cell_id()) + .await + .unwrap() + .unwrap() + .value() + .clone(); + Self { + _root: root, + node, + handle, + authority, + original, + observer, + store, + layout, + } + } + async fn close(&self) { + self.node.shutdown().await.unwrap(); + let stats = self.node.stats(); + assert_eq!(stats.active_cells(), 0); + assert_eq!(stats.retained_bytes(), 0); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.file_descriptors(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); + } + + fn admitted_authority(&self, reads: Arc) -> CellAuthority { + CellAuthority::new(CellStorageLayout::new( + self.store.clone().with_read_admission(reads), + "root-capture".into(), + [82; 16], + )) + } + + async fn roster(&self, cells: usize) -> Vec { + let mut values = vec![self.original.clone()]; + for entity in 1..cells { + provision_entity( + &self.node, + &self.layout, + self._root.path(), + 0, + entity, + "https://node-0".into(), + ) + .await; + values.push( + self.authority + .load(entity_target(self.node.application(), entity).cell_id()) + .await + .unwrap() + .unwrap() + .value() + .clone(), + ); + } + values + } +} + +#[derive(Default)] +struct ReadGate { + mode: usize, + cancellation: CancellationToken, + requests: AtomicUsize, + active: AtomicUsize, + entered: tokio::sync::Notify, +} +struct ActiveRead<'a>(&'a ReadGate); +impl Drop for ActiveRead<'_> { + fn drop(&mut self) { + self.0.active.fetch_sub(1, Ordering::SeqCst); + } +} +#[derive(Debug)] +struct OriginalReadFailure; +impl std::fmt::Display for OriginalReadFailure { + fn fmt(&self, output: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + output.write_str("original root observation read failure") + } +} +impl std::error::Error for OriginalReadFailure {} + +#[async_trait::async_trait] +impl cellule_store::ReadAdmission for ReadGate { + fn cancellation(&self) -> &CancellationToken { + &self.cancellation + } + + async fn request(&self) -> std::result::Result<(), Box> { + self.requests.fetch_add(1, Ordering::SeqCst); + self.active.fetch_add(1, Ordering::SeqCst); + let _active = ActiveRead(self); + self.entered.notify_one(); + match self.mode { + 1 => std::future::pending().await, + 2 => Err(Box::new(OriginalReadFailure)), + 3 => { + tokio::time::sleep(Duration::from_millis(600)).await; + Ok(()) + } + _ => Ok(()), + } + } + + async fn bytes( + &self, + _: u64, + ) -> std::result::Result<(), Box> { + Ok(()) + } +} + +#[tokio::test] +async fn canonical_capture_waits_for_the_original_writer_to_publish_the_required_sequence() { + let fixture = Fixture::new().await; + let minimum = fixture.original.root.as_ref().unwrap().commit_sequence + 1; + let minima = [minimum]; + fixture.observer.armed.store(true, Ordering::SeqCst); + let capture = capture( + &fixture.authority, + std::slice::from_ref(&fixture.original), + &minima, + ); + tokio::pin!(capture); + let advance = async { + fixture.observer.observed.notified().await; + fixture + .handle + .execute( + crate::qualification_identity(98, now_ms()), + Digest::from_bytes([99; 32]), + now_ms(), + 1024, + 1024, + |tx| { + tx.execute("CREATE TABLE root_capture_probe(value INTEGER)", [])?; + Ok(HandlerOutcome::Success(Vec::new())) + }, + ) + .await + .unwrap(); + }; + let (captured, ()) = tokio::join!(&mut capture, advance); + fixture.close().await; + let captured = captured.unwrap(); + assert_eq!(captured.values.len(), 1); + assert!(captured.reads >= 2); + assert_eq!(captured.values[0].epoch, fixture.original.epoch); + assert_eq!(captured.values[0].owner, fixture.original.owner); + assert!(captured.values[0].root.as_ref().unwrap().commit_sequence >= minimum); + assert!(captured.elapsed_us <= DRAIN_GRACE_US); + let control = &captured.values[0]; + let root = control.ltx_root().unwrap(); + let replica = cellule_ltx::CellReplica::new( + fixture.layout.clone(), + *control.cell.as_bytes(), + *control.incarnation.as_bytes(), + cellule_ltx::Limits::default(), + ) + .unwrap() + .with_host(Host::default().with_local_disk_budget(DiskBudget::new(8 << 30))); + let restored = fixture._root.path().join("captured-root.sqlite"); + assert_eq!( + replica + .open_root(&root) + .await + .unwrap() + .restore(&restored) + .await + .unwrap(), + root.position + ); + let database = cellule_ltx::rusqlite::Connection::open(restored).unwrap(); + assert_eq!( + database + .query_row("SELECT count(*) FROM root_capture_probe", [], |row| row + .get::<_, u64>(0)) + .unwrap(), + 0 + ); + assert_eq!( + database + .query_row( + "SELECT count(*) FROM sys_requests WHERE commit_sequence = ?1", + [minimum], + |row| row.get::<_, u64>(0) + ) + .unwrap(), + 1 + ); +} + +#[tokio::test] +async fn every_original_writer_binding_is_required_even_when_the_root_already_covers_writes() { + let fixture = Fixture::new().await; + let mut originals = Vec::new(); + let mut changed = fixture.original.clone(); + changed.owner.as_mut().unwrap().endpoint = "https://unexpected-endpoint".into(); + originals.push(changed); + let mut changed = fixture.original.clone(); + changed.owner.as_mut().unwrap().session = node_session(1); + originals.push(changed); + let mut changed = fixture.original.clone(); + changed.epoch += 1; + originals.push(changed); + let mut changed = fixture.original.clone(); + changed.incarnation = cellule_runtime::identity::IncarnationId::from_bytes([99; 16]); + originals.push(changed); + let mut changed = fixture.original.clone(); + changed.code = Digest::from_bytes([99; 32]); + originals.push(changed); + let mut changed = fixture.original.clone(); + changed.schema += 1; + originals.push(changed); + let mut changed = fixture.original.clone(); + changed.revision += 1; + originals.push(changed); + let mut changed = fixture.original.clone(); + changed.progress += 1; + originals.push(changed); + for original in &originals { + assert!(matches!( + capture(&fixture.authority, std::slice::from_ref(original), &[0]).await, + Err(Error::Fenced) + )); + } + let observed = fixture + .authority + .load(fixture.original.cell) + .await + .unwrap() + .unwrap(); + fixture.close().await; + assert_eq!(observed.value(), &fixture.original); +} + +#[tokio::test] +async fn the_complete_roster_shares_one_original_deadline_including_read_admission() { + let fixture = Fixture::new().await; + let roster = fixture.roster(4).await; + let minimum = roster + .iter() + .map(|control| control.root.as_ref().unwrap().commit_sequence) + .collect::>(); + let reads = Arc::new(ReadGate { + mode: 3, + ..ReadGate::default() + }); + let result = capture( + &fixture.admitted_authority(reads.clone()), + &roster, + &minimum, + ) + .await; + fixture.close().await; + assert!(matches!(result, Err(Error::Deadline))); + assert_eq!(reads.requests.load(Ordering::SeqCst), 4); + assert_eq!(reads.active.load(Ordering::SeqCst), 0); +} + +#[tokio::test] +async fn an_original_writer_that_never_covers_the_minimum_cannot_pass() { + let fixture = Fixture::new().await; + let minimum = fixture.original.root.as_ref().unwrap().commit_sequence + 1; + let result = capture( + &fixture.authority, + std::slice::from_ref(&fixture.original), + &[minimum], + ) + .await; + let observed = fixture + .authority + .load(fixture.original.cell) + .await + .unwrap() + .unwrap(); + fixture.close().await; + assert!(matches!(result, Err(Error::Deadline))); + assert_eq!(observed.value(), &fixture.original); +} + +#[tokio::test] +async fn capture_preserves_the_original_read_failure_instead_of_waiting_or_retrying() { + let fixture = Fixture::new().await; + let reads = Arc::new(ReadGate { + mode: 2, + ..ReadGate::default() + }); + let result = capture( + &fixture.admitted_authority(reads.clone()), + std::slice::from_ref(&fixture.original), + &[0], + ) + .await; + fixture.close().await; + let error = result.err().unwrap(); + let mut source: &(dyn std::error::Error + 'static) = &error; + while !source.is::() { + source = source.source().unwrap(); + } + assert_eq!(reads.requests.load(Ordering::SeqCst), 1); + assert_eq!(reads.active.load(Ordering::SeqCst), 0); +} + +#[tokio::test] +async fn dropping_capture_cancels_only_observation_and_retains_the_original_writer() { + let fixture = Fixture::new().await; + let reads = Arc::new(ReadGate { + mode: 1, + ..ReadGate::default() + }); + let authority = fixture.admitted_authority(reads.clone()); + let mut capture = Box::pin(capture( + &authority, + std::slice::from_ref(&fixture.original), + &[0], + )); + tokio::select! { + result = &mut capture => panic!("unexpected capture result: {}", result.is_ok()), + () = reads.entered.notified() => {}, + } + assert_eq!(reads.active.load(Ordering::SeqCst), 1); + drop(capture); + assert_eq!(reads.active.load(Ordering::SeqCst), 0); + let observed = fixture + .authority + .load(fixture.original.cell) + .await + .unwrap() + .unwrap(); + fixture.close().await; + assert_eq!(observed.value(), &fixture.original); +} + +#[tokio::test] +async fn closed_missing_and_successor_authority_cannot_certify_the_original_serving_roster() { + use cellule_runtime::control::{Owner, Transition}; + let fixture = Fixture::new().await; + fixture.close().await; + let result = capture( + &fixture.authority, + std::slice::from_ref(&fixture.original), + &[0], + ) + .await; + assert!(matches!(result, Err(Error::Fenced))); + let observed = fixture + .authority + .load(fixture.original.cell) + .await + .unwrap() + .unwrap(); + let successor = observed + .value() + .takeover(Owner { + session: node_session(1), + endpoint: "https://node-1".into(), + }) + .unwrap(); + fixture + .authority + .transition(&observed, successor, Transition::Takeover) + .await + .unwrap(); + let result = capture( + &fixture.authority, + std::slice::from_ref(&fixture.original), + &[0], + ) + .await; + assert!(matches!(result, Err(Error::Fenced))); + fixture + .store + .delete( + &fixture + .layout + .control_path(fixture.original.cell.as_bytes()), + ) + .await + .unwrap(); + let result = capture( + &fixture.authority, + std::slice::from_ref(&fixture.original), + &[0], + ) + .await; + assert!(matches!(result, Err(Error::CellNotActive))); +} + +#[tokio::test] +async fn duplicate_empty_mismatched_and_unbounded_rosters_are_rejected_before_reading() { + let fixture = Fixture::new().await; + let reads = Arc::new(ReadGate { + mode: 1, + ..ReadGate::default() + }); + let authority = fixture.admitted_authority(reads.clone()); + for (roster, minimum) in [ + (Vec::new(), Vec::new()), + (vec![fixture.original.clone()], Vec::new()), + (vec![fixture.original.clone(); 2], vec![0; 2]), + (vec![fixture.original.clone(); 81], vec![0; 81]), + ] { + assert!(matches!( + capture(&authority, &roster, &minimum).await, + Err(Error::Control(_)) + )); + } + fixture.close().await; + assert_eq!(reads.requests.load(Ordering::SeqCst), 0); +} diff --git a/crates/cellule-app/tests/host/replicas.rs b/crates/cellule-app/tests/host/replicas.rs index 007ac133..027afeee 100644 --- a/crates/cellule-app/tests/host/replicas.rs +++ b/crates/cellule-app/tests/host/replicas.rs @@ -278,9 +278,9 @@ async fn publication_hints_with_gate(readers: usize, stalled_reader: bool, hold_ .as_ref() .filter(|_| occurrence == 1) .map(|held| held.session); - let started = std::time::Instant::now(); + let started = tokio::time::Instant::now(); let mut notified = std::collections::HashSet::new(); - let completed = tokio::time::timeout(Duration::from_secs(2), async { + let received = tokio::time::timeout(Duration::from_secs(2), async { for _ in 0..healthy.len() - usize::from(held_session.is_some()) { let (cell, session) = hints.recv().await.unwrap(); assert_eq!(cell, target.cell_id()); @@ -293,15 +293,15 @@ async fn publication_hints_with_gate(readers: usize, stalled_reader: bool, hold_ }) .await; assert!( - completed.is_ok(), - "publication {occurrence} timed out after {:?}; recruitment elapsed: {:?}; directory age: {}ms; missing readers: {:?}; pending stalled hints: {}", + received.is_ok(), + "publication {occurrence} timed out after {:?}: received {} of {} healthy hints; missing {:?}; held {:?}; recruitment elapsed: {:?}; directory age: {}ms; pending transports {}", started.elapsed(), + notified.len(), + healthy.len() - usize::from(held_session.is_some()), + healthy.difference(¬ified).filter(|session| Some(**session) != held_session).collect::>(), + held_session, recruitment_started.elapsed(), now_ms() - now, - healthy - .difference(¬ified) - .filter(|session| Some(**session) != held_session) - .collect::>(), pending.load(std::sync::atomic::Ordering::Relaxed), ); if let Some(held) = held.as_ref().filter(|_| occurrence == 1) { diff --git a/crates/cellule-app/tests/performance_fixture.rs b/crates/cellule-app/tests/performance_fixture.rs index 1ef98cbf..4a2409b2 100644 --- a/crates/cellule-app/tests/performance_fixture.rs +++ b/crates/cellule-app/tests/performance_fixture.rs @@ -570,13 +570,46 @@ impl PerfFixture { Ok(()) => {} // The simulated lost owner closes locally but cannot release // authority after its lease is fenced. - Err(Error::Fenced) if self.leases[index].check().is_err() => {} + Err(error) + if self.leases[index].check().is_err() && fenced_runtime_drain(&error) => + { + let stats = node.stats(); + assert_eq!(stats.active_cells(), 0); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.retained_bytes(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.file_descriptors(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); + } Err(error) => panic!("CellNode shutdown failed: {error}"), } } } } +fn fenced_runtime_drain(error: &Error) -> bool { + if matches!(error, Error::Fenced) { + return true; + } + if !matches!( + error, + Error::Facility { + name: "cell-runtime-drain", + .. + } + ) { + return false; + } + // Retained shutdown errors wrap the original runtime failure so repeated + // callers can inspect it. Only the exact terminal fencing error is expected + // after simulated owner loss; unrelated facility failures must still fail. + let mut cause: &(dyn std::error::Error + 'static) = error; + while let Some(source) = cause.source() { + cause = source; + } + matches!(cause.downcast_ref::(), Some(Error::Fenced)) +} + pub(super) fn node_session(node: usize) -> cellule_runtime::SessionId { cellule_runtime::SessionId::from_bytes([24 + node as u8; 16]) } diff --git a/crates/cellule-app/tests/process_follower.rs b/crates/cellule-app/tests/process_follower.rs index fad1fdda..1aed21ed 100644 --- a/crates/cellule-app/tests/process_follower.rs +++ b/crates/cellule-app/tests/process_follower.rs @@ -9,7 +9,6 @@ use cellule_runtime::follower::{FollowerReceipt, FollowerStore, FollowerTailPage use cellule_runtime::identity::NodeId; use cellule_runtime::node::durability::{NodeDurabilityConfig, NodeLogAuthority}; use cellule_runtime::node::lease::NodeLeaseGuard; -use cellule_runtime::node::log::NodeLogRotationBarrier; use cellule_runtime::node::log_transport::{ AppendRequest, NodeLogTransport, RetireRequest, SealRequest, TailRequest, }; @@ -727,7 +726,11 @@ impl NodeLogAuthority for ProcessEnrollment { }) } - fn close<'a>(&'a self, barrier: &'a NodeLogRotationBarrier) -> BoxFuture<'a, Result<()>> { + fn close<'a>( + &'a self, + retirement: &'a cellule_runtime::node::log::NodeLogRetirementObservation, + ) -> BoxFuture<'a, Result<()>> { + let barrier = retirement.barrier(); Box::pin(async move { let mut observed = self.observed.lock().await; if observed diff --git a/crates/cellule-host/Cargo.toml b/crates/cellule-host/Cargo.toml index 1c39c3f9..9ab6951a 100644 --- a/crates/cellule-host/Cargo.toml +++ b/crates/cellule-host/Cargo.toml @@ -8,6 +8,7 @@ rust-version.workspace = true description = "Provider-neutral Cell node lifecycle facade" [dependencies] +blake3.workspace = true cellule-app.workspace = true cellule-runtime.workspace = true futures-util.workspace = true @@ -17,5 +18,14 @@ tracing.workspace = true uuid.workspace = true [dev-dependencies] +bytes.workspace = true +cellule-store.workspace = true +ed25519-dalek.workspace = true +object_store.workspace = true +rusqlite.workspace = true tempfile.workspace = true -tokio = { workspace = true, features = ["macros", "rt-multi-thread"] } +tokio = { workspace = true, features = ["macros", "rt-multi-thread", "test-util"] } + +[[example]] +name = "fleet_operations" +path = "minion/main.rs" diff --git a/crates/cellule-host/docs/README.md b/crates/cellule-host/docs/README.md index 96779713..e04ec68f 100644 --- a/crates/cellule-host/docs/README.md +++ b/crates/cellule-host/docs/README.md @@ -4,6 +4,8 @@ | --- | --- | | [Lifecycle](lifecycle.md) | Start and stop one node safely. | | [Read replicas](read-replicas.md) | Install and supervise immutable views. | +| [Original failed-boot writers](original-writers.md) | Retain complete original ownership metadata before dependent effects. | +| [Fleet journal example](../minion/README.md) | Run the durable local journal foundation and inspect its integration limits. | | [Crate entry](../README.md) | Ownership and test command. | The host is a lifecycle facade. It does not add a second Cell scheduler, @@ -143,6 +145,7 @@ preserves the request; expiry or withdrawal releases it for replacement. | `read_replicas` | Selected immutable views, refresh, eviction, and close. | | `read_replicas/recruitment` | Scoped recruitment, bounded fanout, and replacement. | | `durability` | Node-log supervision and rotation. | +| `fleet` | Journal-bound finite actions, admitted receiver resources, exact source evidence, and checked activation. Applications own authentication, journal, and trusted Cell input lookup. | | `facility`, `tasks`, `status` | Owned facilities, bounded tasks, lifecycle reporting. | ```sh diff --git a/crates/cellule-host/docs/lifecycle.md b/crates/cellule-host/docs/lifecycle.md index 5d409694..5d1f0c0c 100644 --- a/crates/cellule-host/docs/lifecycle.md +++ b/crates/cellule-host/docs/lifecycle.md @@ -13,13 +13,17 @@ sequenceDiagram participant Service participant Node as CellNode + participant Drain as Owned host drain participant Runtime Service->>Node: Build with required components Node->>Runtime: Start and acquire lease Runtime-->>Node: Ready Service->>Node: Shutdown with deadline - Node->>Runtime: Stop admission and drain - Runtime-->>Node: Log closed, session withdrawn + Node->>Drain: Transfer the shared lane and deadline + Drain->>Runtime: Drain facilities, then close runtime + Runtime-->>Drain: Log closed + Drain->>Drain: Join lease maintenance and withdrawal + Drain-->>Node: Original result and task join ``` A deadline bounds releases started after acquiring the drain lane; it does not @@ -27,3 +31,939 @@ bound waiting for that lane. Fleet-level pacing stays in the movement planner. The node retains lease maintenance while accepted work and covered log tails are drained. See [the framework integration guide](../../../docs/framework.md) for service startup order. + +The host retains one runtime shutdown task and its original result. A deadline +cancels the join waiter; accepted runtime drain continues. A later drain joins +that task instead of calling runtime shutdown twice. Facility failures remain +inspectable, and the host reaches Stopped only after every required phase +succeeds. A retained runtime failure preserves its original error as a source +on every subsequent observation. + +The whole host drain also has one retained task slot. It runs the same reverse +facility order, runtime barrier and lease-maintenance phase. Cancelling a +shutdown caller leaves that task and its original callback future owned; the +host continues toward Stopped. The task keeps the shared lane guard, so queued +shutdown or scale-down callers cannot overlap it or repeat an accepted callback. +An idempotent stop still joins the original task epilogue before returning. + +For managed fleet boots, install `CellNode::install_fleet_boot_withdrawal` +after boot confirmation and before `start()`. Supply the original authenticated +directory version, Established boot record, and its enrollment journal. The +same retained drain task withdraws that exact boot after facilities, runtime, +and lease maintenance join. It checks the permanent canonical tombstone and +confirms durable retirement before exposing Stopped. A missing advertisement, +recovery-claim tombstone, failed facility, or ambiguous journal reply leaves the +node Draining. Retry retains the original binding and retirement evidence; +a late heartbeat can be reconciled only for the same signed boot identity. +The directory's `is_withdrawn` query requires no retained log or recovery claim; +the broader `is_retired` query only confirms a permanent fence. +This binding proves boot closure. Fleet relocation and complete reader/follower +settlement remain separate required barriers. + +A deadline keeps its existing phase timeout and ordinary task-abort policies. +After a returned incomplete attempt, the next caller first joins that original +attempt, then resumes the canonical sequence with its new deadline. The runtime +shutdown task is still invoked once. Genuine task-group failures remain errors; +a host task panic retains its original join failure and blocks a new attempt. + +`CellNode::drain_observation()` captures fixed-size local attempt diagnostics. +It retains the original first/latest failure objects after a successful retry. +The embedding application authenticates any route exposing this observation. + +| Local drain phase | Meaning | +| --- | --- | +| No observation | No retained local attempt was started; this proves no fleet absence. | +| Running | No original return or joined failure has been captured. This does not certify task health or progress. | +| Returned | The original sequence result is available; its task epilogue has not been joined. | +| Joined | The original task was joined; inspect its actual result and NodeState. | + +These diagnostics do not authorize maintenance completion. Relocation, +reader/follower policy, current authority and confirmed withdrawal still need +the fleet operation's complete proof. Native rotation handles stay weak; a +successful autonomous shutdown may clear them before a later observer arrives. +Capture and publish their canonical evidence before terminal shutdown. + +The task group retains each join and its original result within its existing +256-task bound. Concurrent drains share those joins; cancelling a caller leaves +the work owned. A failure cannot disappear after its handle is consumed, and +every sibling is joined even when another task fails. A repeated node drain +still reports that original source and keeps lease maintenance and Draining. + +| Task deadline | Retained outcome | +| --- | --- | +| Ordinary work | Abort the work task, retain its supervisor, then join its cancellation and destructor on a later drain. Cancellation remains an original error. | +| Node-log supervisor watcher | Retain the watcher and the existing native facility join. A later drain can succeed after canonical cleanup succeeds. | +| Lease maintenance | Node drain reaches this phase only after required work and runtime closure succeed. A work failure cannot become session withdrawal on retry. | + +Dropping the task group aborts its remaining owned tasks. Stopped still requires +successful joins through the canonical node drain lane. + +## Requested node-log rotation + +Install the existing durability provider during startup, then call +`CellNode::request_node_log_rotation(expected_epoch)` from an authorized +management adapter. The same supervisor wakes immediately and uses confirmed +old-member retirement; maintenance does not wait for the normal frame threshold. +The provider's `rotation_required(live_node_limit)` hook also requests automatic +rotation when its enrolled members expire, before the frame threshold. The +fleet-bound provider delegates this read to `FleetNodeDurabilityProvider`. +A failed membership read emits Failed and retries on a later tick while the +object-proof path remains available. After the read, the supervisor rechecks +its one bounded claim so an accepted maintenance request keeps strict retirement. + +Duplicate requests for that epoch share progress. An automatic best-effort +rotation already in flight is refused; observe the current binding before retry. + +| Inspection | Meaning | +| --- | --- | +| `NodeLogRotationRequest::observe()` | Local Queued, Retiring, Recruiting, Completed or Interrupted progress, with first/latest original errors. | +| `node_log_rotation_request(epoch)` | Recover a retained local request after a lost handle or reply. Missing is not evidence of completion or absence. | +| `fleet_durability_supervisor(now_ms)` | Capture the original supervisor lifecycle, separate supervisor/request-stop errors, and the existing pending/completed request bank, including automatic claims. | +| `retirement()` | Every original member confirmed its exact fence and canonical old-epoch authority close succeeded. | +| `completion()` | The retirement proof plus the newer epoch installed through expected-binding replacement. This does not establish redundancy policy or journal settlement. | + +Member failures remain retryable in strict mode; timer rotation cannot weaken an +accepted maintenance request. Existing Cells may continue through the ordinary +object-proof fallback while retirement or replacement enrollment is pending. +Applications own follower selection, exclusion of maintenance nodes, enrollment +barriers, authority reconciliation, deadlines, authorization and durable results. + +The node charges bounded request storage to its shared byte ledger. It retains +one pending request and one most-recent completed request; handles are weak and +cannot retain the node's resources after drain. Completed local history can be +evicted, so publish its canonical evidence to the fleet journal before finalizing. + +The host retains the single supervisor's join outside its cancellable task-group +watcher. Caller cancellation or a drain deadline drops only a join waiter. +Accepted retirement and recruitment replies still join; a recruited replacement +that returns during drain is canonically closed. The host remains Draining until +all required joins and cleanup succeed, retaining lease maintenance meanwhile. +Lost cleanup replies retry the same generation under the retained owner. +Complete member responses remain cached before the authority-close await. +Closure retries reuse those original fences and call only authority reconciliation; +they do not send retire RPCs after directory authorization has closed. +Foreign boot/node or non-advancing replacement configuration is rejected before +construction or any attempt to close that foreign scope. +Interrupted rotation never counts as +fleet settlement. Complete role observation, failed-owner recovery and journal +retirement remain required before physical-node finalization. + +Supervisor capture uses brief metadata locks and never waits on provider work +or the original join lane. Its installed owner reserves four KiB before work +starts; capture still works during drain after new runtime byte admission closes. +`NotStarted`, `Running`, `FinishedUnobserved` and `Returned` require the original +join. `JoinedUnsettled` retains a failed request stop; `Joined` can still carry +the original supervisor failure. Cancellation alone establishes neither join +nor cleanup. The bounded rotation bank preserves original proof/error Arcs; +an unavailable bank is an explicit error, never fabricated empty coverage. +An absent component supplies no coverage, and wrong-type/poisoned owners fail. +Combine this advisory local capture with producer pages, inbound native lanes, +fresh authority, replacement policy and authenticated revision-bound evidence +before settling fleet roles. + +### Verify a live owner's follower evacuation + +After the original requested rotation completes, call +`CellNode::follower_evacuation(original, operation, minimum_members, deadline)` +on that same live owner. Supply the original Established follower request, the +current Evacuating or Closing operation and the application's nonzero redundancy minimum +(one or two members). This read-only check returns a blocker while recruitment +is incomplete. It never requests another rotation or closes a lane. + +| Returned evidence | Required checks | +| --- | --- | +| Original and Retired rows | Same request, acceptance time and establishment evidence; every old ensemble member is durably Retired. | +| Original rotation Arc | Confirmed object-covered retirement and the newer binding installed by the existing supervisor; retry leaves original errors inspectable. | +| Replacement enrollment | Complete original signed source/member boots and canonical epoch; the donor is excluded and the member minimum is met. | +| Replacement rows and current authority | Every member is Established on an Active managed boot at its exact intent revision; signed boot identities and source authority are rechecked around full journal confirmation. | +| Capture interval and snapshot | At most thirty seconds, bounded by the operation and caller deadlines; original journal timestamps are never refreshed. | + +The returned `FollowerEvacuation` retains its node metadata charge. Publish and +revalidate this per-owner interval before local history can be evicted. A +withdrawn or changed replacement, missing original completion, changed registry, +expired operation or incomplete enrollment produces no settlement evidence. +Complete foreign/native inventory, Pending producers, failed-owner recovery and +terminal action handoff remain required before physical-node shutdown. + +### Persist and revalidate follower replacement evidence + +`FollowerReplacementPolicy` gives the application's nonzero member minimum a +durable revision in the existing fleet journal. `FleetFollowerEvacuationJournal` +loads it at the complete head/registry barrier. Policy changes compare that +barrier and live controller, advance the policy revision exactly once and bump +the shared registry. Absence blocks publication; it cannot silently reduce +redundancy. Applications authorize these changes and make their ordinary +recruitment satisfy the same policy. + +`FollowerEvacuation::durable_record` preserves every original Retired member, +the original Established donor request identity, object-covered watermark, +signed source/member boot identities, complete replacement requests and the +capture interval. The journal atomically commits the bounded manifest and its +latest operation/request pointer. Replay returns original history without +restoring a superseded pointer or refreshing timestamps. + +`FleetSnapshotTransport::capture` routes exact read-only requests to +`CellNode::fleet_snapshot` through its existing finite task owner. Authenticate +the physical endpoint and boot, preserve original request/response/error values, +and return fresh captures. The transport owns no second runtime or role executor. +`FleetFollowerEvacuationVerifier` combines canonical directory/policy reads with +complete source/member native traversals and all-category rechecks. A replacement +epoch can be enrolled and installed before its first append; its native producer, +supervisor and runtime binding prove installation even while the canonical log +is inactive. An inactive advertisement alone grants no such evidence. + +```rust,no_run +use cellule_host::{FollowerEvacuation, fleet::{ + FleetFollowerEvacuationJournal, FleetFollowerEvacuationPublication, + FleetFollowerEvacuationVerifier, FleetSnapshotTransport, +}}; +use cellule_runtime::{Error, Result, node::NodeDirectory}; +use std::sync::Arc; +use tokio::time::Instant; + +async fn persist_follower_replacement( + capture: &FollowerEvacuation, + journal: &dyn FleetFollowerEvacuationJournal, + directory: NodeDirectory, + transport: Arc, + deadline: Instant, + clock: impl FnMut() -> Result, +) -> Result { + let policy = journal.follower_replacement_policy(capture.snapshot()).await + .map_err(|source| Error::Facility { name: "fleet-policy", source })? + .ok_or(Error::Fenced)?; + let verifier = FleetFollowerEvacuationVerifier::new(directory, transport); + FleetFollowerEvacuationPublication::publish( + capture, policy, journal, &verifier, deadline, clock, + ).await +} +``` + +After reconstruction, load immutable history and call `recheck`. A committed +record and a failed final confirmation remain separately inspectable, with +original shared source errors. `refresh` builds new metadata from that committed +original retirement when policy, current ensemble, or operation deadline/session +changes. Ordinary rotation supplies the new ensemble. The original local rotation +receipt may already be evicted; refresh starts no rotation or retirement. Publish +the opaque candidate through `publish_refreshed`. Both Evacuating and Closing +allow this repair before terminal finalization. + +Applications account bounded copied manifest/collector buffers and join accepted +backend/native work. Fresh checks use monotonic thirty-second intervals and caller, +controller and operation deadlines. This supplies live-owner replacement evidence. +A failed original leader requires canonical recovery and affected-Cell successor +evidence; complete fleet role observation, process providers and terminal action +handoff remain required. See the [reference evidence cases](../minion/README.md#durable-follower-replacement-evidence). + +### Publish a recovered owner's follower retirement + +First finish canonical recovery pinning, every original member's native +retirement and the Retired tombstone CAS through the +[runtime recovery path](../../cellule-runtime/docs/failover-and-followers.md#contents). +Then capture `FleetRecoveredFollowerRetirement` against the complete current +`FleetRoster`. It binds the canonical physical leader, session, epoch, full +ensemble and manifest to every original member request. Missing or duplicate +requests, foreign scope, unretired authority and changed journal barriers refuse +capture/publication. Pending establishment replies remain original obligations. + +```rust +use cellule_host::fleet::{ + FleetJournal, FleetRecoveredFollowerPublication, + FleetRecoveredFollowerRetirement, FleetRoster, +}; +use cellule_runtime::{ + fleet::operations::FleetScope, + identity::SessionId, + node::{NodeDirectory, SealedNodeLog}, +}; +use tokio::time::Instant; + +pub async fn publish_recovered_followers( + journal: &dyn FleetJournal, + directory: &NodeDirectory, + scope: FleetScope, + sealed: &SealedNodeLog, + authenticated_claimant: SessionId, + deadline: Instant, + mut clock: impl FnMut() -> cellule_runtime::Result, +) -> Result> { + let snapshot = journal.load_snapshot(scope).await?; + let roster = FleetRoster::collect(journal, &snapshot, deadline).await?; + let retirement = FleetRecoveredFollowerRetirement::capture( + journal, directory, &roster, sealed, + authenticated_claimant, deadline, &mut clock, + ).await?; + Ok(retirement.publish( + journal, directory, authenticated_claimant, deadline, &mut clock, + ).await?) +} +``` + +Own the publication future in accepted finite work and account the copied +collector/record buffers. The API starts no background worker or native retire +RPC. The caller deadline bounds journal waiters; the adapter retains and joins +accepted backend work after deadline or cancellation. `members()` preserves all +returned responses and their separate original errors; `confirmed()` +requires successful replies plus a complete post-publication roster and fresh +canonical authority. Clock regression, stale barriers or an incomplete final +scan preserve successful writes without granting closure. + +After a lost reply or controller restart, collect a fresh roster and recapture +the exact canonical epoch. Deterministic events preserve each original spec, +acceptance, establishment history and retirement time. Replay still works after +native grace collection; it adopts canonical history without reconstructing +missing member receipts. The closure covers these follower enrollment rows. +Failed-boot/process closure, replacement policy, affected writers and terminal +host shutdown remain required before maintenance completion. + +### Publish an original failed receiver's reader closure + +`FleetFailedReaderRetirement` settles one original reader enrollment after +its exact receiver boot is permanently fenced and independently proven unable +to execute again. Capture accepts Pending, Established and exact replayed +Retired requests at a complete bootstrapped roster. The reader's receiver +node/session must match the original Established boot; a failed source writer +cannot authorize closure of a reader on a live receiver. + +`FleetFailedBootProcessRequest::capture` can collect the original boot/fence +request while roles remain unresolved. It permits only read-only process +confirmation. This breaks the dependency between obtaining durable process +evidence and settling the roles that must precede boot retirement. Its digest +excludes mutable enrollment status and collection times, so the same original +process witness survives reader/follower publication and fresh recapture. + +Use the application-owned `FleetFailedBootProcesses` provider described below. +It must join the original process and all accepted native/external work and +producers, authenticate that exact lifetime, and prevent session reuse. Neither +canonical withdrawal nor an expired advertisement supplies process evidence. + +```rust +async fn publish_failed_reader( + journal: &dyn cellule_host::fleet::FleetJournal, + directory: &cellule_runtime::node::NodeDirectory, + processes: &dyn cellule_host::fleet::FleetFailedBootProcesses, + original_boot: &cellule_runtime::fleet::operations::EnrollmentRecord, + original_reader: &cellule_runtime::fleet::operations::EnrollmentRecord, + claimant: cellule_runtime::identity::SessionId, + deadline: tokio::time::Instant, + mut clock: impl FnMut() -> cellule_runtime::Result, +) -> cellule_runtime::Result { + let snapshot = journal.load_snapshot(original_reader.spec().scope).await + .map_err(|source| cellule_runtime::Error::Facility { + name: "example-failed-reader-journal", source, + })?; + let roster = cellule_host::fleet::FleetRoster::collect(journal, &snapshot, deadline).await?; + let retirement = cellule_host::fleet::FleetFailedReaderRetirement::capture( + journal, directory, &roster, original_boot, original_reader, + claimant, deadline, &mut clock, + ).await?; + retirement.publish(journal, directory, processes, claimant, deadline, clock).await +} +``` + +Publication rechecks the original complete barrier after provider confirmation, +then uses the existing enrollment journal. Inspect `record()` even when +`confirmed()` fails: publication replies and final checks retain separate +original errors and durable partial state. Fresh recapture adopts lost replies +without refreshing original acceptance, establishment or retirement times. +The final roster, original history, canonical receiver fence and unchanged +process witness must confirm within the original thirty-second interval. +Backend owners retain and join accepted work after cancellation or deadline. + +This closes one original reader lifetime. Replacement-policy coverage, other +roles, affected writers, boot retirement and physical maintenance finalization +remain separate requirements. For a failed source with a live receiver, use +that receiver's ordinary joined reader lifecycle. The +[native example cases](../minion/README.md#failed-reader-lifetime-evidence) +exercise actual SQLite readers, joined host shutdown and independent journal +reconstruction; OS crash and external-job/provider campaigns remain required. + +### Publish an original failed boot's closure + +Before recovery can change Cell ownership, obtain original process evidence from +`FleetFailedBootProcessRequest::capture_fenced` and `confirm`. The directory's +read-only `fenced_session` supplies the permanent exact physical/session fence +while the leader log can still be Recovering. It does not prove termination, +native/external work joining, log retirement or takeover readiness. + +The fenced request uses a new v2 identity bound to the full original boot, +acceptance/establishment and immutable fence times. Recovery phase/manifest, +claimant, registry and capture times are excluded. It can be recaptured after +recovery and controller/adapter reconstruction with the same original witness. +`confirm` checks the current complete roster, original boot, permanent authority +and stable provider evidence twice in a monotonic thirty-second interval. It +starts no termination, native or recovery effect. Applications still authenticate +and durably join the actual original process and every accepted job first. + +```rust +async fn confirm_original_process( + journal: &dyn cellule_host::fleet::FleetJournal, + directory: &cellule_runtime::node::NodeDirectory, + processes: &dyn cellule_host::fleet::FleetFailedBootProcesses, + original: &cellule_runtime::fleet::operations::EnrollmentRecord, + claimant: cellule_runtime::identity::SessionId, + deadline: tokio::time::Instant, + mut clock: impl FnMut() -> cellule_runtime::Result, +) -> cellule_runtime::Result<( + cellule_host::fleet::FleetFailedBootProcessRequest, + cellule_host::fleet::FleetFailedBootProcessConfirmation, +)> { + let snapshot = journal.load_snapshot(original.spec().scope).await.map_err(|source| { + cellule_runtime::Error::Facility { name: "application-process-journal", source } + })?; + let roster = cellule_host::fleet::FleetRoster::collect(journal, &snapshot, deadline).await?; + let request = cellule_host::fleet::FleetFailedBootProcessRequest::capture_fenced( + journal, directory, &roster, original, claimant, deadline, &mut clock, + ).await?; + let confirmed = request.confirm( + journal, directory, processes, claimant, deadline, clock, + ).await?; + Ok((request, confirmed)) +} +``` + +Retain the complete original Cell set before overlay publication, log sealing or +takeover. A process confirmation permits this collection; it is not the collection +or successor evidence. Full catalog/native inventory and durable operation-bound +retention remain required. An active Recovering log blocks independent takeover; +already Sealed/inactive logs require an earlier retained writer set or verified +complete original ownership history. The canonical authority path now retains +full owner observations before departure, including unpublished and object-covered +controls. Its [per-Cell history read](../../cellule-runtime/docs/storage.md#retain-original-owners-before-departure) +refuses missing legacy/restore history. Authenticated complete catalog traversal, +original process joining, durable operation binding and successor verification +remain required; history alone cannot finalize a node. Re-reading only current +failed-owner controls cannot reconstruct Cells that already moved. + +After recovery and all related roles close, use +`FleetFailedBootRetirement::capture_retained` with the original process request +and a freshly collected roster. It preserves the original request interval and +identity, and requires fresh terminal-log and physical-reference checks. Publication +uses the same provider and enrollment journal. Process confirmation cannot bypass +any boot-retirement barrier. The existing terminal `capture` retains its v1 +digest and event identity. `request.fence()` is always available; +`request.canonical()` now returns an optional original terminal-log observation. +It is absent for fenced captures even after their original log later retires. +After restart, recapture a fenced request with `capture_fenced` and use +`capture_retained` again. Preserve the basis chosen for the original publication: +terminal v1 and fenced v2 requests have distinct identities. Changing basis or +witness cannot adopt an already committed retirement. The provider durably retains +that original request/witness binding; mixed-binary/provider rollout qualification +remains required. + +After recovered follower publication, `FleetFailedBootRetirement` closes the +original boot row through the same enrollment journal. It requires a complete +bootstrapped roster, the exact Established original request and all its related +reader/follower requests settled. A fresh `NodeDirectory::closed_session` checks +the physical node/session's permanent fence and Retired leader log. Sealed or +inactive unretired logs, missing records and expired advertisements refuse. +Complete physical follower discovery also rejects a live reference belonging to +that original boot; registered obligations of a different boot remain separate. + +The application implements `FleetFailedBootProcesses`. Its read-only +`confirm_stopped` checks durable evidence that the original process and its +accepted external jobs/producers have joined and that the same session cannot +execute again. Authentication, provider termination and evidence persistence +belong to the application. A witness digest identifies that evidence; its public +constructor checks shape and supplies no authentication or process observation. +Do not create a witness from a reusable PID, expiry, missing inventory, timeout, +recovery result or replacement process. Start and own termination through the +application's supervised finite work before requesting confirmation. + +```rust +async fn publish_failed_boot( + journal: &dyn cellule_host::fleet::FleetJournal, + directory: &cellule_runtime::node::NodeDirectory, + processes: &dyn cellule_host::fleet::FleetFailedBootProcesses, + original: &cellule_runtime::fleet::operations::EnrollmentRecord, + claimant: cellule_runtime::identity::SessionId, + deadline: tokio::time::Instant, + mut clock: impl FnMut() -> cellule_runtime::Result, +) -> cellule_runtime::Result { + let snapshot = journal.load_snapshot(original.spec().scope).await.map_err(|source| { + cellule_runtime::Error::Facility { name: "application-failed-boot-journal", source } + })?; + let roster = cellule_host::fleet::FleetRoster::collect(journal, &snapshot, deadline).await?; + let retirement = cellule_host::fleet::FleetFailedBootRetirement::capture( + journal, directory, &roster, original, claimant, deadline, &mut clock, + ).await?; + retirement.publish(journal, directory, processes, claimant, deadline, clock).await +} +``` + +Inspect `record()` even if `confirmed()` fails: a lost reply or final check can +leave an actual committed retirement. The original source error remains shared +and inspectable. Fresh recapture and the same durable process witness adopt the +original request, acceptance, establishment and retirement time after controller +or adapter restart. A changed witness conflicts with committed retirement. +The final complete roster, exact returned row, canonical authority, physical +references and process proof must revalidate within the original thirty-second +interval. Accepted backend jobs remain with the journal/provider owner after a +waiter cancellation or deadline. + +This closure settles that boot's enrollment. It does not convert a recovered +tombstone into planned withdrawal or prove replacement policy, affected-writer +relocation, operation completion or permission to stop the physical node. +`SettleRoles`/`Finalize` still require those additional barriers and the existing +native shutdown handoff. The [focused example cases](../minion/README.md#failed-boot-process-evidence) +exercise a joined child lifetime and independently reconstructed evidence; +they do not qualify a complete multi-process fleet or provider deployment. + +## Journal bound fleet actions + +### Fleet boot admission + +For a fleet-managed host, load its retained `NodeIntent` and pass it to +`CellNodeBuilder::with_fleet_startup_intent` before building. The leased runtime +holds its shared writer, reader and follower gate before the application can +install a lease. An Active row must name this boot; a maintenance predecessor +can build with admission held until the journal validates its successor. +The offline unleased maintenance builder rejects this configuration. + +| Startup boundary | Required behavior | +| --- | --- | +| Build | Validate the retained intent, hold new roles, and install its sticky Cordoned/Draining mode. Missing or ambiguous rows fail before build. | +| Enroll boot | Journal Pending before canonical directory creation. Only New permits first execution; an Existing request requires canonical inspection and its exact original evidence. | +| Confirm | Call `confirm_fleet_startup(journal, enrollment_key)` after Established publication and lease installation. `FleetEnrollmentJournal::load_boot` reads the exact established boot and current intent in one transaction. Missing, pending, retired or foreign evidence cannot open admission. | +| Bind closure | Before readiness, call `install_fleet_boot_withdrawal(directory, original_version, established_record, journal)`. The binding is immutable and requires the exact confirmed boot. The native closing owner checks withdrawal and durable retirement before Stopped. | +| Start | Install required facilities/task supervision and finish probes, then call `start()`. Confirmation alone keeps the hold. Start checks the owned components and opens Active admission, or enters `NodeState::Maintenance` under retained cordon/drain. | +| Route requests | Serving probes use `is_ready()`. Authorized management uses `is_management_ready()`, which includes Maintenance. Fleet action/inspection endpoints remain available there; new roles stay closed. | +| Refresh intent | Call `refresh_fleet_intent(journal, deadline)` from the supervised membership/lease loop before renewal. It rechecks the original Established boot and current physical intent atomically, then applies sticky cordon/drain to the shared gate. A failed or expired read grants no renewal authority. | +| Close | The bound native drain joins accepted work, runtime and lease maintenance before withdrawing the boot and retiring its registry obligation. Only checked ordinary withdrawal and its permanent session tombstone settle a lost reply; absence, expiry or a recovery claimant cannot. | + +Confirmation is a cancellable read. Concurrent confirmations cannot overwrite a +newer checked intent with an older reply; a read that finishes after shutdown +cannot reopen the node. Local startup checks do not replace an application's +atomic enrollment policy, authentication or complete registry import. Startup +retains the original enrollment record, including its acceptance time and evidence; +another Established request cannot replace it. Intent refresh rejects older or +contradictory replies and rechecks lifecycle after the read. A delayed reply cannot +reopen shutdown, and an Active reply cannot clear a local cordon. Cancellation +drops only the read waiter; retry against the original boot. Existing owners keep +serving while live intent closes new roles. The application owns supervision, +polling cadence, error handling and the lease renewal policy; this API starts no task. +Authorized Cordon actions use the same gate and remain replayable after refresh. + +The [reference example](../minion/README.md) wires this order +to actual signed directory enrollment and one durable SQLite transaction domain. +Managed reader and follower bindings provide the local producer barriers below. +Complete fleet observation remains integration work. + +### Managed follower enrollment + +Install `install_fleet_node_durability_provider` before readiness, using the +shared `FleetJournal` and an application-owned `FleetNodeDurabilityProvider`. +Preparation returns `FleetNodeLogRecruitment`: an opaque exact directory attempt, +authenticated transport, authority, lease and telemetry. It performs no enrollment +CAS or append. Configured fleet startup rejects an ordinary unmanaged durability +provider and a manually installed runtime epoch without this typed binding. +Recruitment waits for the startup hold to clear; confirmed maintenance mode can +still replace outbound follower obligations while new inbound roles stay closed. + +| Boundary | Managed behavior | +| --- | --- | +| Prepare | Validate canonical shipper limits; capture exact source/follower boots and fresh endpoint intent revisions. Reserve one MiB from the shared node ledger before accepting any request. | +| Accept | Retain every original request before awaits. Every selected member must return a validated New Pending acceptance before the single original-token directory CAS. | +| Unknown enrollment | Inspect only the retained attempt. An original-token conditional refusal may exclude its delayed CAS. Missing, expired or newer empty records remain unknown; no reselection or CAS rebasing is allowed. | +| No native dispatch | Atomically refuse every original member, including members whose acceptance reply was lost or acceptance never started. An absent key becomes an exclusion tombstone. Delayed acceptance returns Existing with that original terminal row. | +| Establish | Replay the original checked enrollment events and acceptance times until all publications confirm. No shipper/configuration is delivered before that barrier. | +| Retire | Join object coverage and every exact native member fence, retain the complete observation, reconcile canonical authority close, then durably retire every original member. Lost publication never repeats a confirmed native close. | +| Drain | Join the existing supervisor before settling its undelivered attempt, then let runtime drain close delivered epochs. Deadlines cancel waiters; owned cleanup and byte charges remain until journal settlement. | + +There is one pending preparation and at most 32 inventoried epochs. Retirement +removes local inventory and releases its reservation only after durable settlement. +`follower_enrollment_completion(epoch)` exposes immutable inputs, native proofs, +all member responses, publication state, and separate original native/journal +errors without waiting on RPCs. Missing local history supplies no fleet proof. + +`fleet_follower_enrollments_page(cursor, limit, now_ms)` enumerates every retained +epoch, including original member requests whose acceptance is unknown. A page +reserves one MiB before copying fixed-size follower metadata, admits 1–32 epochs, +and hashes signed inputs/proofs in place. It preserves original retirement/error +Arcs without deep-copying signed ensembles or provider CAS tokens. Drop the page +to release its charge. A process-local continuation binds all epochs, native and +publication progress, pending attempt, shared mode, and protocol state. Changes +or missing continuation keys require restarting from the first page. + +| Capture | Interpretation | +| --- | --- | +| Busy protocol, no epochs | Preparation may be awaiting provider/journal I/O before a request exists; zero rows supply no absence proof. | +| Pending epoch | Preserve all original selected members, including missing acceptance replies; capture starts no CAS or cleanup. | +| Enrollment/refusal digest | Identifies an original checked proof; it supplies no authentication, current authority or replacement-policy proof. | +| Retirement observation | Keeps original successful and failed member replies before separate canonical closure and durable publication. | +| Missing or failed producer | An unbound node returns `None`; wrong component type, failed lock, exhausted or closed runtime byte ledger returns an error. Neither supplies empty coverage. | + +An idle protocol and zero retained epochs do not prove a joined supervisor, +empty inbound follower lanes, withdrawal, or safe finalization. This is an +interval scan. Applications still collect authenticated request-bound envelopes, +the complete durable roster, fresh directory authority, native lanes and readers, +and supervisor/facility completion. `try_owned_component` preserves typed lookup +failures for these collectors; optional legacy lookups still return `None`. + +The provider allocates a never-reused advancing epoch for each leader boot. Its +transport must address the originally selected signed follower boots. Its authority +serializes heartbeat/activation/coverage/closure, fresh-loads the exact source +epoch after enrollment, and reconciles ambiguous close from the original checked +close receipt. Arbitrary absence cannot prove closure. Applications retain durable +backend, authentication, replacement policy and failed-process reconciliation. +Complete role observation and these remaining proofs precede fleet finalization. + +Install `CellNode::install_fleet_actions` during startup with a fleet/application +scope, stable physical NodeId, an application-owned `FleetActionJournal`, and +trusted `FleetCellProvider` inputs. The provider resolves catalog, replica, +canonical authority, private local destination, and this node's leased owner. +Its lookup performs no hydration or takeover; remote callers supply no local +paths or credentials. +For failed-source recovery, `recovery_inputs` supplies an existing canonical +`NodeTakeoverProof` and recovery manifest store. Ordinary node recovery first +fences the failed session and seals/pins its required tail; the provider lookup +performs none of those effects. +The application authenticates management requests. The journal atomically +checks the current controller epoch, head, permit, intent, endpoint, and +deadline when first accepting an action. `AcceptedFleetAction` validates its +shape and exact inputs; its Rust type supplies no remote authorization. + +`apply_fleet_action` executes settled source Release with +`release_idle_cell_at`, actual receiver preparation, canonical activation, +resource cleanup and failed-source recovery. Maintenance Cordon closes the +same new-writer, reader, and follower admission gate after exact boot-bound +journal acceptance. Existing owners keep serving until their movement barrier. +Fresh inspection uses its separate request-bound method. The caller-driven +reconciler supports cordon and settled movement; complete observation wiring, +complete primitive maintenance coverage and role finalization remain under implementation in the +[fleet operations plan](../../../docs/fleet-operations-plan.md). + +| Event | Action executor behavior | +| --- | --- | +| First acceptance | Commit exact acceptance before its effect. Movement Release rechecks deadline and invokes the canonical actor release for the exact session, generation, incarnation and epoch. | +| Source preflight refusal | The actor returns `Error::CellReleaseRefused` only before this request begins canonical deactivation. Record Rejected with its original source error; the reconciler then cancels and joins unused receiver credit. Independent local eviction grants no release proof for this attempt. | +| Prepare receiver | Validate catalog/Cell/incarnation and current release/schema support; reserve actual runtime/LTX resources before returning Reserved. A refusal preserves the original error and leaves the source serving. | +| Maintenance Cordon | Apply retained Draining intent through the shared admission gate. Replays retain the exact acceptance/result; a dropped waiter leaves publication owned. This preserves existing obligations and uses no shutdown lane. | +| Other maintenance effects | Refuse role settlement, finalization and maintenance inspection until their inventory and host barriers are implemented. A cordon receipt cannot establish Stopped or withdrawal. | +| Activate receiver | Confirm the exact Idle acquisition basis is durably retained before ordinary ownership CAS; consume prepared credit through canonical restore and actor activation. | +| Acquisition-basis reply lost | Keep authority untouched and the prepared credit charged. Repeating the same accepted action confirms the original basis and capture time before takeover. | +| Unused credit after release | Journal CleaningReceiver, cancel/join that credit, and retain the source position and fleet permit. Cleanup returns to Released, with no claim of pre-release cancellation. After confirmed cleanup, ordinary admitted acquisition may resume. | +| Ordinary acquisition on the preferred session | Free the unused preparation without closing the ordinary writer. Verify current authority, actor readiness, and the required release position. Matching session alone never proves prepared credit was consumed. | +| Recover unresolved source release | Journal Recovering, cancel proved-unused preparation, validate canonical failed-session proof, and confirm `RecoveryBasis` before ownership CAS. Confirm `RecoveryEvidence` after exact recovery publication and before actor admission. Return Recovered only with current actor/authority proof. | +| Recovery input reply lost | No ownership CAS starts. Retain the original input/time; an unchanged full canonical control permits confirming that basis and continuing the accepted action. | +| Recovery position reply lost | No actor is admitted. Canonical rollback preserves the materialized root; the attempt stays charged and Unknown. Ordinary acquisition can restore serving, after which inspection verifies it against the retained recovery evidence. | +| Duplicate | Compare the full immutable specification, including cost and physical identities. Join current work or return its committed result. | +| Dropped caller | Retain and finish execution plus journal publication independently of the caller. | +| Result publication failure | Retain the original checked result. Result delivery joins the owned task, so an immediate subsequent dispatch can retry publication. Drain also retries; neither repeats source release. | +| Existing acceptance without a result | Record Unknown and retain the fleet permit; never infer the original release root from current authority. | +| Node drain | Stop new action admission and join retained finite work before runtime shutdown. Keep task handles and checked results when a deadline cancels the drain waiter. | +| Action task fails | Join every accepted sibling job, settle resources, and retain the original join error across repeated drain calls. Runtime cleanup still runs; consuming a task handle cannot make a later shutdown report success. | + +`FleetActionCompletion::committed` is true only after confirmed durable result +publication. Preserve its separate execution and journal errors in correlated +diagnostics. A committed Released result proves source release; destination +serving still requires fresh authority and actor evidence. A replay returns a +retained observation; it does not refresh that proof. The executor checks a +new serving observation through the normal FIFO query/lease boundary and +compares actor inventory with current authority. The receiver's local receipt +retires only after confirmed result publication and accepted-task completion. +Later inspection uses retained journal evidence after local retirement. + +Clean release and recovery remain separate in status and history. A source-epoch +Idle root after a lost release reply can be a recovery input only with canonical +failed-session proof; a later root alone cannot manufacture Released. Recovery +uses `AcquisitionObserver` recording points through the same ordinary acquisition, +exact-root restoration, and rollback path. The journal atomically binds basis +and evidence to original acceptance, rejects changed inputs, and returns the +original time for identical writes. + +The executor uses at most two retained action jobs and charges their envelopes, +results, and acquisition inputs to the existing node retained-byte ledger. +Unknown work retains its fleet permit in the journal. A missing local receipt, +caller timeout, or expired reservation alone proves neither resource settlement +nor successful movement. These local gates do not establish complete fleet +maintenance or process/provider qualification. + +## Fresh fleet inspection + +Use `CellNode::inspect_fleet_action` for current evidence. A retained +Inspect reply is historical and cannot establish current serving. +`apply_fleet_action` refuses raw Inspect dispatch; use the request-bound method. +Effects and observations use the same finite task owner, shared action bound and +runtime ledger; shutdown joins accepted inspections after dropped waiters. + +| Boundary | Required behavior | +| --- | --- | +| Request | Build `FleetInspectionRequest` from the current head's Inspect action and exact `RegistryVersion`. Name the physical node/boot, a new nonce for this pass and an exclusive capture deadline. | +| Authorization | `FleetActionJournal::authorize_inspection` checks current head, registry revision and endpoint intent in one read transaction. Cached acceptance/result records cannot satisfy it. | +| Capture | Inspect actual actor/authority state. Recovery inspection cannot start acquisition; missing recovery position remains Unknown. Retained release and resource-settlement proofs remain historical facts. | +| Response | `FleetInspectionObservation` retains the original capture start/end and full request. `validate_for` rejects mismatched nonce, endpoint, payload, registry or authorization, future time, expired deadline and an over-age capture interval. | +| Dependent decision | Authenticate the reply, check its required position, and compare the head/registry versions again when committing the reducer transition. A failed inspection preserves the charged attempt. | + +The observation codecs use new record kinds 17 and 18. Existing effect and +result encodings remain unchanged. Deploy readers before these producers; +decoding a record does not authenticate its origin or prove a complete roster. + +## Request bound native fleet pages + +`CellNode::fleet_snapshot` captures one native category through the same two-job +bank as effects and inspection. `FleetSnapshotRequest` pins the complete journal +head/registry, physical node and boot, nonce, category, native continuation, +limit, issue time and exclusive deadline. The interval is at most 30 seconds +and ends within the original controller lease. Applications authenticate callers. + +`FleetActionJournal::authorize_snapshot` must compare the full current barrier +and endpoint intent in one transaction, before and after the native read. The +reference SQLite journal implements that transaction. Changed revisions or +boots fail; retries cannot restamp the original request or interval. Waiter +cancellation leaves accepted capture owned until its original join, including +blocking follower inventory reads. Shutdown joins that work. + +| Subject | Original native source | +| --- | --- | +| Host | Lifecycle, installed owner bindings, mode and local node-log identity. | +| Cells | Generation-bound live and transitioning actor inventory. | +| Readers / ReaderEnrollments | Managed native views and original producer requests/jobs. | +| FollowerLanes / FollowerEnrollments | Persisted inbound lanes and original managed leader enrollment progress. | +| DurabilitySupervisor | Retained task lifecycle, original failures and bounded rotation bank. | + +The response retains each native page's allocation token and original errors. +Drop it to release its page charge. `FleetNodeSnapshot::validate` checks the +full request, capture interval and available native scope/boot identities. +Missing owners return `Unbound`; missing or failed capture supplies no role +coverage. Local bindings, counts and log identity remain advisory. Complete +observation still requires stable traversal, durable roster confirmation, +current remote authority and replacement policy. This in-process API defines +no persisted or wire format; applications own authenticated transport encoding. + +### Full native traversal and revalidation + +`FleetNodeInventoryScan::new(&roster, node, session)` pins a current Established +managed boot in a bootstrapped full roster. Serve each `next_subject()` through +the authenticated snapshot transport and pass the exact request and original +response to `accept()`. Drop the response after acceptance to release its native +page charge. A failed acceptance poisons that traversal. `finish()` requires +every continuation and a fresh first-page recheck of all seven categories. +Applications account the collector's bounded copied buffers. + +| Retained input | Required interpretation | +| --- | --- | +| Writer rows and transitioning Cell IDs | Transitions remain obligations; actor maps can overlap. They never imply absent ownership or complete signed counts. | +| Reader views and producer records/jobs | Preserve original requests, accepted timestamps, publication state and shared errors. Preparation with no Pending row remains visible through job state. | +| Persisted lanes and follower producer state | Include cold/retired lanes, quarantined files and unknown preparation. An idle producer does not prove a joined supervisor. | +| Supervisor observation | Preserve the original lifecycle, rotation barriers, completion and failures. | +| Missing bindings | Retain None/Unbound identity. Absence requires the application's bootstrapped closed composition proof. | + +`inventory.validate_enrollments(&roster)` matches the retained native inputs to +the same full roster. A mismatch disables complete coverage; callers may retain +independently fresh writers for pressure relief. After scanning **all** nodes, +unexpected directory records, exact Cell/log authority and replacement policy, +use `inventory.recheck()` to fetch seven fresh first pages. These pages fingerprint +the entire category, including rows outside the page. Finish every recheck and +then reconfirm the full journal barrier. Revalidation keeps the original start +clock, rejects nonce reuse across rounds, and never refreshes the original rows. +Matching fingerprints supply interval evidence, not an atomic fleet snapshot. +Failed-process closure, policy satisfaction and the finalization transaction +remain separate requirements; these collectors grant no shutdown permission. + +`FleetFollowerReferences::collect(directory, roster, member, page_limit, deadline, +clock)` traverses all authoritative log references for a physical follower, +including expired advertisements and fenced tombstones. It preserves exact +leader/session, epoch, phase, ensemble and coverage. Its enrollment check matches +retained original requests; Pending requests without a current reference still +require their original nonexecution or closure evidence. Applications account +the bounded copied buffer, with at most 10,000 rows and 128 rows per page. + +After native and policy collection, `references.recheck(...)` traverses every +page again and compares exact rows. Native topology fingerprints intentionally +omit volatile coverage and leader liveness, so a first-page fingerprint cannot +replace this authority recheck. A changed row, deadline, missing page or changed +roster returns an error and preserves the original interval. Reconfirm the full +roster after both authority and native rechecks. A closed local lane can still +have a foreign authority reference; zero local writers do not settle that tail. + +`FleetRoleCoverage::check(roster, native, foreign, now_ms)` matches the combined +graph. Supply every required original boot and every retained physical node's +reference scan. Complete all initial captures, all seven-category native +rechecks, then every exact foreign recheck, in that order; reconfirm the full +journal after the check. Missing or duplicate inputs, stale intervals and +incomplete rechecks are refused. Starting another native or foreign recheck +invalidates its earlier confirmation, including cancellation and failure. + +A follower can be Established before its first append creates a persisted lane. +The combined check observes that obligation through the exact delivered managed +source producer, installed binding, original member rows and current Open +authority on every ensemble member. Local `validate_enrollments` remains strict +without this cross-node witness. Failed or unobserved owners remain blockers. +The returned digest binds the original roster, category fingerprints, exact +foreign rows and collection/recheck times; it is an in-process input identifier. +Authentication, unexpected advertisements, current Cell authority, replacement +policy and failed-process closure remain separate adapter duties. Pending rows +remain obligations. Coverage grants neither role settlement nor permission to +stop a node. + +Attach the original graph with `observation.with_role_coverage(coverage)`. +`FleetObservation` retains it and includes its digest in canonical planner +inputs. The graph's collection/recheck interval must fit inside the original +observation interval; attachment rejects a different scope or registry and +cannot replace an already retained graph. The reconciler compares its exact +head, registry and full roster digest again after confirming the journal. +A controller renewal changing only the head invalidates an earlier graph even +when the registry is unchanged. Attachment never upgrades the adapter's +`complete` assertion. The reference observer retains successful graph checks +through this path; partial pressure inputs retain their incomplete status. + +## Caller driven fleet reconciliation + +### Reader producer inventory + +`ReadReplicaManager::fleet_reader_enrollments_page(cursor, limit, now_ms)` +captures retained original reader requests and progress without awaiting journal +or native opening/closure calls. The canonical activation lane still serializes +all mutations; short index/progress locks make Pending and unknown work visible +during paused replies. Each page admits one MiB from the shared node byte ledger +before copying rows, permits 1–128 entries, and scans at most 10,000 obligations. +Drop the page to release its reservation. A closed runtime or exhausted ledger +returns an error; an unbound manager returns `None`, supplying no coverage. +Variable payloads may fill the byte budget before the requested row limit; +continue from the returned cursor. An oversized first row fails explicitly. + +| Observed state | Meaning | +| --- | --- | +| Original request, no acceptance reply | Acceptance is unknown; even an absent journal row cannot authorize another opening. | +| `opening_started`, no `opening_joined` | Native opening may still run or its task failed without a joined result. | +| Original event, `published == false` | Native progress is retained while durable publication remains unconfirmed. Execution and journal errors preserve separate original sources. | +| Running job, no request row | Accepted preparation is still work; zero local enrollment rows cannot prove absence of an obligation. | +| Unobserved or joining job | The original protocol result is unknown. A vanished/completed task handle does not prove closure. | +| Returned protocol, retained handle | The original response is available, but the retained drain owner still joins the task epilogue. | + +Cursors bind manager scope/boot, mode, job counts and every original row's +progress. Changed progress, retirement or mode requires restarting pagination. +This is an advisory interval scan, not an atomic or authenticated fleet proof. +Also traverse `fleet_readers_page` for installed native views and recheck the +durable roster and ordinary authority. These pages neither settle responsibility +nor certify replacement policy, shutdown or maintenance finalization. + +### Reconciler integration + +Construct `FleetReconciler::new(scope, claimant, profile, journal, observer, +transport)` with the same validated profile as the journal. Supervise one +application loop calling `reconcile_once(clock, deadline)`. The facade starts +no timer, runtime, or detached effect task. Time is logical milliseconds in the +node/journal clock domain. Supply a clock callback returning `Result`; +each boundary reads it directly and rejects regression. The deadline is a +Tokio monotonic instant. +Honor the report's suggested wake time: committed forward progress requests an +immediate next pass, while unresolved or cancelled work uses periodic retry. +Counts for clean release, fresh activation, canonical recovery, and cancellation +are separate and describe transitions committed during this pass. + +| Boundary | Reconciler behavior | +| --- | --- | +| Controller | Claim or renew by CAS, preserving every charged attempt across lease replacement. | +| Existing work | Inspect uncertain phases first; commit each dependent transition against the complete head and registry version. | +| Maintenance | Dispatch Cordon from Requested, commit Cordoned only after a durable bound result, then commit BeginEvacuation on a later pass. These steps continue when optional scheduling is stopped. | +| Planning | Overlay retained intents before selecting donors or receivers. Verify signed inputs, require complete fresh membership for count moves, and project unresolved receive costs before relief uses any remaining shared budget. Partial fresh inputs can evacuate maintenance Cells at normal pressure using their configured peak cost; the exact runtime quiescence and release barrier remains required. | +| Dispatch | Commit the phase before sending; check full retained acceptance and result binding. Transport timeout retains the phase and permit. | +| Unaccepted action | Prove absence in the same journal CAS that advances the head revision, fencing delayed old requests before retry. Accepted work remains charged and is inspected. Expired preparation/release returns to cleanup. | +| Expired unused receiver | After proven release, cancel and join expired prepared credit before first activation. Retain the release position and fleet permits, then use canonical ordinary admission after committed cleanup. | +| Endpoint failure | Reserve bounded deadline shares for charged attempts, maintenance and planning. Retain attempt errors in `failures` and the maintenance error in `maintenance_failure`. After an ambiguous timeout, reread the journal and controller epoch before continuing. Journal errors stop the pass. | +| Serving | Consume request-bound fresh actor/authority evidence before activation or retirement. A historical receipt cannot replace the current check. | +| History | Atomically retire with progress; read incarnation-specific cooldown and the global post-batch time from committed history. | +| Operator stop | Disable new allocations while accepted work continues through inspection and cleanup. | +| Maintenance deadline | Keep the cordon; stop new maintenance allocations and bound their deadlines by the operation deadline. Already accepted effects remain retained. Empty movement permits cannot prove completed role evacuation or shutdown. | + +The reconciler collects `FleetRoster` from every intent and enrollment page, +then passes it to `FleetObserver::observe`. The observer owns authenticated +native page collection, including every original failed boot named by the +roster. The reconciler rechecks the entire head and registry after capture. +Count balancing requires bootstrapped coverage, exact established boot rows, +fresh signed advertisements for every required boot, no Pending enrollment, +and ownership rows matching signed counts. Unknown live boots, failed missing +boots, replaced sessions and omitted actors disable count balancing. Pressure +relief still uses its existing source/receiver gates. The roster and all retained +row evidence enter the planner digest; a digest supplies no authority. + +`FleetObserver` additionally proves signing-key enrollment, discovery of +unexpected live records and complete native reader/follower role coverage. +Roster traversal and matching writer counts alone cannot finalize maintenance. +`FleetTransport` owns endpoint authorization and exact boot routing to the public +node action/inspection APIs. Driver models combine SQLite with simulated effects. +The `fleet_operations overload` command and its shared test now execute two +real moves across three leased nodes, reopen an independent controller client, +and verify original receipts and readback. `controller-restart` also loses both +source replies, waits for actual controller expiry, changes claimant/epoch, +adopts retained releases and joins expired receiver credit before activation. +Both scenarios verify empty resource ledgers after shutdown. `balance` also +honors real residence and post-batch samples, renews canonical boots, and +drives repeated bounded batches toward even counts with original receipt checks. +The request-bound collector supports the example's closed writer profile; +native reader/follower installations and unresolved role enrollment keep that +profile incomplete. Production complete observations, +source/receiver failure adoption, busy maintenance, and role finalization remain +required by the [fleet plan](../../../docs/fleet-operations-plan.md). + +## Shared fleet journal and enrollment transactions + +`FleetJournal` extends the action and enrollment journal contracts. Its +`FleetJournalSnapshot` reads the head and shared `RegistryVersion` together. +`compare_exchange` checks both exact versions, controller fencing, and reducer +invariants in the same durable transaction. Allocate also checks scheduling +policy and current retained source/receiver intents. Maintenance changes retain +the physical-node row and operation; retirement commits exact history with its +permit release. Older completed operations and cordons remain reachable. + +`FleetEnrollmentJournal` records Pending before starting ordinary boot, reader, +or follower enrollment. Only New authorizes first execution. Existing compares +the full spec and returns its original state; an unknown Pending record needs +inspection. Producers check intent revisions inside acceptance, retain failed +boots, and publish checked completion/refusal/closure evidence. Reboot lease +enrollment carries the retained mode and cannot open readiness under a cordon. + +Registry pages require an exact shared revision and a limit in 1..=128. A changed +revision restarts the scan. Bootstrap requires a controlled pause and complete +import; an empty live directory does not establish coverage. Stop/resume updates +this same registry version and blocks only new planned allocations. Previously +accepted actions and their resource obligations still reconcile. + +The application supplies the durable backend, authorization, and canonical +enrollment evidence. The [fleet journal example](../minion/README.md) +implements all three journal contracts in one local SQLite transaction domain. +Its focused tests exercise independent clients, lost commit replies and +reconstruction. Boot production uses the startup barrier above. The +[managed reader producer](read-replicas.md#bind-the-durable-fleet-producer) binds +ordinary activation and joined closure to this journal. Install it after read +replicas and before start; configured fleet hosts require it for readiness. +Complete observer coverage, failed-owner follower reconciliation, replacement policy, +maintenance finalization, and process/provider qualification remain required. + + +## Runtime maintenance quiescence checkpoint + +The canonical runtime now exposes exact-source foreground quiescence and +separate primitive maintenance readiness. See +[foreground quiescence](../../cellule-runtime/docs/deployment.md#foreground-quiescence-for-planned-maintenance) +for admission and proof limits. Fleet observations hash the quiescence flag, +readiness presence and all blocker classes under planner input domain +`cellule.fleet-planner-inputs.v3`. Existing retained attempt digests remain +opaque and unchanged; this is a new producer domain, not a journal rewrite. +The v3 producer also binds the separate peak maintenance cost (presence and all +admission dimensions), role, executable identity and individual blocker classes. +The driver can plan busy demand only for an exact Evacuating maintenance node +and boot, within its retained deadline. It tolerates local execution, publication, +lease and unknown primitive readiness until canonical release rechecks them; +Blob and foreign reader/follower/facility blockers remain blocked. A Draining +advertisement alone cannot authorize this demand. Receiver projection and the +shared count/byte limits still apply. Full role evacuation remains incomplete. + +## Explicit maintenance release checkpoint + +The committed `BeginMaintenanceRelease` transition requires the exact Evacuating +maintenance operation, physical source node and boot. It retains +`AttemptPhase::MaintenanceReleasing` (tag 13) and dispatches +`MovementAction::ReleaseMaintenance` (tag 8). Existing ordinary Releasing/Release +records retain their original policy. Deploy readers that understand these new +tags before enabling their producers; older readers reject unknown tags. + +The node-owned executor calls the canonical runtime maintenance release with the +minimum attempt/reservation deadline. A definite preflight refusal produces a +checked Rejected result and preserves its source error. Unresolved release keeps +both fleet permits and requires inspection or canonical recovery. Lost waiters +cannot cancel accepted work; source inspection loads the exact retained release +kind and never repeats an accepted release without evidence. These steps do not +prove reader/follower evacuation, Stopped, withdrawal or complete maintenance. diff --git a/crates/cellule-host/docs/original-writers.md b/crates/cellule-host/docs/original-writers.md new file mode 100644 index 00000000..c1f075ab --- /dev/null +++ b/crates/cellule-host/docs/original-writers.md @@ -0,0 +1,130 @@ +# Retain the original failed-boot writer set + +Use `FleetOriginalWriterCapture` to retain every original ownership observation +before starting effects that depend on a failed physical boot's complete writer +set. The application authenticates the process and catalog providers. The host +uses the existing authority, catalog readers, fleet registry and finite journal +work owner. + +## Integration order + +1. Capture and retain the original `FleetFailedBootProcessRequest` with + `capture_fenced`, before recovering or retiring its leader log. Preserve its + identity across controller reconstruction. +2. Publish the maintenance operation and its Draining intent in the existing + journal. Confirm a live controller and recovery claimant. +3. Supply `FleetOriginalCatalogs` with every canonical application/tenant storage + source the original boot could write. Its durable witness binds authenticated + original configuration and canonical backend mappings. An empty set requires + explicit authenticated no-writer configuration. +4. Capture and publish the original set using the same journal transaction + domain. Confirm publication before dependent effects. On an ambiguous reply, + use `FleetOriginalWriterInventory::load` to reconstruct the committed manifest + and every exact page; preserve their original bytes and capture interval. +5. Independently verify every original acknowledged prefix, exact dependency + availability and current successor serving. Combine complete writer, reader, + follower and accepted-work evidence before role settlement or node finalization. + +The caller accounts bounded metadata work through its existing finite owner and +uses one absolute deadline. Application providers own credentials, authorization, +accepted external jobs and durable process joining. A PID, expired lease, filtered +resident list or successful reconstruction cannot establish a complete set. + +```rust +use cellule_host::fleet::{ + FleetFailedBootProcessRequest, FleetFailedBootProcesses, + FleetOriginalCatalogs, FleetOriginalWriterCapture, FleetOriginalWriterJournal, +}; +use cellule_runtime::{Result, identity::SessionId, node::NodeDirectory}; +use cellule_runtime::fleet::operations::OriginalWriterInventoryRecord; +use tokio::time::Instant; + +async fn retain_original_writers( + journal: &dyn FleetOriginalWriterJournal, + directory: &NodeDirectory, + processes: &dyn FleetFailedBootProcesses, + catalogs: &dyn FleetOriginalCatalogs, + original: &FleetFailedBootProcessRequest, + claimant: SessionId, + deadline: Instant, + mut clock: impl FnMut() -> Result, +) -> Result { + let capture = FleetOriginalWriterCapture::capture( + journal, directory, processes, catalogs, original, + claimant, deadline, &mut clock, + ).await?; + capture.publish( + journal, directory, processes, catalogs, + claimant, deadline, &mut clock, + ).await +} +``` + +After controller restart, load the current complete journal snapshot and supply +the retained operation ID and original process-request digest. The loader checks +the committed pointer and all ordered pages before exposing an inventory. `None` +means no committed set at that read; only a returned manifest with zero owners +establishes retained empty input. Neither result proves accepted capture work +cannot still publish. + +```rust +use cellule_host::fleet::{ + FleetJournalSnapshot, FleetOriginalWriterInventory, FleetOriginalWriterJournal, +}; +use cellule_runtime::{Result, identity::Digest}; +use cellule_runtime::fleet::operations::OperationId; +use tokio::time::Instant; + +async fn reload_original_writers( + journal: &dyn FleetOriginalWriterJournal, + current: &FleetJournalSnapshot, + operation: OperationId, + original_process_request: Digest, + deadline: Instant, +) -> Result> { + FleetOriginalWriterInventory::load( + journal, current, operation, original_process_request, deadline, + ).await +} +``` + +The inventory remains immutable historical input. Its `writers()` iterator spans +every validated page. It does not refresh the collection interval or confirm +current native roles. Aggregate collectors must independently recheck current +authority, every successor prefix, process closure and the complete operation +barrier before acting. + +## Checks and limits + +| Boundary | Required behavior | +| --- | --- | +| Original lifetime | Join the exact original process and every accepted native/external owner through the existing provider; recheck canonical fence, claimant and full roster. | +| Complete sources | At most 128 unique sorted application/tenant scopes; reread the same request-bound source set and durable witness. Providers attest the actual canonical backend, including independent adapters. | +| Catalog traversal | Capture all 256 original heads before pages; ordinary verified page reads and final head/ETag revalidation must complete. | +| Authority history | Inspect every catalog entry, including unused bootstrap absence and owners outside the original boot. Missing legacy/restored history is `OwnerHistoryIncomplete`, never empty. | +| Retained owners | Preserve full original Controls and targets for every matching ownership epoch, including rootless, recovering and object-covered writers. | +| Bounds | At most 10,000 total catalog entries and 10,000 inspected ownership epochs; at most 10,000 retained observations, 64 per page. Record is at most 64 KiB; page is at most 1 MiB. Excess refuses without truncation. | +| First publication | Within the original monotonic 30-second interval and operation/controller deadlines, compare full head/registry, boot and intent; atomically publish all pages, the original pointer and one registry advance. | +| Replay | Exact committed bytes return as historical retention, even after the original deadline. A different original set conflicts. No refreshed timestamp or latest-set replacement. | +| Reconstruction | `FleetOriginalWriterInventory::load` checks the full pointer barrier and validates every immutable page before returning. Missing/corrupt pages refuse the entire set; source errors are preserved. | + +Catalog heads are sequential observations. These metadata records provide no +atomic global snapshot, root-retention pin, authority grant or successor proof. +The original boot must already be joined; accepted original work cannot mutate +its catalog after that barrier. Other boots may continue ordinary authority work. + +## Reference evidence + +The canonical [minion](../minion/README.md) SQLite adapter commits manifest/pages +in the existing accepted blocking-job owner. Its public capture tests traverse +two independently configured application/tenant catalogs and retain writers +removed by actual canonical takeover. They exercise exact replay, reconstruction, +provider errors, missing history, competing clients and canceled/lost replies. +Reload cases distinguish absent and authenticated empty sets, refuse a stale +snapshot or missing/corrupt final page, and preserve the original SQL error. +Their joined child is a lifetime stand-in; OS-crashed CellNode, accepted external +jobs, successor prefixes and provider fault qualification remain required. + +```sh +cargo test -p cellule-host --example fleet_operations --all-features --locked writer_tests +``` diff --git a/crates/cellule-host/docs/read-replicas.md b/crates/cellule-host/docs/read-replicas.md index a7d44d27..74905ce4 100644 --- a/crates/cellule-host/docs/read-replicas.md +++ b/crates/cellule-host/docs/read-replicas.md @@ -6,7 +6,9 @@ flowchart LR Hint --> Select[Validate selected reader boot] Select --> View[Authenticated immutable view] Poll[Periodic reconciliation] --> Select - Drain[Node drain] --> Cancel[Cancel activation and join work] + Drain[Node drain] --> Cancel[Cancel activation] + Cancel --> Close[Fence and detach views] + Close --> Join[Join accepted native and refresh work] ``` | Boundary | Behavior | @@ -15,9 +17,318 @@ flowchart LR | Installation | Manager starts after the node task group; dispatcher uses the same resolver. | | Recruitment | Locally owned Cells send scoped hints through an authorized peer client. | | Refresh | New view must prove its root and receipt before selection. | +| Periodic repair | Scan installed views and retained producer requests; replay original results and joined nonexecution exclusions without another hint. | | Missing reader | Queries return a replica error; they do not trigger activation. | -| Drain | Cancels activation before closing views. | +| Drain | Cancels activation, closes admission, detaches views, and joins accepted work before removing ownership. | Commands do not wait for readers. Reconciliation repairs dropped hints and membership changes; it is not a freshness guarantee. See [application read policies](../../cellule-app/docs/invocation.md). + +## Bind the durable fleet producer + +Install read replicas, then `CellNode::install_fleet_reader_enrollment` before +startup and the first activation. Hosts configured with a fleet startup intent +require this owned binding before readiness can open. A rejected activation +before binding creates no responsibility and does not prevent installation. + +```rust +use std::sync::Arc; +use cellule_host::{CellNode, fleet::FleetJournal}; +use cellule_runtime::{fleet::operations::FleetScope, identity::NodeId}; + +fn bind_reader_registry( + node: &CellNode, + scope: FleetScope, + physical_node: NodeId, + journal: Arc, +) -> cellule_runtime::Result<()> { + node.install_fleet_reader_enrollment(scope, physical_node, journal) +} +``` + +| Boundary | Bound manager behavior | +| --- | --- | +| Pending | Verify source and receiver signed physical boots, read bounded exact-version intent pages, and accept the pinned root and both intent revisions atomically. Only New starts native opening. | +| Ownership | Retain up to 32 finite activation jobs using the runtime byte ledger. A returned activation joins its exact task and byte-token release. Dropping the hint/prepared-activation or join waiter cannot cancel accepted opening or publication. | +| Established | Check the exact initial receipt and publish evidence binding the original request, source code/schema, owner endpoint and pinned root. A lost publication reply retains the same result for replay before refresh. | +| Removal | Join canonical closure before Retired. Keep the fenced view, original request and event across cancellation or failed publication. `remove` and `shutdown` return errors and can be retried. | +| Unknown acceptance | Preserve the original request. After its original activation releases the lane, periodic repair or removal atomically publishes the exact Refused exclusion for a never-started opening. Missing/Pending acceptance cannot authorize another opening. Unjoined native work remains blocked. | +| Diagnostics | `enrollment_completion(cell)` returns the original source/request, acceptance, current event and independently retained native/publication errors. These are diagnostic facts, not current serving or replacement-policy proof. | + +Refresh, policy eviction, peer hints and shutdown share the existing manager +lane. New local roles check the shared admission gate before acceptance and +again at native opening. The manager bounds retained responsibilities at 10,000 +and charges three record envelopes per entry; ordinary native view admission +still applies. A task failure cannot turn an unjoined native opening into +retirement. Shutdown joins healthy siblings and preserves the original task +failure. + +The existing five-second reconciliation loop scans the sorted union of installed +views and retained enrollment requests, with at most 64 attempts per batch. It +charges the bounded temporary index to the runtime metadata ledger. The loop +may scan before lease installation and after local fencing; this charge grants +no native admission or lease authority. New openings and native inventory reads +retain their live-lease checks. A missing view is +reconciled only after acquiring the original activation lane; a still-owned +opening cannot be mistaken for nonexecution. Selected open views replay their +original Established evidence before refresh. Fenced views left by cancelled +removal resume canonical joining and the same Retired event before remote I/O. +A failed result does not erase its original error or prevent other rows from +progressing. Capacity refusal retains all responsibilities for a later tick. + +On a managed Draining node, periodic reconciliation preserves an open reader +even when cordon removes it from placement. It repairs original establishment +without opening or refreshing a view. Explicit maintenance checks below govern +evacuation; fenced views still resume their original closure and retirement. + +The journal retains failed-boot and Pending rows across restart. Reconstructing +a manager does not erase or automatically settle them. Applications still own +complete roster collection, evidence storage/validation, failed-process closure +proof, replacement redundancy and maintenance finalization. The reference +example tests this producer with real SQLite readers and its durable journal; +its movement commands still report incomplete role observation. + +## Prepare exact enrollment inputs + +| Method | Boundary | +| --- | --- | +| `prepare_source(target, origin)` | Observe canonical authority, verify the live signed owner boot and current reader selection, and return opaque `ReadReplicaSource` metadata. Reserve no view resources and create no reader. | +| `activate_source(source)` | Initially open the original pinned root through ordinary admitted activation. Recheck selection, incarnation/code, owner/epoch and signed boot identity. Refuse an already installed view without refreshing it. | +| `activate(target, origin)` | Ordinary authenticated hints retain their existing refresh behavior, using the same source preparation and native opening path. | + +An enrollment adapter can derive immutable role inputs before Pending acceptance: + +```rust +use cellule_runtime::ReadReplicaSource; +use cellule_runtime::fleet::operations::{EnrollmentRole, PublishedPosition}; + +fn reader_role(source: &ReadReplicaSource) -> EnrollmentRole { + EnrollmentRole::Reader { + target: source.target().clone(), + position: PublishedPosition { + incarnation: source.description().incarnation, + epoch: source.epoch(), + root: source.root().clone(), + }, + } +} +``` + +Bind `source.node()`, `source.owner().session` and `source.fleet()` to the +source endpoint and fleet scope. Without a managed binding, the application must obtain current endpoint +intent revisions and journal Pending atomically before activation. Only New +acceptance permits first execution; an ambiguous or Existing reply requires +inspection. Later publication under the same owner does not change the root +opened by `activate_source`. Ordinary subsequent refresh remains available. +After entering the activation lane, prepared opening joins its work on manager +closure. The adapter must retain that future; dropping its waiter without an +owner still leaves an unknown enrollment outcome. + +With the fleet binding, both activation methods use the owned producer above. +Without it, the adapter must retain accepted activation across waiter cancellation, +publish checked completion and retire only after joined closure. Source metadata +and snapshot receipts cannot establish complete enrollment coverage. + +If the retained owner joins without ever starting native opening, removal uses +`refuse_unexecuted_enrollment` in the shared journal transaction domain. It retains +an exact Refused exclusion row whether acceptance is absent, Pending, or its reply +was lost. A delayed acceptance must replay that terminal row. Reading absence +alone cannot close this obligation. Established or differently settled rows +conflict; they require their original native closure evidence. Once native opening +started, joined opening/closure continues to publish the original Retired event. + +## Join reader closure + +`CellReadReplica::close()` fences new work across every clone. +`close_and_join().await` additionally detaches the shared snapshot and waits for +accepted queries, authority reads, refreshes and native SQLite opens. This +includes older snapshots still used by queries and native jobs whose request +waiters were cancelled. Retaining a peer clone cannot retain a detached view's +memory, descriptor or disk charges after joined closure. + +The manager retains each reader until joined removal and, when bound, durable +retirement succeed. Cancelling a +removal or shutdown waiter leaves that obligation inventoried; a later call +joins the same work. Shutdown fences all views before joining up to 16 at once. +A host drain deadline can return while native work remains owned: the host +stays Draining until a later shutdown joins it. + +The returned receipt and `receipt()` describe the last installed snapshot, +including after closure. They do not prove current authority, replacement +redundancy, or durable fleet registry retirement. Publish retirement only after +the canonical closure and the application's checked enrollment evidence. + +## Evacuate a managed reader + +Use `ReadReplicaManager::evacuate` with the original Established enrollment and +the current journal operation in Evacuating. The owner-side recruiter supplies +replacement activation through the existing authenticated peer path. + +| Check | Required observation | +| --- | --- | +| Donor | Exact physical node/session, current Draining intent, Draining signed boot, and Established boot enrollment. | +| Journal | Complete bounded intent/enrollment traversal and unchanged head/registry before closure and after retirement. | +| Replacements | Current canonical policy selection on other physical nodes; exact Active managed boots and Established reader enrollments. | +| Native readiness | Authenticated status replies cover the original and current published prefixes. A second probe after closure also covers any refresh completed through a retained peer clone. | +| Closure | Canonical local joining followed by confirmed retirement of the exact original request. | +| Freshness | Same authority lifetime, nonregressing publication, unchanged policy, and fresh selected boot identity/liveness checks. | + +```rust +use cellule_host::read_replicas::{ReadReplicaManager, ReaderEvacuation}; +use cellule_runtime::{ + fleet::operations::{EnrollmentRecord, MaintenanceOperation}, + peer::ReplicaPeerClient, +}; + +async fn evacuate_reader( + manager: &ReadReplicaManager, + original: &EnrollmentRecord, + operation: &MaintenanceOperation, + peer: &ReplicaPeerClient, + deadline: tokio::time::Instant, +) -> cellule_runtime::Result { + manager.evacuate(original, operation, peer, deadline).await +} +``` + +No adequate spare means no new local close. Zero desired readers permits checked +closure without replacement. An owner, policy, boot or registry change refuses +evidence; a change after closure can leave the original retired while the +operation remains incomplete. Cancellation, deadlines and lost retirement +replies preserve the original closure and producer event. Retry that same +Established request; absence from the local view map alone proves nothing. +The effective deadline is the earliest of the caller's monotonic deadline, +the operation's remaining wall-clock deadline and thirty-second capture bound. Timeout errors retain their +original source. + +Roster pages, retained records, temporary boot observations and returned +evidence use the runtime's shared metadata ledger. Drop `ReaderEvacuation` when +finished consuming it to release its retained charge. Its capture interval +belongs to that attempt; replayed retirement keeps its original journal times. + +This result settles one local reader. It does not settle foreign followers, +failed processes, writers, complete fleet membership, or terminal shutdown. +The controller must persist and revalidate the observed replacements and all +remaining obligations before finalization. A receiver can fail after capture. + +The reference fixture exercises real managed boots, signed peer dispatch, +native readers, cancellation, lost replies and the post-probe refresh race: + +```sh +cargo test -p cellule-host --example fleet_operations --locked reader_tests::evacuation +``` + +## Persist and revalidate reader replacement coverage + +`ReaderEvacuation::durable_record` builds a `ReaderEvacuationRecord` and every +canonical replacement page. The manifest retains the original full head digest, +registry version, maintenance operation, retired request/history, authority, +policy revision/count, closed prefix and collection interval. Each replacement +binds its signed boot identity, exact Established enrollment digest and observed +native prefix. Pages hold at most 128 entries and cover the full 10,000-reader +policy bound; missing, reordered, foreign or duplicate entries refuse. Decoding +validates historical shape and supplies no authentication or readiness. + +Implement `FleetReaderEvacuationJournal` in the same transaction domain as the +existing fleet/enrollment/action journal. Commit every immutable page, manifest +and latest operation/request pointer together, advancing the shared registry. +Compare the full original snapshot, current operation/controller, original +Retired row and replacement Established rows/intents inside that transaction. +Exact replay retains original times and cannot restore a superseded pointer. +The SQLite example supplies this implementation through its existing finite +backend jobs; cancellation does not cancel an accepted commit. + +```rust +async fn persist_reader_evacuated( + capture: &cellule_host::read_replicas::ReaderEvacuation, + journal: &dyn cellule_host::fleet::FleetReaderEvacuationJournal, + verifier: &cellule_host::fleet::FleetReaderEvacuationVerifier, + deadline: tokio::time::Instant, + clock: impl FnMut() -> cellule_runtime::Result, +) -> cellule_runtime::Result { + cellule_host::fleet::FleetReaderEvacuationPublication::publish( + capture, journal, verifier, deadline, clock, + ).await +} +``` + +Construct the verifier from the existing `NodeDirectory`, `CellAuthority`, +`ReadPolicyStore` and authenticated `ReplicaPeerClient`. Publication performs +fresh pre/post checks. Inspect `record()` even if `confirmed()` fails: a lost +reply or subsequent policy/authority/boot change can leave committed history +with an independently retained source error. Reconstruct the client, load the +same digest or latest original request, then call `verifier.recheck`. It reloads +every page, traverses the complete current roster and probes native readiness +twice around authority, policy, selection and signed-boot rechecks. Fresh +confirmation has its own interval; it never restamps historical capture. + +After policy, replacement boot, owner, deadline or operation-session changes, +`verifier.refresh` captures a new immutable policy candidate from the previously +committed original retirement. The ordinary recruiter must supply current ready +replacements. `publish_refreshed` commits the candidate under its new exact +barrier. This can repair coverage while Closing without repeating native reader +closure; old manifests and original retirement times remain retained. Missing +capacity, a changed incarnation or incomplete historical pages stay blocked. + +Applications account these bounded copied metadata buffers and authenticate +provider/peer calls. One checked reader still supplies no complete role graph, +failed-process proof, affected-writer relocation or permission to finalize a +physical node. Persisted follower policy, aggregate controller settlement and +process/provider qualification remain required. The +[executable evidence](../minion/README.md#durable-reader-replacement-evidence) +uses real managed readers, signed status probes and independent SQLite clients. + +## Observe managed reader obligations + +`ReadReplicaManager::fleet_readers_page(cursor, limit, now_ms)` returns sorted +snapshot positions, canonical lifetime observations and the total managed-view +count. Choose 1 through 128 entries. Each page retains one MiB from the node's existing native-byte ledger +until dropped. The manager limits its collection to 10,000 views; ordinary +memory and descriptor admission can refuse a view sooner. + +Observation waits in the existing activation/removal lane, so an accepted open +cannot disappear between its installation and the scan. Activation, removal, +replacement, or shutdown invalidates the continuation. Refreshing an existing +view updates its receipt without creating a new topology. Cordon remains +visible independently of reader count; stale remote advertisements cannot +admit a new local view. + +```rust +use cellule_host::read_replicas::ReadReplicaManager; + +async fn managed_reader_count( + manager: &ReadReplicaManager, + now_ms: i64, +) -> cellule_runtime::Result { + let page = manager.fleet_readers_page(None, 128, now_ms).await?; + Ok(page.total_views()) +} +``` + +The continuation fingerprint covers the exact manager/session, node admission, +manager closure and every managed reader's receipt, lifetime count, admission +closure and snapshot attachment. Each page hashes the complete bounded set in +place, including rows outside that page; it copies at most the requested rows. +Refresh or closure through a retained peer handle can invalidate continuation +without changing the manager's view set. A changed scan must restart. Admission +is checked again after capture; its change rejects the page. + +Matching fingerprints describe capture intervals. An open query can start and +finish between captures, returning to the same count. They do not establish an +atomic view or upgrade an open reader to joined. Closed, detached zero remains +the canonical stable local-join condition. + +Each entry is a `cellule_runtime::client::ReadReplicaLifecycleObservation`. +`receipt()` returns its last installed position. `admission_closed()` and +`snapshot_attached()` report shared state across every retained reader clone. +`retained_lifetimes()` counts canonical guards for snapshots and accepted +operations, including native queries and refresh jobs whose callers cancelled. +It is neither a query count nor a count of handle copies. + +`locally_joined()` requires closed admission, detached shared state and zero +original lifetimes. Closed plus zero cannot admit another operation or attach a +new snapshot. A closed reader with pending native work stays visible and reports +false. These local observations do not prove replacement policy, current remote +authority, durable producer retirement or successful host shutdown. Maintenance +must independently settle those obligations before declaring a node safe to stop. diff --git a/crates/cellule-host/minion/README.md b/crates/cellule-host/minion/README.md new file mode 100644 index 00000000..847d4640 --- /dev/null +++ b/crates/cellule-host/minion/README.md @@ -0,0 +1,537 @@ +# Minion fleet operations reference + +Status: durable journal and public-driver models, plus a finite real three-node +admission-overload and controller-restart scenarios with receipt readback. +The `balance` command exercises real count convergence. Complete maintenance, +receiver-loss, production observations and +qualification remain required by the [fleet plan](../../../docs/fleet-operations-plan.md). + +## Run + +Run the real-node scenario from the workspace root: + +```sh +cargo run -p cellule-host --example fleet_operations --locked -- overload +``` + +It creates three independently leased CellNodes over shared strict in-memory +object storage, initializes twelve real SQLite Cells, and acknowledges a command +on each. A held seven-GiB disk admission token causes the existing actor's +measured ledger classifier to enter Shedding after its normal dwell. The public +reconciler allocates its bounded two-move batch; the scenario releases that +pressure token, stops new scheduling, prepares real receivers, and reopens an +independent SQLite controller client before settling movement. Both receiver +nodes restore state, resolve the original command outcomes and confirm the old +source handles are fenced. Exit joins all three runtimes and the journal and +checks their resource ledgers are empty. Output separates release, activation, +retirement, receipt checks, shared budget maxima and advisory blockers. It also +reports three confirmed boot retirements after joined runtime shutdown. + +Each boot now registers its retained physical intent, builds with the shared +startup hold, journals Pending, and creates its actual signed advertisement in +the canonical `NodeDirectory`. Established publication and an atomic boot/intent +read precede `start()`. Confirmation preserves the role hold until the ordinary +required-component checks pass. Boot advertisements expose zero receive capacity +before readiness; later observations publish actual runtime samples through +canonical directory CAS. The finite example derives its local lease guard from +the canonical advertisement's expiry and advances it only after confirmed CAS. +The caller drives these heartbeats while observing; there is no new background +scheduler or production heartbeat/provider implementation. +Before each renewal it calls `refresh_fleet_intent` against the original Established +boot. A retained maintenance intent closes the shared role gate and is reflected +in the actual signed operational sample even if the Cordon RPC was lost. Failed +intent reads prevent that heartbeat and guard renewal. Existing writer routing +remains available until its own quiescence or terminal drain. + +The application retains each original boot advertisement and binds confirmed +boots to native closing before readiness. That owner joins runtime and lease +maintenance, withdraws through the canonical directory, checks the permanent +session tombstone, and retires the exact registry obligation. The retained +application cleanup also handles partial startup and fences its lease guard. Both +paths require `is_withdrawn`: a tombstone retaining a log or recovery claimant +cannot settle boot retirement. Replay +can adopt an already committed withdrawal or retirement. A missing advertisement, +lease expiry or unresolved Pending record alone cannot prove closure. + +Movement serving checks now prove native derivation from the exact released or +materialized recovery root, then verify every current origin dependency and +recheck actor/authority state. Compaction cannot substitute higher counters for +the original prefix. Verification uses shared runtime memory/I/O admission and +bounded inventories; missing legacy lineage or current origin bytes refuse the +observation. See the [prefix contract](../../cellule-runtime/docs/storage.md#prove-an-exact-root-prefix-after-compaction). +This per-movement evidence does not certify complete physical maintenance. + +This measures **admission pressure**, not physical disk usage or throughput. The +actor retains its existing local eviction budget; fleet counts cover only its +journal-backed batch. The fixed three-boot transport pins identities in trusted +composition. The reconciler supplies its fully traversed durable `FleetRoster` +to the collector, including Pending work, failed boots and terminal records, +and rechecks the original full snapshot after collection. Count balancing also +requires exact established boot advertisements, settled enrollment and writer +rows matching signed counts. Its collector uses request-bound native snapshots, +the original enrolled signing keys, both directory session scans and fresh Cell +authority. It uses `FleetNodeInventoryScan` for every native category and continuation, +rereads authority, rechecks all seven complete category fingerprints after the +fleet-wide authority scan, and confirms the full journal barrier. Complete coverage is supported for this private constructor's bounded +writer-only profile: twelve catalog-backed SQL Cells and three managed boots. +Foreign log discovery uses `FleetFollowerReferences` to traverse every page and +recheck exact coverage/liveness after native capture, including expired or fenced +owners. Its complete listing supplies no permission to discard a required tail. +Unexpected advertised boots, Pending/role enrollment, native role installations, +expired/fenced log obligations, changed topology or stale ownership prevent +complete counts. Unbound pages alone supply no absence proof; the closed +constructor profile and durable/current-authority checks are also required. +It stops new scheduling after the bounded batch; this does not demonstrate +sustained-overload convergence. It is not a production complete observer, deployment authentication system, +process-crash test, or distributed-provider qualification. Reader/follower +deployment collectors still need complete native-role and replacement-policy +evidence before enabling maintenance finalization. + +Run ordinary count balancing with real residence and repeated batches: + +```sh +cargo run -p cellule-host --example fleet_operations --locked -- balance +``` + +The scenario starts at 12/0/0 and honors the planner's actual 60-second residence +rule and post-batch sample barrier. Each pass renews canonical boot leases and +drives the same shared two-move budget. Eight distinct Cells converge to 4/4/4, +with durable cooldowns preventing repeat movement. The scenario resolves all +eight original acknowledged outcomes, reads their SQLite values, verifies +successor authority and fences the old handles. It issues five-minute command +receipts for this longer scenario; overload and controller-restart retain their +original one-minute receipts and profiles. A 120-second convergence deadline +fails with retained attempt diagnostics, then exit joins all nodes and retires +their exact boot obligations. This is real-time in-process count convergence; +sustained workload, process/provider and complete role qualification remain +required. + +Run controller replacement after ambiguous source replies: + +```sh +cargo run -p cellule-host --example fleet_operations --locked -- controller-restart +``` + +This uses the same real nodes, journal, bounded driver and receipt checks. +The transport loses both replies after source release commits. The scenario +keeps both attempts charged, waits for the actual controller lease to expire, +reopens an independent journal client and resumes with a different claimant. +Epoch 2 adopts retained source proofs; the old claimant's renewal is fenced. +Expired unused receiver credit is cancelled and joined before first activation +through the canonical ordinary admission path. Output also reports lost replies, +controller epoch and confirmed expired-credit cleanup. No release is repeated. +The fixed test profile uses a three-second controller lease and 500-ms interval; +node leases, reservation timestamps and observations use actual time. This is +in-process controller replacement with durable local reconstruction; process +crash, provider and receiver-session loss qualification remain outstanding. + +Recovered movement's fresh serving check requires canonical acquisition metadata +matching the full journal input/result. A pinned suffix additionally requires the +original digest-verified manifest row, original closed owner, exact canonical +materialization and native lineage/current origin verification. Interrupted +claims may materialize at a later epoch while preserving the original manifest +scope. The existing provider's read-only lookup must remain available after +reconstruction and publication; missing/corrupt metadata refuses fresh serving. +The host charges transient acquisition/manifest bytes through its runtime ledger, +then repeats the ordinary actor/authority/inventory boundary. Historical outcome +replay does not refresh this observation. Complete original writer/suffix +aggregation and maintenance finalization remain required. + +To inspect a retained example journal, supply a database path in an existing directory, +outside canonical Cell storage. The command creates the journal if absent or +reopens it with the same scope and profile, prints its bounded head summary, +and closes the connection after joining accepted work. + +```sh +cargo run -p cellule-host --example fleet_operations --locked -- inspect-journal /tmp/cellule-fleet.sqlite +``` + +The example uses fleet identity `[200; 32]`, application identity `[3; 16]`, and +`FleetProfile::default()` for overload and journal inspection. A new journal starts with bootstrap incomplete and +scheduling stopped. Inspection does not bootstrap the registry or enable +movement. Reopening with a different scope or profile fails. + +On workstations with the mounted Workspace volume, set `CARGO_TARGET_DIR` to +a directory for this checkout beneath `$HOME/Workspace/crabbuild-target`. + +## Transaction and lifecycle contract + +Original-process cases capture and confirm a permanent boot fence while the +leader log is still Recovering, before the existing recovery coordinator runs. +They retain the same process identity through native log retirement, journal +publication, lost replies and independent adapter reconstruction. Changed provider +evidence, suspended reads across registry changes, foreign identities and regressing +clocks refuse. Boot retirement still waits for the complete terminal role barrier. +These cases use a joined child lifetime stand-in and a cold follower ensemble with +no Cell suffix or external jobs; they do not qualify an OS-crashed CellNode or +prove complete affected-writer retention and successors. + +```sh +cargo test -p cellule-host --example fleet_operations --all-features --locked recovered_followers::failed_boot::process_tests +``` + +The original writer capture traverses every authenticated application/tenant +catalog and canonical owner history, including rootless originals already removed +by takeover. `FleetOriginalWriterJournal` retains all pages and one immutable +operation/process pointer in the same SQLite transaction that advances the +registry. Exact replay and independent reconstruction preserve original Controls, +intervals and identities. `FleetOriginalWriterInventory::load` follows that +committed pointer and validates every page before exposing the historical set; +absence differs from an explicitly retained empty set. Missing/corrupt final +pages refuse the whole set and journal source errors remain inspectable. +See the [integration recipe](../docs/original-writers.md) +for source authentication, bounds and successor requirements. This metadata +foundation does not establish successor availability or maintenance completion. + +```sh +cargo test -p cellule-host --example fleet_operations --all-features --locked writer_tests +``` + +The reader producer cases use `fleet_reader_enrollments_page` while original +acceptance, establishment and native open replies are paused. They check original +requests, separate errors, accepted preparation before a row exists, page memory +admission/release, stable pagination and cursor invalidation. Native reader views +remain a separate inventory; the collector still requires complete role and +authority evidence before maintenance can finalize. + +The managed reader evacuation fixture starts three real leased, enrolled boots +and activates native readers through signed peer dispatch. It checks nonzero +replacement readiness before closure and again after original retirement; +missing spares preserve the Draining donor's view. Cases cover lost retirement +replies, cancellation, deadline source errors, policy changes, metadata refusal, +replacement withdrawal, and refresh through a retained peer clone. Every case +joins native shutdown and checks empty resource ledgers and original boot +withdrawal/retirement. This supplies per-reader evidence; the public driver still +needs complete role settlement and finalization. + +```sh +cargo test -p cellule-host --example fleet_operations --locked reader_tests::evacuation +``` + +Follower producer cases capture `fleet_follower_enrollments_page` while provider +preparation and member acceptance are paused. They preserve all original requests +and proof identities, check the shared page charge and memory refusal, reject +changed/missing continuations, and keep original failed retirement/error Arcs. +Idle protocol or absent rows do not establish a joined supervisor or empty native +lanes. The reference collector still needs authenticated complete role envelopes. + +The public host durability suite also captures `fleet_durability_supervisor` +while original retirement/recruitment is unresolved, after byte admission +closes during drain, and after local rotation completion. It compares original +proof/error references and verifies fixed metadata admission and release. +Supervisor completion and bank availability are separate from producer row +counts; these captures still cannot certify fleet settlement. + +```sh +cargo test -p cellule-host --test node --all-features --locked node::durability +``` + +[`SqliteJournal`](journal/mod.rs) implements all three host journal interfaces +in one local SQLite transaction domain. It lives in the embedding example; +the host library remains provider neutral. + +| Boundary | Implementation | +| --- | --- | +| Controller and permits | `BEGIN IMMEDIATE` compares the complete head and registry version, then invokes the pure reducer and current intent allocation gate. | +| Maintenance | Publish the physical-node intent and retained operation with the head. Keep the original request separately from later deadline and boot changes. Earlier cordons survive later operations. | +| Enrollment | Check exact source/target intent rows and publish Pending in one transaction. Full input comparison precedes current intent checks on replay. Pending remains an obligation after expiry. | +| Boot confirmation | Read current physical intent and the exact Established node enrollment in one transaction; reject pending/settled/foreign records. No cached older intent can open readiness. | +| Action acceptance | Check current head, permit, endpoint and intents before inserting the immutable acceptance. Key includes action digest, physical node and boot; source and receiver inspections remain distinct. | +| Unaccepted action | `ResolveUnaccepted` checks that the exact attempt/effect/endpoint has no acceptance in the same `BEGIN IMMEDIATE` transaction that advances the head revision. A delayed old envelope then fails; an acceptance that won the race prevents retry. Both permits remain charged. | +| Results and recovery | Retain original acquisition/recovery inputs and positions. Require matching basis/evidence before publishing successful activation or recovery. Unknown may advance to checked completion; terminal results are immutable. | +| Fresh inspection | Check the request's exact head action, registry version and endpoint intent in one read transaction. Create no effect acceptance or cached observation result. The public host path performs the actual current check. | +| Original writers | Compare the full barrier, original boot and current Draining operation; atomically insert every page and the immutable operation/process manifest. Exact replay preserves bytes without a registry advance. | +| History | Insert the exact progress page in the same transaction that retires permits. Loading verifies the page digest and head reference. | +| Durability | SQLite WAL with `synchronous=FULL`, foreign keys and a five-second busy timeout. Each client owns one connection; independently opened clients serialize through SQLite. | +| Bounds | At most 32 accepted blocking jobs per client. Excess calls return Capacity. Pages contain at most 128 sorted records; canonical codecs bound record sizes. | +| Cancellation and close | An accepted native transaction survives a dropped waiter. Close stops admission, joins all accepted jobs, releases their slots and closes SQLite. Backend and codec errors retain their sources. | + +Applications must retry an ambiguous commit with the same immutable request +identity and reread retained state. A timeout cannot free a permit or turn an +unknown enrollment into a refusal. + +## Evidence and integration work + +```sh +cargo test -p cellule-host --example fleet_operations --locked +cargo clippy -p cellule-host --example fleet_operations --tests --locked -- -D warnings +``` + +The focused tests exercise independent connections, competing controller and +allocation CAS, lost replies after commit, rollback before commit, reopening, +immutable action inputs/results, retained cordons/enrollments, malformed head +and missing operation references, bounded admission, and drain after waiter +cancellation. Recovery fixtures use canonical failed-session proof and synthetic +record positions; actual Cell restoration is covered separately by runtime and +host integration tests. These journal tests do not activate Cell actors. + +SQLite reconstruction here establishes the local adapter behavior. Process +crash, filesystem faults, distributed journal providers and mixed deployed +readers require the plan's qualification campaign. The schema has no migration +contract yet; format mismatch fails closed. + +The example test target also invokes the exported `FleetReconciler` with this +SQLite adapter and signed synthetic observations. Its simulated transport checks +phase publication before dispatch, fresh activation/retirement, independent +cleanup, retained permits after lost replies/timeouts, competing controllers, +stop-new-moves behavior, and pressure relief using remaining shared budget. +The separate `scenario::tests` cases invoke the same real-node implementations +as the `overload` and `controller-restart` commands. The model tests do not create Cell actors. Planned +maintenance and receiver-loss scenarios still require their complete barriers; +controller replacement here retains all three live node sessions. + +The host now owns each whole closing attempt after a shutdown waiter disappears. +It retains the same lane through facility/runtime/lease cleanup and exposes local +`drain_observation()` results with original failure history. Successful stop +joins the original task; retrying an incomplete deadline resumes the same native +resource owners. Each confirmed example boot now installs its original directory +version and SQLite enrollment row into that owner before readiness. Native +shutdown checks canonical withdrawal and committed boot retirement before +Stopped; a lost retirement reply retains Draining until an exact replay confirms +the original evidence. Partial startup still uses the retained application's +boot cleanup. These local observations supply no relocation or redundancy proof +and do not enable the still-refused fleet Finalize action. +Publish native rotation/reader proofs before starting terminal shutdown: weak +native handles can disappear when an autonomous successful stop clears owners. + +Public startup cases cover missing/Pending/foreign records, lost acceptance and +Established replies, a cordon racing accepted enrollment, a drained-mode reboot +with management available, delayed stale confirmation, shutdown racing a read, +original backend errors, and canonical boot withdrawal/retirement replay. Native +closing cases cover sole-waiter cancellation, an ambiguous committed retirement, +a retirement deadline, a late signed heartbeat, missing canonical storage and +failed role closure without invented retirement or Stopped. They +use real local runtimes/directory/SQLite with explicit reply faults. They do not +cover process crashes, a distributed journal or a complete role registry. + +The large reader pagination case opens 128 native views and checks all restored +values and joined ledgers. Its explicit fixture budgets are 128 slots, 256 MiB +of retained credit and four GiB of native memory credit. It holds 127 views +through at least 1,500 ms of real pressure samples and requires Normal admission +before the final Pending request. This avoids depending on a fast opening loop +to beat the classifier's 1,000-ms dwell. It renews both original signed boot +advertisements through directory CAS while opening and reading; its receiver +guard advances only after confirmation. Ordinary memory budgets and production +pressure thresholds and qualification profiles are unchanged. + +The public managed follower scenarios use two native `FollowerStore` instances, +signed directory authorization/CAS, and this same SQLite journal. They check all +members Pending before enrollment; partial acceptance and receiver cordon; lost +acceptance, establishment, close and retirement replies; shutdown deadlines and +cancelled waiters; cold retirement and actual acknowledged SQL command/root +readback; and invalid shipper limits before acceptance. They exercise +`CellNode::install_fleet_node_durability_provider` through the existing supervisor. +The fixture provider prepares signed immutable inputs only; its authority +fresh-loads exact scope and retains its checked close receipt across lost replies. +These in-process scenarios do not establish complete role coverage or sustained +replacement policy for role-enabled deployments. The executable's count +collector supports only its declared writer profile. + +`FleetEnrollmentJournal::refuse_unexecuted_enrollment` is one SQLite transaction: +validate complete inputs, refuse an exact Pending row or create an exact terminal +exclusion when absent, and advance RegistryVersion only on a real change. The +retained finite owner must prove native work never started (or retain the checked +original-token exclusion proof). Reader removal and follower refusal use it to +fence delayed acceptance; neither treats an absence read as settlement. Independent +clients exercise lost commit replies, reconstruction, and acceptance/refusal races. + +The journal's cooldown and post-batch queries compare the complete expected +snapshot and walk only its committed progress chain, one bounded page at a time. +Cancellation history and orphan pages cannot establish movement time. This +reference traversal is not a measured large-history performance result; a +production index must update atomically with the same retirement transaction. + +Fresh inspection records bind a nonce and original capture interval. Use the +host's `inspect_fleet_action` and validate its response against the complete +request and a finite age bound. Legacy retained Inspect results are historical. +The journal authorizes a capture; it does not supply current actor evidence. + +Before enabling fleet execution, the embedding application must provide the +controlled bootstrap/import barrier, wire every enrollment producer, authenticate +management requests, retain canonical enrollment proofs, produce complete fresh +observations, and supervise the public reconciliation driver. The reference +adapter does not infer these facts from an empty directory or a bootstrap flag. +See the [implementation evidence](../../../docs/fleet-operations-progress.md) +for exact source fingerprints and remaining work. + +The synthetic controller deadline cases use Tokio's test clock after fixture +setup to inject timeout at an observed endpoint boundary. They separately prove +retained permits before acceptance and after durable acceptance without a +result. The two-endpoint fault cases keep their 150 ms pass budget. The +single-cordon and unavailable-first-endpoint cases keep their 100 ms and 300 ms +budgets, advancing only after the original first acceptance. Journal rereads and +the healthy sibling finish on the paused clock without a fabricated second +fault. Original Elapsed sources and all phase/permit assertions remain required. +These cases measure protocol behavior; native/process latency qualification +uses its real clocks and committed profiles. + +## Recovered follower enrollment evidence + +The recovered closure and live replacement evidence have distinct requirements; +the durable live path is described [below](#durable-follower-replacement-evidence). + +The test target drives canonical recovery of a cold two-member ensemble, +native retirement and `FleetRecoveredFollowerRetirement` through this durable +journal. It checks Pending/Established original rows, lost publication replies, +independent-client reconstruction after native collection, unchanged replay +timestamps, duplicate requests, stale barriers, expired claimants and regressed +clocks. Cancellation joins accepted backend jobs before reconstruction; a delayed +new epoch request blocks the final roster barrier. Successful sibling writes +remain visible when complete closure fails. +The failed leader's boot remains an obligation. These cold-lane cases do not +qualify Cell suffix pinning, failed-process closure or complete maintenance. +See the [host publication recipe](../docs/lifecycle.md#publish-a-recovered-owners-follower-retirement). + +```sh +cargo test -p cellule-host --example fleet_operations --all-features --locked recovered_publication +``` + +## Failed boot process evidence + +On Unix, the test target combines the cold recovered ensemble with an actual +child lifetime. Its test provider retains request-bound termination evidence +only after child kill/wait joins, then rereads the same file after adapter/client +reconstruction. `FleetFailedBootRetirement` refuses a running original, +unretired leader logs, unsettled/duplicate requests, stale barriers, foreign +process evidence and regressed clocks. Lost retirement replies and failed or +changed final process confirmation retain the original committed row/errors; +a delayed original-session request prevents closure even after boot retirement. + +The child represents only a process lifetime; it does not run a CellNode or +external provider jobs. This is adapter-contract evidence, not the plan's +multi-process, storage-fault, workload, replacement or maintenance qualification. +Production adapters authenticate and retain the actual process and all its +accepted external-job evidence. See the +[host recipe](../docs/lifecycle.md#publish-an-original-failed-boots-closure). + +```sh +cargo test -p cellule-host --example fleet_operations --all-features --locked failed_boot_closure +``` + +## Failed reader lifetime evidence + +The `failed_reader_closure` cases use application-owned enrollments around two +real canonical SQLite readers on an actual receiver host. The application calls `prepare_source`, journals Pending, then opens through +`activate_source`. One original establishment +remains unresolved to exercise closure after a lost result. The fixture retains +the original lifetime witness only after host shutdown joins native work, +retained reader clones report local joining and reject queries, and resource +ledgers return to zero. The source retains readable acknowledged data. + +`FleetFailedReaderRetirement` requires the exact receiver boot, complete roster, +permanent canonical fencing and durable original process evidence. Tests cover +live receiver/source-failure refusal, Pending/Established history, foreign +process/payload/acceptance refusal, replay, independent adapter reconstruction, +lost replies, cancelled waiters, a suspended provider read, changed/failing final +evidence, stale barriers and monotonic collection expiry. Boot retirement stays +blocked until both original reader rows settle; process identity survives those +publications. The journal owner joins accepted work after waiter cancellation. + +This is native in-process lifetime evidence with no external-job workload. It +does not qualify OS crashes, provider termination, replacement redundancy or a +complete maintenance command. Production providers authenticate and retain the +actual original process and every accepted external job/producer. See the +[host recipe](../docs/lifecycle.md#publish-an-original-failed-receivers-reader-closure). + +```sh +cargo test -p cellule-host --example fleet_operations --all-features --locked failed_reader_closure +``` + +## Durable reader replacement evidence + +See also the [durable follower path](#durable-follower-replacement-evidence). + +The `reader_policy_publication` cases use the same three real leased nodes, +managed boot/reader producers, canonical SQL views, authenticated peer status +path and SQLite journal as live evacuation. They persist every original policy +manifest/page and latest pointer in one registry transaction, then recheck fresh +authority, policy, exact enrollment, selected signed boots and native prefixes. +Independent clients reload committed history after reconstruction. Exact replay +retains the original capture and cannot restore a superseded pointer. + +Cases cover lost commit replies and their original shared I/O source, cancelled +waiters, policy changes after commit, fresh zero-reader policy publication, +refresh during Closing, unavailable replacements, stale registry/corrupt pages +and authority change during a suspended native probe. Refresh reuses the exact +original retirement without another close. Codec cases cover all 10,000 readers, +complete page validation, duplicate boots/requests, prefix/time/shape faults, +operation boot adoption and malformed bounded envelopes. + +These are local native/transaction cases. They do not supply complete controller +settlement, affected-writer evidence, follower replacement policy, OS/process +faults or provider/load/mixed-version qualification. See the +[host contract](../docs/read-replicas.md#persist-and-revalidate-reader-replacement-coverage). + +```sh +cargo test -p cellule-host --example fleet_operations --all-features --locked reader_policy_publication +cargo test -p cellule-runtime --lib --all-features --locked reader_policy_ +``` + +## Durable follower replacement evidence + +The `follower_policy_publication` cases use four actual managed leased nodes, +native follower stores, an acknowledged SQLite Cell, the original durability +supervisor and the shared SQLite journal. They persist complete original and +replacement ensembles with immutable boot/request identities, policy revision, +object-covered retirement watermark and original timestamps. Reconstructed +clients recheck the current policy and canonical ensemble together with complete +source/receiver native traversal and all-category rechecks. The transport routes +fresh nonce-bound pages to the existing finite `fleet_snapshot` owner. + +Cases cover duplicate and lost replies, cancelled publication waiters, policy +CAS races and changes during publication, superseded-pointer replay, missing +history and stale barriers, regressing clocks, withdrawn receivers, original +leader shutdown, local fencing while the directory remains live, Closing and +deadline repair. A third native rotation exercises refresh after the original +local receipt is evicted. Installed epochs before their first append remain +valid through their actual native binding and producer; inactive directory +enrollment alone cannot prove installation. Fixture shutdown joins the original +owners and checks native resource ledgers. Five codec cases cover bounded +ensembles, malformed data and scope/history/policy/lifetime rules. + +These cases establish selected in-process live-owner behavior. Failed-owner +replacement policy still needs canonical recovery plus affected-Cell successors; +complete fleet observation, maintenance action orchestration, OS/process/provider, +mixed-binary and measured load qualification remain required. The public recipe +is in the [host lifecycle guide](../docs/lifecycle.md#persist-and-revalidate-follower-replacement-evidence). + +```sh +cargo test -p cellule-host --example fleet_operations --all-features --locked follower_policy_publication +cargo test -p cellule-runtime --lib --all-features --locked follower_policy_ +``` + +## Managed reader producer evidence + +The example's test target also exercises `install_fleet_reader_enrollment` with +real admitted SQLite readers and this journal. It checks Pending before native +opening, caller cancellation, lost acceptance/publication replies, joined +retirement, cancelled removal, a native VFS stall across a host deadline, +independent journal reconstruction and startup binding requirements. A typed +counter query checks the exact reader receipt. The installed periodic loop +also repairs original lost replies and cancelled removals without another hint +or shutdown, retains a still-owned opening, and retries temporary inventory +credit refusal. A fenced-node case confirms that the original never-started +request can receive its nonexecution exclusion while new activation and native +inventory remain fenced. A public host case drives reconciliation before lease +installation and confirms healthy startup and joined shutdown. These fixtures do not +establish complete observer coverage or replacement redundancy for the movement +commands. + +```sh +cargo test -p cellule-host --example fleet_operations --all-features --locked +``` + +The long-partition pagination case retains 128 source Cells and 128 native reader +views in one process. Provision OS descriptor headroom for both collections. +Colima's default 1,024 soft limit fails during source bootstrap in both the +parent and current code. The controlled Linux run passed with the existing +524,288 hard limit exposed as the soft limit, preserving two CPUs, four GiB, +all 110 cases and their original assertions. In a dedicated verification +container, inspect and raise the soft limit before invoking the target: + +```sh +ulimit -Sn +ulimit -Hn +ulimit -Sn "$(ulimit -Hn)" +cargo test -p cellule-host --example fleet_operations --all-features --locked +``` diff --git a/crates/cellule-host/minion/journal/actions.rs b/crates/cellule-host/minion/journal/actions.rs new file mode 100644 index 00000000..957bd455 --- /dev/null +++ b/crates/cellule-host/minion/journal/actions.rs @@ -0,0 +1,424 @@ +use super::*; + +impl Db<'_> { + fn accepted( + &self, + key: Digest, + node: NodeId, + session: SessionId, + ) -> JournalResult)>> { + let row = self + .tx + .query_row( + "SELECT accepted,result FROM actions WHERE key=?1 AND node=?2 AND session=?3", + params![ + key.as_bytes().as_slice(), + node.as_bytes().as_slice(), + session.as_bytes().as_slice() + ], + |row| { + let result = if matches!(row.get_ref(1)?, rusqlite::types::ValueRef::Null) { + None + } else { + Some(blob(row, 1, MAX_RECORD_BYTES)?) + }; + Ok((blob(row, 0, MAX_RECORD_BYTES)?, result)) + }, + ) + .optional()?; + row.map(|(accepted, result)| { + let accepted = AcceptedFleetAction::from_bytes(&accepted)?; + self.check_scope(accepted.action().scope())?; + if accepted.action().key()? != key + || accepted.node() != node + || accepted.session() != session + { + return Err(OperationError::Conflict.into()); + } + let result = result + .map(|bytes| FleetActionOutcome::from_bytes(&bytes)) + .transpose()?; + if let Some(result) = &result { + accepted.validate_result(result)?; + } + Ok((accepted, result)) + }) + .transpose() + } + fn original(&self, accepted: &AcceptedFleetAction) -> JournalResult<()> { + let (original, _) = self + .accepted( + accepted.action().key()?, + accepted.node(), + accepted.session(), + )? + .ok_or(OperationError::NotFound)?; + if original != *accepted { + return Err(OperationError::Conflict.into()); + } + Ok(()) + } + fn basis(&self, accepted: &AcceptedFleetAction, kind: u8) -> JournalResult>> { + self.original(accepted)?; + Ok(self + .tx + .query_row( + "SELECT body FROM bases WHERE key=?1 AND node=?2 AND session=?3 AND kind=?4", + params![ + accepted.action().key()?.as_bytes().as_slice(), + accepted.node().as_bytes().as_slice(), + accepted.session().as_bytes().as_slice(), + kind + ], + |row| blob(row, 0, MAX_RECORD_BYTES), + ) + .optional()?) + } + fn write_basis( + &self, + accepted: &AcceptedFleetAction, + kind: u8, + bytes: Vec, + ) -> JournalResult<()> { + self.original(accepted)?; + self.tx.execute( + "INSERT INTO bases(key,node,session,kind,body) VALUES (?1,?2,?3,?4,?5)", + params![ + accepted.action().key()?.as_bytes().as_slice(), + accepted.node().as_bytes().as_slice(), + accepted.session().as_bytes().as_slice(), + kind, + bytes + ], + )?; + Ok(()) + } + fn check_intents( + &self, + action: &FleetAction, + node: NodeId, + session: SessionId, + ) -> JournalResult<()> { + let local = self.required_intent(node)?; + if local.session() != session { + return Err(OperationError::Conflict.into()); + } + match action.kind() { + FleetActionKind::Maintenance { operation, .. } => { + if local.operation() != Some(operation.id()) + || local.revision() != operation.intent_revision() + || local.mode() == NodeMode::Active + { + return Err(OperationError::Conflict.into()); + } + } + FleetActionKind::Movement { + action: effect, + attempt, + } => { + let spec = attempt.spec(); + if matches!( + effect, + MovementAction::Prepare + | MovementAction::Release + | MovementAction::ReleaseMaintenance + | MovementAction::Activate + | MovementAction::Recover + ) { + let receiver = self.required_intent(spec.destination_node)?; + if receiver.session() != spec.destination || receiver.mode() != NodeMode::Active + { + return Err(OperationError::Conflict.into()); + } + } + if matches!( + effect, + MovementAction::Prepare + | MovementAction::Release + | MovementAction::ReleaseMaintenance + ) { + let source = self.required_intent(spec.source_node)?; + if source.session() != spec.source { + return Err(OperationError::Conflict.into()); + } + if source.mode() != NodeMode::Active + && self + .snapshot()? + .head() + .maintenance() + .is_none_or(|operation| { + source.operation() != Some(operation.id()) + || source.revision() != operation.intent_revision() + || source.session() != operation.session() + || spec.id.operation != operation.id() + || operation.phase() != MaintenancePhase::Evacuating + }) + { + return Err(OperationError::Conflict.into()); + } + } + } + } + Ok(()) + } +} + +impl FleetActionJournal for SqliteJournal { + fn authorize_snapshot<'a>( + &'a self, + request: &'a FleetSnapshotRequest, + now_ms: i64, + ) -> FleetAdapterFuture<'a, ()> { + let request = request.clone(); + Box::pin(self.run(move |db| { + db.check_scope(request.expected().head().scope())?; + let snapshot = db.snapshot()?; + let intent = db.required_intent(request.node())?; + request.authorize_against(&snapshot, &intent, now_ms)?; + Ok(()) + })) + } + + fn authorize_inspection<'a>( + &'a self, + request: &'a FleetInspectionRequest, + now_ms: i64, + ) -> FleetAdapterFuture<'a, ()> { + let request = request.clone(); + Box::pin(self.run(move |db| { + db.check_scope(request.action().scope())?; + let snapshot = db.snapshot()?; + request.authorize_against(snapshot.head(), snapshot.registry(), now_ms)?; + db.check_intents(request.action(), request.node(), request.session()) + })) + } + + fn accept_action<'a>( + &'a self, + action: &'a FleetAction, + node: NodeId, + session: SessionId, + now_ms: i64, + ) -> FleetAdapterFuture<'a, FleetActionAcceptance> { + let action = action.clone(); + Box::pin(self.run(move |db| { + db.check_scope(action.scope())?; + let key = action.key()?; + if let Some((accepted, result)) = db.accepted(key, node, session)? { + accepted.validate_replay(&action, node, session)?; + return Ok(FleetActionAcceptance::Existing { accepted, result: result.map(Box::new) }); + } + let accepted = AcceptedFleetAction::new(action.clone(), db.snapshot()?.head(), node, session, now_ms)?; + db.check_intents(&action, node, session)?; + let (operation, sequence, effect) = match action.kind() { + FleetActionKind::Movement { action, attempt } => (attempt.spec().id.operation, attempt.spec().id.sequence.to_be_bytes().to_vec(), *action as u8), + FleetActionKind::Maintenance { action, operation } => (operation.id(), vec![], *action as u8), + }; + db.tx.execute("INSERT INTO actions(key,node,session,operation,sequence,effect,accepted) VALUES (?1,?2,?3,?4,?5,?6,?7)", params![key.as_bytes().as_slice(), node.as_bytes().as_slice(), session.as_bytes().as_slice(), operation.as_bytes().as_slice(), sequence, effect, accepted.to_bytes()?])?; + Ok(FleetActionAcceptance::New(accepted)) + })) + } + fn publish_action_result<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + result: &'a FleetActionOutcome, + ) -> FleetAdapterFuture<'a, ()> { + let accepted = accepted.clone(); + let result = result.clone(); + Box::pin(self.run(move |db| { + db.original(&accepted)?; + accepted.validate_result(&result)?; + if let FleetActionKind::Movement { + action: MovementAction::Activate, + .. + } = accepted.action().kind() + && matches!(result.outcome, FleetOutcome::Activated(_)) + { + let basis = AcquisitionBasis::from_bytes( + &db.basis(&accepted, 1)?.ok_or(OperationError::NotFound)?, + )?; + basis.validate_result(&result)?; + } + if let FleetOutcome::Recovered(recovered) = &result.outcome + && let FleetActionKind::Movement { + action: MovementAction::Recover, + .. + } = accepted.action().kind() + { + let evidence = RecoveryEvidence::from_bytes( + &db.basis(&accepted, 3)?.ok_or(OperationError::NotFound)?, + )?; + if evidence != recovered.recovery { + return Err(OperationError::Conflict.into()); + } + } + let (_, previous) = db + .accepted( + accepted.action().key()?, + accepted.node(), + accepted.session(), + )? + .ok_or(OperationError::NotFound)?; + if let Some(previous) = previous { + if previous == result { + return Ok(()); + } + if !matches!(previous.outcome, FleetOutcome::Unknown) { + return Err(OperationError::Conflict.into()); + } + if matches!(result.outcome, FleetOutcome::Unknown) { + return Ok(()); + } + } + db.tx.execute( + "UPDATE actions SET result=?1 WHERE key=?2 AND node=?3 AND session=?4", + params![ + result.to_bytes()?, + accepted.action().key()?.as_bytes().as_slice(), + accepted.node().as_bytes().as_slice(), + accepted.session().as_bytes().as_slice() + ], + )?; + Ok(()) + })) + } + fn load_movement_action<'a>( + &'a self, + scope: FleetScope, + attempt: AttemptId, + effect: MovementAction, + node: NodeId, + session: SessionId, + ) -> FleetAdapterFuture<'a, Option> { + Box::pin(self.run(move |db| { + db.check_scope(scope)?; + let key = db.tx.query_row("SELECT key FROM actions WHERE operation=?1 AND sequence=?2 AND effect=?3 AND node=?4 AND session=?5", params![attempt.operation.as_bytes().as_slice(), attempt.sequence.to_be_bytes().as_slice(), effect as u8, node.as_bytes().as_slice(), session.as_bytes().as_slice()], |row| blob(row, 0, 32)).optional()?; + key.map(|bytes| { + let key = Digest::from_bytes(bytes.as_slice().try_into().map_err(|_| OperationError::Invalid("action index width"))?); + let (accepted, result) = db.accepted(key, node, session)?.ok_or(OperationError::NotFound)?; + match accepted.action().kind() { + FleetActionKind::Movement { action, attempt: original } if *action == effect && original.spec().id == attempt => {} + _ => return Err(OperationError::Conflict.into()), + } + Ok(FleetActionAcceptance::Existing { accepted, result: result.map(Box::new) }) + }).transpose() + })) + } + fn record_acquisition_basis<'a>( + &'a self, + basis: &'a AcquisitionBasis, + ) -> FleetAdapterFuture<'a, AcquisitionBasis> { + let basis = basis.clone(); + Box::pin(self.run(move |db| { + let bytes = basis.to_bytes()?; + if let Some(bytes) = db.basis(basis.accepted(), 1)? { + let original = AcquisitionBasis::from_bytes(&bytes)?; + if original.accepted() != basis.accepted() || original.control() != basis.control() + { + return Err(OperationError::Conflict.into()); + } + return Ok(original); + } + db.write_basis(basis.accepted(), 1, bytes)?; + Ok(basis) + })) + } + fn load_acquisition_basis<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + ) -> FleetAdapterFuture<'a, Option> { + let accepted = accepted.clone(); + Box::pin(self.run(move |db| { + db.basis(&accepted, 1)? + .map(|bytes| { + let basis = AcquisitionBasis::from_bytes(&bytes)?; + if basis.accepted() != &accepted { + return Err(OperationError::Conflict.into()); + } + Ok(basis) + }) + .transpose() + })) + } + fn record_recovery_basis<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + basis: &'a RecoveryBasis, + ) -> FleetAdapterFuture<'a, RecoveryBasis> { + let accepted = accepted.clone(); + let basis = basis.clone(); + Box::pin(self.run(move |db| { + basis.validate_acceptance(&accepted)?; + let bytes = basis.to_bytes()?; + if let Some(bytes) = db.basis(&accepted, 2)? { + let original = RecoveryBasis::from_bytes(&bytes)?; + original.validate_acceptance(&accepted)?; + if original.control() != basis.control() { + return Err(OperationError::Conflict.into()); + } + return Ok(original); + } + db.write_basis(&accepted, 2, bytes)?; + Ok(basis) + })) + } + fn load_recovery_basis<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + ) -> FleetAdapterFuture<'a, Option> { + let accepted = accepted.clone(); + Box::pin(self.run(move |db| { + db.basis(&accepted, 2)? + .map(|bytes| { + let basis = RecoveryBasis::from_bytes(&bytes)?; + basis.validate_acceptance(&accepted)?; + Ok(basis) + }) + .transpose() + })) + } + fn record_recovery_evidence<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + evidence: &'a RecoveryEvidence, + ) -> FleetAdapterFuture<'a, RecoveryEvidence> { + let accepted = accepted.clone(); + let evidence = evidence.clone(); + Box::pin(self.run(move |db| { + evidence.basis().validate_acceptance(&accepted)?; + let basis = RecoveryBasis::from_bytes( + &db.basis(&accepted, 2)?.ok_or(OperationError::NotFound)?, + )?; + if &basis != evidence.basis() { + return Err(OperationError::Conflict.into()); + } + let bytes = evidence.to_bytes()?; + if let Some(bytes) = db.basis(&accepted, 3)? { + let original = RecoveryEvidence::from_bytes(&bytes)?; + if original.basis() != evidence.basis() + || original.restored() != evidence.restored() + { + return Err(OperationError::Conflict.into()); + } + return Ok(original); + } + db.write_basis(&accepted, 3, bytes)?; + Ok(evidence) + })) + } + fn load_recovery_evidence<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + ) -> FleetAdapterFuture<'a, Option> { + let accepted = accepted.clone(); + Box::pin(self.run(move |db| { + db.basis(&accepted, 3)? + .map(|bytes| { + let evidence = RecoveryEvidence::from_bytes(&bytes)?; + evidence.basis().validate_acceptance(&accepted)?; + Ok(evidence) + }) + .transpose() + })) + } +} diff --git a/crates/cellule-host/minion/journal/controller.rs b/crates/cellule-host/minion/journal/controller.rs new file mode 100644 index 00000000..0b79ab6d --- /dev/null +++ b/crates/cellule-host/minion/journal/controller.rs @@ -0,0 +1,199 @@ +use super::*; + +impl FleetJournal for SqliteJournal { + fn load_snapshot(&self, scope: FleetScope) -> FleetAdapterFuture<'_, FleetJournalSnapshot> { + Box::pin(self.run(move |db| { + db.check_scope(scope)?; + db.snapshot() + })) + } + fn claim_controller( + &self, + scope: FleetScope, + expected_revision: u64, + claimant: SessionId, + now_ms: i64, + ) -> FleetAdapterFuture<'_, FleetJournalSnapshot> { + Box::pin(self.run(move |db| { + db.check_scope(scope)?; + let current = db.snapshot()?; + let head = current + .head() + .claim(db.profile, expected_revision, claimant, now_ms)?; + db.set_head(&head)?; + db.snapshot() + })) + } + fn compare_exchange<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + epoch: u64, + now_ms: i64, + transition: &'a JournalTransition, + ) -> FleetAdapterFuture<'a, FleetJournalSnapshot> { + let expected = expected.clone(); + let transition = transition.clone(); + Box::pin(self.run(move |db| { + db.check_scope(expected.head().scope())?; + let current = db.snapshot()?; + if current != expected { return Err(OperationError::Conflict.into()); } + if let JournalTransition::ResolveUnaccepted { id, effect } = &transition { + let attempt = current.head().attempts().iter().find(|attempt| attempt.spec().id == *id).ok_or(OperationError::NotFound)?; + let spec = attempt.spec(); + let (node, session) = if effect.is_source_release() { (spec.source_node, spec.source) } else { (spec.destination_node, spec.destination) }; + let accepted = db.tx.query_row("SELECT EXISTS(SELECT 1 FROM actions WHERE operation=?1 AND sequence=?2 AND effect=?3 AND node=?4 AND session=?5)", + params![id.operation.as_bytes().as_slice(), id.sequence.to_be_bytes().as_slice(), *effect as u8, node.as_bytes().as_slice(), session.as_bytes().as_slice()], |row| row.get::<_, bool>(0))?; + if accepted { return Err(OperationError::Busy.into()); } + } + if let JournalTransition::Allocate(spec) = &transition { + current.registry().authorize_allocation(current.head(), spec, &db.required_intent(spec.source_node)?, &db.required_intent(spec.destination_node)?)?; + } + if let JournalTransition::BeginMaintenance(request) = &transition + && let Some(original) = db.tx.query_row("SELECT request FROM operations WHERE key=?1", [request.id().as_bytes().as_slice()], |row| blob(row, 0, MAX_RECORD_BYTES)).optional()? { + let original = MaintenanceOperation::from_bytes(&original)?; + if original != *request { return Err(OperationError::Conflict.into()); } + let lease = current.head().controller().ok_or(OperationError::Fenced)?; + if epoch != lease.epoch || now_ms >= lease.expires_at_ms { return Err(OperationError::Fenced.into()); } + // The canonical calculation checks revision and monotonic + // time; its successor is not published for an idempotent request. + current.head().claim(db.profile, current.head().revision(), lease.claimant, now_ms)?; + // The original request is retained independently of mutable + // deadline/session progress. Replays cannot start it again. + return Ok(current); + } + let head = current.head().transition(db.profile, current.head().revision(), epoch, now_ms, transition.clone())?; + if head == *current.head() { return Ok(current); } + if head.maintenance() != current.head().maintenance() && let Some(operation) = head.maintenance() { + let old = db.required_intent(operation.node())?; + let next = old.advance_maintenance(operation)?; + if next != old { db.write_intent(&next)?; } + let request = if let JournalTransition::BeginMaintenance(request) = &transition { request.to_bytes()? } + else { db.tx.query_row("SELECT request FROM operations WHERE key=?1", [operation.id().as_bytes().as_slice()], |row| blob(row, 0, MAX_RECORD_BYTES))? }; + db.tx.execute("INSERT INTO operations(key,request,body) VALUES (?1,?2,?3) ON CONFLICT(key) DO UPDATE SET body=excluded.body", params![operation.id().as_bytes().as_slice(), request, operation.to_bytes()?])?; + } + if let JournalTransition::Retire { progress } = &transition { + db.tx.execute("INSERT INTO progress(key,body) VALUES (?1,?2)", params![progress.digest()?.as_bytes().as_slice(), progress.to_bytes()?])?; + } + db.set_head(&head)?; + db.snapshot() + })) + } + fn load_operation( + &self, + scope: FleetScope, + operation: OperationId, + ) -> FleetAdapterFuture<'_, Option> { + Box::pin(self.run(move |db| { + db.check_scope(scope)?; + db.operation(operation) + })) + } + fn load_progress( + &self, + scope: FleetScope, + digest: Digest, + ) -> FleetAdapterFuture<'_, Option> { + Box::pin(self.run(move |db| { + db.check_scope(scope)?; + db.progress(digest) + })) + } + fn last_moved_at<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + cell: cellule_runtime::identity::CellId, + incarnation: cellule_runtime::identity::IncarnationId, + ) -> FleetAdapterFuture<'a, Option> { + let expected = expected.clone(); + Box::pin(self.run(move |db| db.movement_time(&expected, Some((cell, incarnation))))) + } + fn last_movement_at<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + ) -> FleetAdapterFuture<'a, Option> { + let expected = expected.clone(); + Box::pin(self.run(move |db| db.movement_time(&expected, None))) + } + fn intents_page( + &self, + version: RegistryVersion, + after: Option, + limit: usize, + ) -> FleetAdapterFuture<'_, IntentPage> { + Box::pin(self.run(move |db| { + page_limit(limit)?; + db.check_version(version)?; + if let Some(after) = after { + db.required_intent(after)?; + } + let mut statement = db.tx.prepare( + "SELECT key,body FROM intents WHERE (?1 IS NULL OR key>?1) ORDER BY key LIMIT ?2", + )?; + let mut rows = statement.query(params![ + after.map(|key| key.as_bytes().to_vec()), + i64::try_from(limit + 1)? + ])?; + let mut entries = Vec::with_capacity(limit); + while entries.len() < limit { + let Some(row) = rows.next()? else { + break; + }; + let key = blob(row, 0, 16)?; + let intent = NodeIntent::from_bytes(&blob(row, 1, MAX_RECORD_BYTES)?)?; + if intent.node().as_bytes().as_slice() != key { + return Err(OperationError::Conflict.into()); + } + entries.push(intent); + } + let next = if rows.next()?.is_some() { + entries.last().map(NodeIntent::node) + } else { + None + }; + Ok(IntentPage::new(version, after, entries, next)?) + })) + } + fn enrollments_page( + &self, + version: RegistryVersion, + after: Option, + limit: usize, + ) -> FleetAdapterFuture<'_, EnrollmentPage> { + Box::pin(self.run(move |db| { + page_limit(limit)?; + db.check_version(version)?; + if let Some(after) = after && db.enrollment(after)?.is_none() { return Err(OperationError::NotFound.into()); } + let mut statement = db.tx.prepare("SELECT key,body FROM enrollments WHERE (?1 IS NULL OR key>?1) ORDER BY key LIMIT ?2")?; + let mut rows = statement.query(params![after.map(|key| key.as_bytes().to_vec()), i64::try_from(limit + 1)?])?; + let mut entries = Vec::with_capacity(limit); + while entries.len() < limit { + let Some(row) = rows.next()? else { break; }; + let key = blob(row, 0, 32)?; + let record = EnrollmentRecord::from_bytes(&blob(row, 1, MAX_RECORD_BYTES)?)?; + if record.spec().key()?.as_bytes().as_slice() != key { return Err(OperationError::Conflict.into()); } + entries.push(record); + } + let next = if rows.next()?.is_some() { entries.last().map(|record| record.spec().key()).transpose()? } else { None }; + Ok(EnrollmentPage::new(version, after, entries, next)?) + })) + } + fn set_scheduling( + &self, + expected: RegistryVersion, + enabled: bool, + ) -> FleetAdapterFuture<'_, RegistryVersion> { + Box::pin(self.run(move |db| { + db.check_version(expected)?; + let next = expected.set_scheduling(expected.revision(), enabled)?; + db.set_registry(next)?; + Ok(next) + })) + } +} + +fn page_limit(limit: usize) -> JournalResult<()> { + if limit == 0 || limit > MAX_PAGE_ENTRIES { + return Err(OperationError::Invalid("invalid registry page limit").into()); + } + Ok(()) +} diff --git a/crates/cellule-host/minion/journal/enrollment.rs b/crates/cellule-host/minion/journal/enrollment.rs new file mode 100644 index 00000000..2576ae07 --- /dev/null +++ b/crates/cellule-host/minion/journal/enrollment.rs @@ -0,0 +1,208 @@ +use super::*; + +impl FleetEnrollmentJournal for SqliteJournal { + fn refuse_unexecuted_enrollment<'a>( + &'a self, + spec: &'a EnrollmentSpec, + evidence: Digest, + now_ms: i64, + ) -> FleetAdapterFuture<'a, EnrollmentRecord> { + let spec = spec.clone(); + Box::pin(self.run(move |db| { + db.check_scope(spec.scope)?; + let original = db.enrollment(spec.key()?)?; + let next = match &original { + Some(original) => { + original.validate_replay(&spec)?; + original.refuse(evidence, now_ms)? + } + None => EnrollmentRecord::unexecuted_refusal(spec, evidence, now_ms)?, + }; + if original.as_ref() != Some(&next) { + db.write_enrollment(&next)?; + } + Ok(next) + })) + } + + fn load_boot( + &self, + scope: FleetScope, + node: NodeId, + key: Digest, + ) -> FleetAdapterFuture<'_, Option> { + Box::pin(async move { + let observed = self + .run(move |db| { + db.check_scope(scope)?; + let (Some(intent), Some(enrollment)) = (db.intent(node)?, db.enrollment(key)?) + else { + return Ok(None); + }; + Ok(Some(cellule_host::fleet::FleetBootObservation::new( + intent, enrollment, + )?)) + }) + .await?; + #[cfg(test)] + { + // Hold only the reply after the read transaction has joined. + // Another client can advance intent without a locked database. + let pause = self.inner.boot_reply.lock().unwrap().take(); + if let Some(pause) = pause { + let _ = pause.captured.send(()); + let _ = pause.resume.await; + } + } + Ok(observed) + }) + } + fn register_initial_intent<'a>( + &'a self, + intent: &'a NodeIntent, + ) -> FleetAdapterFuture<'a, NodeIntent> { + let intent = intent.clone(); + Box::pin(self.run(move |db| { + db.check_scope(intent.scope())?; + if NodeIntent::initial(intent.scope(), intent.node(), intent.session())? != intent { + return Err(OperationError::Conflict.into()); + } + if let Some(existing) = db.intent(intent.node())? { + if existing != intent { + return Err(OperationError::Conflict.into()); + } + return Ok(existing); + } + db.write_intent(&intent)?; + Ok(intent) + })) + } + fn rebind_active_intent<'a>( + &'a self, + original: &'a NodeIntent, + session: SessionId, + revision: u64, + ) -> FleetAdapterFuture<'a, NodeIntent> { + let original = original.clone(); + Box::pin(self.run(move |db| { + db.check_scope(original.scope())?; + let next = original.rebind_active(session, revision)?; + let current = db.required_intent(original.node())?; + if current == next { + return Ok(current); + } + if current != original { + return Err(OperationError::Conflict.into()); + } + db.write_intent(&next)?; + Ok(next) + })) + } + fn return_to_service( + &self, + scope: FleetScope, + node: NodeId, + operation: OperationId, + session: SessionId, + revision: u64, + ) -> FleetAdapterFuture<'_, NodeIntent> { + Box::pin(self.run(move |db| { + db.check_scope(scope)?; + let completed = db.operation(operation)?.ok_or(OperationError::NotFound)?; + let old = db.required_intent(node)?; + let predecessor = NodeIntent::maintenance(scope, &completed)?; + let next = predecessor.return_to_service(&completed, session, revision)?; + if old == next { + return Ok(old); + } + if old != predecessor { + return Err(OperationError::Conflict.into()); + } + db.write_intent(&next)?; + Ok(next) + })) + } + fn bootstrap_registry( + &self, + expected: RegistryVersion, + ) -> FleetAdapterFuture<'_, RegistryVersion> { + Box::pin(self.run(move |db| { + db.check_version(expected)?; + let next = expected.bootstrap(expected.revision())?; + db.set_registry(next)?; + Ok(next) + })) + } + fn accept_enrollment<'a>( + &'a self, + spec: &'a EnrollmentSpec, + now_ms: i64, + ) -> FleetAdapterFuture<'a, FleetEnrollmentAcceptance> { + let spec = spec.clone(); + Box::pin(async move { + #[cfg(test)] + self.enrollment_reply(false, true).await?; + let result = self + .run(move |db| { + db.check_scope(spec.scope)?; + if let Some(original) = db.enrollment(spec.key()?)? { + original.validate_replay(&spec)?; + return Ok(FleetEnrollmentAcceptance::Existing(original)); + } + let source = spec + .source + .map(|endpoint| db.required_intent(endpoint.node)) + .transpose()?; + let target = db.required_intent(spec.target.node)?; + let pending = + EnrollmentRecord::pending(spec, source.as_ref(), &target, now_ms)?; + db.write_enrollment(&pending)?; + Ok(FleetEnrollmentAcceptance::New(pending)) + }) + .await?; + #[cfg(test)] + self.enrollment_reply(false, false).await?; + Ok(result) + }) + } + fn publish_enrollment_result<'a>( + &'a self, + original: &'a EnrollmentRecord, + event: EnrollmentEvent, + now_ms: i64, + ) -> FleetAdapterFuture<'a, EnrollmentRecord> { + let original = original.clone(); + Box::pin(async move { + let result = self + .run(move |db| { + db.check_scope(original.spec().scope)?; + let current = db + .enrollment(original.spec().key()?)? + .ok_or(OperationError::NotFound)?; + current.validate_replay(original.spec())?; + if current.accepted_at_ms() != original.accepted_at_ms() { + return Err(OperationError::Conflict.into()); + } + let next = current.apply(event, now_ms)?; + if next != current { + db.write_enrollment(&next)?; + } + Ok(next) + }) + .await?; + #[cfg(test)] + self.enrollment_reply(true, false).await?; + Ok(result) + }) + } + fn load_enrollment( + &self, + scope: FleetScope, + key: Digest, + ) -> FleetAdapterFuture<'_, Option> { + Box::pin(self.run(move |db| { + db.check_scope(scope)?; + db.enrollment(key) + })) + } +} diff --git a/crates/cellule-host/minion/journal/follower_evacuation.rs b/crates/cellule-host/minion/journal/follower_evacuation.rs new file mode 100644 index 00000000..ad873056 --- /dev/null +++ b/crates/cellule-host/minion/journal/follower_evacuation.rs @@ -0,0 +1,203 @@ +use super::*; + +impl FleetFollowerEvacuationJournal for SqliteJournal { + fn follower_replacement_policy<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + ) -> FleetAdapterFuture<'a, Option> { + let expected = expected.clone(); + Box::pin(self.run(move |db| { + db.check_scope(expected.head().scope())?; + if db.snapshot()? != expected { + return Err(OperationError::Conflict.into()); + } + db.follower_policy() + })) + } + fn set_follower_replacement_policy<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + policy: FollowerReplacementPolicy, + now_ms: i64, + ) -> FleetAdapterFuture<'a, FollowerReplacementPolicy> { + let expected = expected.clone(); + Box::pin(self.run(move |db| { + db.check_scope(policy.scope())?;policy.to_bytes()?; + let snapshot=db.snapshot()?; + if snapshot!=expected || snapshot.registry().bootstrap_revision().is_none() + || snapshot.head().controller().is_none_or(|lease| now_ms>=lease.expires_at_ms) + || now_ms<0 { + return Err(OperationError::Conflict.into()); + } + let expected_revision=db.follower_policy()?.map_or(Some(1),|old| old.revision().checked_add(1)).ok_or(OperationError::Conflict)?; + if policy.revision()!=expected_revision {return Err(OperationError::Conflict.into());} + db.tx.execute("INSERT INTO follower_policy(singleton,body) VALUES(1,?1) ON CONFLICT(singleton) DO UPDATE SET body=excluded.body",[policy.to_bytes()?])?; + db.advance_registry()?; + Ok(policy) + })) + } + fn persist_follower_evacuation<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + record: &'a FollowerEvacuationRecord, + now_ms: i64, + ) -> FleetAdapterFuture<'a, FollowerEvacuationRecord> { + Box::pin(async move { + record.to_bytes()?; + let expected = expected.clone(); + let record = record.clone(); + let result=self.run(move |db| { + db.check_scope(record.policy().scope())?; + let digest=record.digest()?; + if let Some(original)=db.follower_evacuation(digest)? { + if original!=record {return Err(OperationError::Conflict.into());} + // Historical replay cannot restore a superseded latest pointer. + return Ok(original); + } + let snapshot=db.snapshot()?;let operation=record.operation();let (started,finished)=record.interval(); + if snapshot!=expected || record.registry()!=expected.registry() + || record.head_digest()!=Digest::from_bytes(*blake3::hash(&expected.head().to_bytes()?).as_bytes()) + || snapshot.registry().bootstrap_revision().is_none() + || snapshot.head().maintenance()!=Some(operation) + || !matches!(operation.phase(),MaintenancePhase::Evacuating | MaintenancePhase::Closing) + || snapshot.head().controller().is_none_or(|lease| now_ms>=lease.expires_at_ms) + || now_ms>=operation.deadline_ms() || now_ms30_000 + || db.follower_policy()?!=Some(record.policy()) { + return Err(OperationError::Conflict.into()); + } + let donor=db.required_intent(operation.node())?; + if donor.session()!=operation.session() || donor.revision()!=operation.intent_revision() || donor.mode()!=NodeMode::Draining { + return Err(OperationError::Conflict.into()); + } + for row in record.retired() { + if db.enrollment(row.spec().key()?)?.as_ref()!=Some(row) {return Err(OperationError::Conflict.into());} + } + let source=record.source()?;let source_intent=db.required_intent(source.node)?; + if db.follower_rows(source.node,source.session,record.original_epoch()?)?!=record.retired() + || db.follower_rows(source.node,source.session,record.replacement_epoch())? + !=record.replacements().iter().map(|entry| entry.enrollment.clone()).collect::>() { + return Err(OperationError::Conflict.into()); + } + if source_intent.session()!=source.session || source_intent.mode()!=NodeMode::Active {return Err(OperationError::Conflict.into());} + for entry in record.replacements() { + let row=&entry.enrollment;let target=row.spec().target;let intent=db.required_intent(target.node)?; + if db.enrollment(row.spec().key()?)?.as_ref()!=Some(row) + || intent.session()!=target.session || intent.revision()!=target.intent_revision || intent.mode()!=NodeMode::Active + || row.spec().source.is_none_or(|endpoint| endpoint.intent_revision!=source_intent.revision()) { + return Err(OperationError::Conflict.into()); + } + } + db.tx.execute("INSERT INTO follower_evacuations(key,body) VALUES(?1,?2)",params![digest.as_bytes().as_slice(),record.to_bytes()?])?; + db.tx.execute("INSERT INTO latest_follower_evacuations(operation,original,witness) VALUES(?1,?2,?3) ON CONFLICT(operation,original) DO UPDATE SET witness=excluded.witness", + params![operation.id().as_bytes().as_slice(),record.original_key().as_bytes().as_slice(),digest.as_bytes().as_slice()])?; + db.advance_registry()?; + Ok(record) + }).await?; + #[cfg(test)] + self.follower_evacuation_reply().await?; + Ok(result) + }) + } + fn load_follower_evacuation( + &self, + scope: FleetScope, + digest: Digest, + ) -> FleetAdapterFuture<'_, Option> { + Box::pin(self.run(move |db| { + db.check_scope(scope)?; + db.follower_evacuation(digest) + })) + } + fn latest_follower_evacuation<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + operation: OperationId, + original: Digest, + ) -> FleetAdapterFuture<'a, Option> { + let expected = expected.clone(); + Box::pin(self.run(move |db| { + db.check_scope(expected.head().scope())?; + if db.snapshot()?!=expected {return Err(OperationError::Conflict.into());} + let key=db.tx.query_row("SELECT witness FROM latest_follower_evacuations WHERE operation=?1 AND original=?2", + params![operation.as_bytes().as_slice(),original.as_bytes().as_slice()],|row| blob(row,0,32)).optional()?; + key.map(|key| -> JournalResult<_> { + let record=db.follower_evacuation(Digest::from_bytes(key.as_slice().try_into()?))?.ok_or(OperationError::NotFound)?; + if record.operation().id()!=operation || record.original_key()!=original {return Err(OperationError::Conflict.into());} + Ok(record) + }).transpose() + })) + } +} +impl Db<'_> { + fn follower_rows( + &self, + node: NodeId, + session: SessionId, + epoch: u64, + ) -> JournalResult> { + let mut statement = self + .tx + .prepare("SELECT key,body FROM enrollments ORDER BY key")?; + let rows = statement.query_map([], |row| { + Ok((blob(row, 0, 32)?, blob(row, 1, MAX_RECORD_BYTES)?)) + })?; + let mut matches = Vec::new(); + for row in rows { + let (key, body) = row?; + let record = EnrollmentRecord::from_bytes(&body)?; + self.check_scope(record.spec().scope)?; + if record.spec().key()?.as_bytes().as_slice() != key { + return Err(OperationError::Conflict.into()); + } + if record + .spec() + .source + .is_some_and(|source| source.node == node && source.session == session) + && record.spec().role == (EnrollmentRole::Follower { log_epoch: epoch }) + { + if matches.len() == 2 { + return Err(OperationError::Conflict.into()); + } + matches.push(record); + } + } + matches.sort_by_key(|row| *row.spec().target.node.as_bytes()); + Ok(matches) + } + fn follower_policy(&self) -> JournalResult> { + self.tx + .query_row( + "SELECT body FROM follower_policy WHERE singleton=1", + [], + |row| blob(row, 0, MAX_RECORD_BYTES), + ) + .optional()? + .map(|bytes| -> JournalResult<_> { + let policy = FollowerReplacementPolicy::from_bytes(&bytes)?; + self.check_scope(policy.scope())?; + Ok(policy) + }) + .transpose() + } + fn follower_evacuation( + &self, + digest: Digest, + ) -> JournalResult> { + self.tx + .query_row( + "SELECT body FROM follower_evacuations WHERE key=?1", + [digest.as_bytes().as_slice()], + |row| blob(row, 0, MAX_PAGE_BYTES), + ) + .optional()? + .map(|bytes| -> JournalResult<_> { + let record = FollowerEvacuationRecord::from_bytes(&bytes)?; + self.check_scope(record.policy().scope())?; + if record.digest()? != digest { + return Err(OperationError::Conflict.into()); + } + Ok(record) + }) + .transpose() + } +} diff --git a/crates/cellule-host/minion/journal/mod.rs b/crates/cellule-host/minion/journal/mod.rs new file mode 100644 index 00000000..9d2e29ae --- /dev/null +++ b/crates/cellule-host/minion/journal/mod.rs @@ -0,0 +1,441 @@ +//! Local durable reference adapter. One SQLite file is one journal scope. +//! This proves local transactions, not a distributed provider's linearizability. + +use std::path::PathBuf; +use std::sync::{Arc, Mutex}; +use std::time::Duration; + +use cellule_host::fleet::*; +use cellule_runtime::fleet::operations::*; +use cellule_runtime::identity::{Digest, NodeId, SessionId}; +use cellule_runtime::node::NodeMode; +use rusqlite::{Connection, OptionalExtension, Transaction, TransactionBehavior, params}; +use tokio::sync::{Notify, Semaphore}; + +mod actions; +mod controller; +mod enrollment; +mod follower_evacuation; +mod reader_evacuation; +mod records; +mod writer_inventory; +use records::{Db, blob, profile_bytes, scope_bytes}; + +#[cfg(test)] +mod tests; + +pub type JournalError = Box; +pub type JournalResult = std::result::Result; + +const MAX_PENDING_JOBS: usize = 32; +const FORMAT: i64 = 1; + +struct Inner { + connection: Mutex>, + activity: Mutex, + changed: Notify, + slots: Arc, + scope: FleetScope, + profile: FleetProfile, + #[cfg(test)] + lose_commit_reply: std::sync::atomic::AtomicBool, + #[cfg(test)] + boot_reply: Mutex>, + #[cfg(test)] + enrollment_reply: Mutex>, + #[cfg(test)] + reader_evacuation_reply: Mutex>, + #[cfg(test)] + follower_evacuation_reply: Mutex>, + #[cfg(test)] + original_writer_reply: Mutex>, +} + +#[cfg(test)] +struct BootReplyPause { + captured: tokio::sync::oneshot::Sender<()>, + resume: tokio::sync::oneshot::Receiver<()>, +} + +#[cfg(test)] +struct EnrollmentReplyPause { + before_acceptance: bool, + publication: bool, + lose_reply: bool, + captured: tokio::sync::oneshot::Sender<()>, + resume: tokio::sync::oneshot::Receiver<()>, +} + +struct Activity { + accepting: bool, + pending: usize, +} + +/// Independently reconstructable client of the example's shared SQLite file. +/// Each call owns its bounded blocking job through commit, even if its waiter +/// disappears. `close` stops admission and joins all accepted jobs first. +#[derive(Clone)] +pub struct SqliteJournal { + inner: Arc, +} + +struct ActiveJob { + inner: Arc, + permit: Option, +} +impl Drop for ActiveJob { + fn drop(&mut self) { + // A poisoned supervisor closes future admission, but still settles the + // count so shutdown cannot strand a successfully accepted blocking job. + drop(self.permit.take()); + let mut state = self + .inner + .activity + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + state.pending = state.pending.saturating_sub(1); + self.inner.changed.notify_waiters(); + } +} + +impl SqliteJournal { + pub async fn open( + path: PathBuf, + scope: FleetScope, + profile: FleetProfile, + now_ms: i64, + ) -> JournalResult { + let profile = profile.validate()?; + let head = FleetHead::new(scope, now_ms)?.to_bytes()?; + let registry = RegistryVersion::new(scope)?.to_bytes()?; + let connection = tokio::runtime::Handle::try_current()?.spawn_blocking(move || -> JournalResult { + let mut connection = Connection::open(path)?; + connection.busy_timeout(Duration::from_secs(5))?; + connection.execute_batch("PRAGMA journal_mode=WAL; PRAGMA synchronous=FULL; PRAGMA foreign_keys=ON;")?; + let tx = connection.transaction_with_behavior(TransactionBehavior::Immediate)?; + tx.execute_batch(include_str!("schema.sql"))?; + tx.execute("INSERT OR IGNORE INTO state (singleton, format, scope, profile, head, registry) VALUES (1, ?1, ?2, ?3, ?4, ?5)", params![FORMAT, scope_bytes(scope), profile_bytes(profile)?, head, registry])?; + let (format, stored_scope, stored_profile) = tx.query_row("SELECT format, scope, profile FROM state WHERE singleton=1", [], |row| Ok((row.get::<_, i64>(0)?, blob(row, 1, 48)?, blob(row, 2, 32)?)))?; + if format != FORMAT || stored_scope != scope_bytes(scope) || stored_profile != profile_bytes(profile)? { + return Err(OperationError::Conflict.into()); + } + Db { tx: &tx, scope, profile }.snapshot()?; + tx.commit()?; + Ok(connection) + }).await??; + Ok(Self { + inner: Arc::new(Inner { + connection: Mutex::new(Some(connection)), + activity: Mutex::new(Activity { + accepting: true, + pending: 0, + }), + changed: Notify::new(), + slots: Arc::new(Semaphore::new(MAX_PENDING_JOBS)), + scope, + profile, + #[cfg(test)] + lose_commit_reply: std::sync::atomic::AtomicBool::new(false), + #[cfg(test)] + boot_reply: Mutex::new(None), + #[cfg(test)] + enrollment_reply: Mutex::new(None), + #[cfg(test)] + reader_evacuation_reply: Mutex::new(None), + #[cfg(test)] + follower_evacuation_reply: Mutex::new(None), + #[cfg(test)] + original_writer_reply: Mutex::new(None), + }), + }) + } + + async fn run( + &self, + work: impl FnOnce(&Db<'_>) -> JournalResult + Send + 'static, + ) -> JournalResult { + let runtime = tokio::runtime::Handle::try_current()?; + let active = { + let mut state = self + .inner + .activity + .lock() + .map_err(|_| std::io::Error::other("journal supervisor lock poisoned"))?; + if !state.accepting { + return Err(cellule_runtime::Error::RuntimeClosed.into()); + } + let permit = Arc::clone(&self.inner.slots) + .try_acquire_owned() + .map_err(|_| cellule_runtime::Error::Capacity("reference journal jobs"))?; + state.pending += 1; + ActiveJob { + inner: Arc::clone(&self.inner), + permit: Some(permit), + } + }; + let inner = Arc::clone(&self.inner); + runtime + .spawn_blocking(move || { + let _active = active; + let mut guard = inner + .connection + .lock() + .map_err(|_| std::io::Error::other("journal connection lock poisoned"))?; + let connection = guard + .as_mut() + .ok_or(cellule_runtime::Error::RuntimeClosed)?; + // Acceptance, intent checks and evidence publication all use this + // transaction boundary. Never release it between validation/write. + let tx = connection.transaction_with_behavior(TransactionBehavior::Immediate)?; + let value = work(&Db { + tx: &tx, + scope: inner.scope, + profile: inner.profile, + })?; + tx.commit()?; + #[cfg(test)] + if inner + .lose_commit_reply + .swap(false, std::sync::atomic::Ordering::SeqCst) + { + return Err( + std::io::Error::other("injected lost reply after durable commit").into(), + ); + } + Ok(value) + }) + .await? + } + + #[cfg(test)] + pub(crate) fn pause_next_boot_reply( + &self, + ) -> ( + tokio::sync::oneshot::Receiver<()>, + tokio::sync::oneshot::Sender<()>, + ) { + let (captured, observed) = tokio::sync::oneshot::channel(); + let (resume, paused) = tokio::sync::oneshot::channel(); + let mut slot = self.inner.boot_reply.lock().unwrap(); + assert!(slot.is_none()); + *slot = Some(BootReplyPause { + captured, + resume: paused, + }); + (observed, resume) + } + + #[cfg(test)] + pub(crate) fn pause_next_original_writer_reply( + &self, + ) -> ( + tokio::sync::oneshot::Receiver<()>, + tokio::sync::oneshot::Sender<()>, + ) { + let (captured, observed) = tokio::sync::oneshot::channel(); + let (resume, paused) = tokio::sync::oneshot::channel(); + let mut slot = self.inner.original_writer_reply.lock().unwrap(); + assert!(slot.is_none()); + *slot = Some(BootReplyPause { + captured, + resume: paused, + }); + (observed, resume) + } + + #[cfg(test)] + async fn original_writer_reply(&self) { + let pause = self.inner.original_writer_reply.lock().unwrap().take(); + if let Some(pause) = pause { + let _ = pause.captured.send(()); + let _ = pause.resume.await; + } + } + + #[cfg(test)] + pub(crate) fn pause_next_enrollment_reply( + &self, + publication: bool, + lose_reply: bool, + ) -> ( + tokio::sync::oneshot::Receiver<()>, + tokio::sync::oneshot::Sender<()>, + ) { + let (captured, observed) = tokio::sync::oneshot::channel(); + let (resume, paused) = tokio::sync::oneshot::channel(); + let mut slot = self.inner.enrollment_reply.lock().unwrap(); + assert!(slot.is_none()); + *slot = Some(EnrollmentReplyPause { + before_acceptance: false, + publication, + lose_reply, + captured, + resume: paused, + }); + (observed, resume) + } + + #[cfg(test)] + pub(crate) fn pause_next_reader_evacuation_reply( + &self, + lose_reply: bool, + ) -> ( + tokio::sync::oneshot::Receiver<()>, + tokio::sync::oneshot::Sender<()>, + ) { + let (captured, observed) = tokio::sync::oneshot::channel(); + let (resume, paused) = tokio::sync::oneshot::channel(); + let mut slot = self.inner.reader_evacuation_reply.lock().unwrap(); + assert!(slot.is_none()); + *slot = Some(EnrollmentReplyPause { + before_acceptance: false, + publication: true, + lose_reply, + captured, + resume: paused, + }); + (observed, resume) + } + #[cfg(test)] + async fn reader_evacuation_reply(&self) -> JournalResult<()> { + let pause = self.inner.reader_evacuation_reply.lock().unwrap().take(); + if let Some(pause) = pause { + let _ = pause.captured.send(()); + let _ = pause.resume.await; + if pause.lose_reply { + return Err(std::io::Error::other( + "injected lost reader evacuation reply after durable commit", + ) + .into()); + } + } + Ok(()) + } + #[cfg(test)] + pub(crate) fn pause_next_follower_evacuation_reply( + &self, + lose_reply: bool, + ) -> ( + tokio::sync::oneshot::Receiver<()>, + tokio::sync::oneshot::Sender<()>, + ) { + let (captured, observed) = tokio::sync::oneshot::channel(); + let (resume, paused) = tokio::sync::oneshot::channel(); + let mut slot = self.inner.follower_evacuation_reply.lock().unwrap(); + assert!(slot.is_none()); + *slot = Some(EnrollmentReplyPause { + before_acceptance: false, + publication: true, + lose_reply, + captured, + resume: paused, + }); + (observed, resume) + } + #[cfg(test)] + async fn follower_evacuation_reply(&self) -> JournalResult<()> { + let pause = self.inner.follower_evacuation_reply.lock().unwrap().take(); + if let Some(pause) = pause { + let _ = pause.captured.send(()); + let _ = pause.resume.await; + if pause.lose_reply { + return Err(std::io::Error::other( + "injected lost follower evacuation reply after durable commit", + ) + .into()); + } + } + Ok(()) + } + #[cfg(test)] + pub(crate) fn pause_before_enrollment_acceptance( + &self, + ) -> ( + tokio::sync::oneshot::Receiver<()>, + tokio::sync::oneshot::Sender<()>, + ) { + let result = self.pause_next_enrollment_reply(false, false); + self.inner + .enrollment_reply + .lock() + .unwrap() + .as_mut() + .unwrap() + .before_acceptance = true; + result + } + + #[cfg(test)] + async fn enrollment_reply( + &self, + publication: bool, + before_acceptance: bool, + ) -> JournalResult<()> { + let pause = { + let mut slot = self.inner.enrollment_reply.lock().unwrap(); + if slot.as_ref().is_some_and(|pause| { + pause.publication == publication && pause.before_acceptance == before_acceptance + }) { + slot.take() + } else { + None + } + }; + if let Some(pause) = pause { + let _ = pause.captured.send(()); + let _ = pause.resume.await; + if pause.lose_reply { + return Err(std::io::Error::other( + "injected lost enrollment reply after durable commit", + ) + .into()); + } + } + Ok(()) + } + + #[cfg(test)] + pub(crate) fn lose_next_commit_reply(&self) { + self.inner + .lose_commit_reply + .store(true, std::sync::atomic::Ordering::SeqCst); + } + + pub async fn close(&self) -> JournalResult<()> { + let runtime = tokio::runtime::Handle::try_current()?; + loop { + let changed = self.inner.changed.notified(); + tokio::pin!(changed); + changed.as_mut().enable(); + let pending = { + let mut state = self + .inner + .activity + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + state.accepting = false; + state.pending + }; + if pending == 0 { + break; + } + changed.await; + } + let inner = Arc::clone(&self.inner); + runtime + .spawn_blocking(move || -> JournalResult<()> { + let mut guard = inner + .connection + .lock() + .map_err(|_| std::io::Error::other("journal connection lock poisoned"))?; + if let Some(connection) = guard.take() + && let Err((connection, error)) = connection.close() + { + *guard = Some(connection); + return Err(error.into()); + } + Ok(()) + }) + .await? + } +} diff --git a/crates/cellule-host/minion/journal/reader_evacuation.rs b/crates/cellule-host/minion/journal/reader_evacuation.rs new file mode 100644 index 00000000..d03d0ebf --- /dev/null +++ b/crates/cellule-host/minion/journal/reader_evacuation.rs @@ -0,0 +1,160 @@ +use super::*; + +impl FleetReaderEvacuationJournal for SqliteJournal { + fn persist_reader_evacuation<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + record: &'a ReaderEvacuationRecord, + pages: &'a [ReaderEvacuationPage], + now_ms: i64, + ) -> FleetAdapterFuture<'a, ReaderEvacuationRecord> { + Box::pin(async move { + record.validate_pages(pages)?; + let expected = expected.clone(); + let record = record.clone(); + let pages = pages.to_vec(); + let result=self.run(move|db| { + db.check_scope(record.retired().spec().scope)?; + let digest=record.digest()?; + if let Some(original)=db.reader_evacuation(digest)? { + if original!=record {return Err(OperationError::Conflict.into());} + for page in &pages { + if db.reader_evacuation_page(page.digest()?)?.as_ref()!=Some(page) {return Err(OperationError::Conflict.into());} + } + // Replay returns history only. It never restores a pointer + // superseded by a fresh capture for the same responsibility. + return Ok(original); + } + let current=db.snapshot()?; + let operation=record.operation(); + let lease=current.head().controller().ok_or(OperationError::Fenced)?; + let (started,finished)=record.interval(); + if current!=expected || record.registry()!=expected.registry() + || record.head_digest()!=Digest::from_bytes(*blake3::hash(&expected.head().to_bytes()?).as_bytes()) + || current.head().maintenance()!=Some(operation) + || current.registry().bootstrap_revision().is_none() + || !matches!(operation.phase(),MaintenancePhase::Evacuating|MaintenancePhase::Closing) + || now_ms>=lease.expires_at_ms || now_ms>=operation.deadline_ms() + || now_ms30_000 { + return Err(OperationError::Conflict.into()); + } + let donor=db.required_intent(operation.node())?; + if donor.session()!=operation.session() || donor.revision()!=operation.intent_revision() || donor.mode()!=NodeMode::Draining + || db.enrollment(record.retired().spec().key()?)?.as_ref()!=Some(record.retired()) { + return Err(OperationError::Conflict.into()); + } + for page in &pages { + for entry in page.entries() { + let row=db.enrollment(entry.enrollment_key)?.ok_or(OperationError::NotFound)?; + let intent=db.required_intent(entry.node)?; + if row.status()!=EnrollmentStatus::Established || row.spec().target.node!=entry.node || row.spec().target.session!=entry.session + || intent.session()!=entry.session || intent.mode()!=NodeMode::Active + || Digest::from_bytes(*blake3::hash(&row.to_bytes()?).as_bytes())!=entry.enrollment_digest + || !matches!(&row.spec().role,EnrollmentRole::Reader {target,position} + if target.cell_id()==record.authority().cell && position.incarnation==record.authority().incarnation + && entry.commit_sequence>=position.root.commit_sequence) { + return Err(OperationError::Conflict.into()); + } + } + let key=page.digest()?; + if let Some(original)=db.reader_evacuation_page(key)? { + if original!=*page {return Err(OperationError::Conflict.into());} + } else { + db.tx.execute("INSERT INTO reader_evacuation_pages(key,body) VALUES(?1,?2)",params![key.as_bytes().as_slice(),page.to_bytes()?])?; + } + } + db.tx.execute("INSERT INTO reader_evacuations(key,body) VALUES(?1,?2)",params![digest.as_bytes().as_slice(),record.to_bytes()?])?; + db.tx.execute("INSERT INTO latest_reader_evacuations(operation,original,witness) VALUES(?1,?2,?3) ON CONFLICT(operation,original) DO UPDATE SET witness=excluded.witness", + params![operation.id().as_bytes().as_slice(),record.retired().spec().key()?.as_bytes().as_slice(),digest.as_bytes().as_slice()])?; + db.advance_registry()?; + Ok(record) + }).await?; + #[cfg(test)] + self.reader_evacuation_reply().await?; + Ok(result) + }) + } + fn load_reader_evacuation( + &self, + scope: FleetScope, + digest: Digest, + ) -> FleetAdapterFuture<'_, Option> { + Box::pin(self.run(move |db| { + db.check_scope(scope)?; + db.reader_evacuation(digest) + })) + } + fn load_reader_evacuation_page( + &self, + scope: FleetScope, + digest: Digest, + ) -> FleetAdapterFuture<'_, Option> { + Box::pin(self.run(move |db| { + db.check_scope(scope)?; + db.reader_evacuation_page(digest) + })) + } + fn latest_reader_evacuation<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + operation: OperationId, + original: Digest, + ) -> FleetAdapterFuture<'a, Option> { + let expected = expected.clone(); + Box::pin(self.run(move|db|{ + db.check_scope(expected.head().scope())?; + if db.snapshot()?!=expected {return Err(OperationError::Conflict.into());} + let key=db.tx.query_row("SELECT witness FROM latest_reader_evacuations WHERE operation=?1 AND original=?2",params![operation.as_bytes().as_slice(),original.as_bytes().as_slice()],|row|blob(row,0,32)).optional()?; + key.map(|bytes| ->JournalResult<_> { + let digest=Digest::from_bytes(bytes.as_slice().try_into()?); + let record=db.reader_evacuation(digest)?.ok_or(OperationError::NotFound)?; + if record.operation().id()!=operation || record.retired().spec().key()?!=original {return Err(OperationError::Conflict.into());} + Ok(record) + }).transpose() + })) + } +} +impl Db<'_> { + fn reader_evacuation(&self, digest: Digest) -> JournalResult> { + let bytes = self + .tx + .query_row( + "SELECT body FROM reader_evacuations WHERE key=?1", + [digest.as_bytes().as_slice()], + |row| blob(row, 0, MAX_PAGE_BYTES), + ) + .optional()?; + bytes + .map(|bytes| -> JournalResult<_> { + let record = ReaderEvacuationRecord::from_bytes(&bytes)?; + self.check_scope(record.retired().spec().scope)?; + if record.digest()? != digest { + return Err(OperationError::Conflict.into()); + } + Ok(record) + }) + .transpose() + } + fn reader_evacuation_page( + &self, + digest: Digest, + ) -> JournalResult> { + let bytes = self + .tx + .query_row( + "SELECT body FROM reader_evacuation_pages WHERE key=?1", + [digest.as_bytes().as_slice()], + |row| blob(row, 0, MAX_RECORD_BYTES), + ) + .optional()?; + bytes + .map(|bytes| -> JournalResult<_> { + let page = ReaderEvacuationPage::from_bytes(&bytes)?; + if page.digest()? != digest { + return Err(OperationError::Conflict.into()); + } + Ok(page) + }) + .transpose() + } +} diff --git a/crates/cellule-host/minion/journal/records.rs b/crates/cellule-host/minion/journal/records.rs new file mode 100644 index 00000000..9db9e977 --- /dev/null +++ b/crates/cellule-host/minion/journal/records.rs @@ -0,0 +1,240 @@ +use super::*; + +pub(super) fn scope_bytes(scope: FleetScope) -> Vec { + [ + scope.fleet.as_bytes().as_slice(), + scope.application.as_bytes().as_slice(), + ] + .concat() +} +pub(super) fn profile_bytes(profile: FleetProfile) -> JournalResult> { + let count = u64::try_from(profile.max_inflight)?; + Ok([ + count.to_be_bytes(), + profile.max_restore_bytes.to_be_bytes(), + profile.controller_lease_ms.to_be_bytes(), + profile.reconcile_interval_ms.to_be_bytes(), + ] + .concat()) +} + +pub(super) fn blob(row: &rusqlite::Row<'_>, column: usize, max: u32) -> rusqlite::Result> { + let bytes = row.get_ref(column)?.as_blob()?; + if bytes.len() > max as usize { + return Err(rusqlite::Error::FromSqlConversionFailure( + column, + rusqlite::types::Type::Blob, + Box::new(cellule_runtime::Error::Capacity("reference journal record")), + )); + } + Ok(bytes.to_vec()) +} + +pub(super) struct Db<'a> { + pub tx: &'a Transaction<'a>, + pub scope: FleetScope, + pub profile: FleetProfile, +} + +impl Db<'_> { + pub fn check_scope(&self, scope: FleetScope) -> JournalResult<()> { + if scope != self.scope { + return Err(OperationError::Conflict.into()); + } + Ok(()) + } + pub fn movement_time( + &self, + expected: &FleetJournalSnapshot, + target: Option<( + cellule_runtime::identity::CellId, + cellule_runtime::identity::IncarnationId, + )>, + ) -> JournalResult> { + self.check_scope(expected.head().scope())?; + if self.snapshot()? != *expected { + return Err(OperationError::Conflict.into()); + } + // Constant resident history memory; only the CAS-reachable chain counts. + // An indexed production backend can satisfy the same atomic contract. + let mut cursor = expected.head().progress(); + let mut latest = None; + while let Some(head) = cursor { + let page = self + .progress(head.digest)? + .ok_or(OperationError::NotFound)?; + if page.sequence() != head.sequence { + return Err(OperationError::Conflict.into()); + } + for entry in page.entries() { + if target.is_none_or(|(cell, incarnation)| { + entry.spec().target.cell_id() == cell && entry.spec().incarnation == incarnation + }) && matches!( + entry.phase(), + AttemptPhase::Activated | AttemptPhase::Recovered + ) { + let at = entry.completed_at_ms().ok_or(OperationError::Conflict)?; + latest = Some(latest.map_or(at, |old: i64| old.max(at))); + } + } + cursor = page.previous().map(|digest| ProgressHead { + digest, + sequence: head.sequence - 1, + }); + } + Ok(latest) + } + pub fn snapshot(&self) -> JournalResult { + let (head, registry) = self.tx.query_row( + "SELECT head, registry FROM state WHERE singleton=1", + [], + |row| { + Ok(( + blob(row, 0, MAX_RECORD_BYTES)?, + blob(row, 1, MAX_RECORD_BYTES)?, + )) + }, + )?; + let head = FleetHead::from_bytes(&head)?; + let registry = RegistryVersion::from_bytes(®istry)?; + self.check_scope(head.scope())?; + if let Some(operation) = head.maintenance() + && self.operation(operation.id())?.as_ref() != Some(operation) + { + return Err(OperationError::Invalid( + "maintenance head lacks its exact retained operation", + ) + .into()); + } + if let Some(progress) = head.progress() { + let page = self + .progress(progress.digest)? + .ok_or(OperationError::NotFound)?; + if page.sequence() != progress.sequence { + return Err(OperationError::Conflict.into()); + } + } + Ok(FleetJournalSnapshot::new(head, registry)?) + } + pub fn set_head(&self, head: &FleetHead) -> JournalResult<()> { + self.check_scope(head.scope())?; + self.tx.execute( + "UPDATE state SET head=?1 WHERE singleton=1", + [head.to_bytes()?], + )?; + Ok(()) + } + pub fn set_registry(&self, version: RegistryVersion) -> JournalResult<()> { + self.check_scope(version.scope())?; + self.tx.execute( + "UPDATE state SET registry=?1 WHERE singleton=1", + [version.to_bytes()?], + )?; + Ok(()) + } + pub fn advance_registry(&self) -> JournalResult<()> { + let version = self.snapshot()?.registry(); + self.set_registry(version.advance(version.revision())?) + } + pub fn intent(&self, node: NodeId) -> JournalResult> { + let bytes = self + .tx + .query_row( + "SELECT body FROM intents WHERE key=?1", + [node.as_bytes().as_slice()], + |row| blob(row, 0, MAX_RECORD_BYTES), + ) + .optional()?; + bytes + .map(|bytes| { + let intent = NodeIntent::from_bytes(&bytes)?; + self.check_scope(intent.scope())?; + if intent.node() != node { + return Err(OperationError::Conflict.into()); + } + Ok(intent) + }) + .transpose() + } + pub fn required_intent(&self, node: NodeId) -> JournalResult { + self.intent(node)? + .ok_or_else(|| OperationError::NotFound.into()) + } + pub fn write_intent(&self, intent: &NodeIntent) -> JournalResult<()> { + self.check_scope(intent.scope())?; + self.tx.execute("INSERT INTO intents(key,body) VALUES (?1,?2) ON CONFLICT(key) DO UPDATE SET body=excluded.body", params![intent.node().as_bytes().as_slice(), intent.to_bytes()?])?; + self.advance_registry() + } + pub fn operation(&self, id: OperationId) -> JournalResult> { + let bytes = self + .tx + .query_row( + "SELECT body FROM operations WHERE key=?1", + [id.as_bytes().as_slice()], + |row| blob(row, 0, MAX_RECORD_BYTES), + ) + .optional()?; + bytes + .map(|bytes| { + let operation = MaintenanceOperation::from_bytes(&bytes)?; + if operation.id() != id { + return Err(OperationError::Conflict.into()); + } + Ok(operation) + }) + .transpose() + } + + pub fn progress(&self, digest: Digest) -> JournalResult> { + let bytes = self + .tx + .query_row( + "SELECT body FROM progress WHERE key=?1", + [digest.as_bytes().as_slice()], + |row| blob(row, 0, MAX_PAGE_BYTES), + ) + .optional()?; + bytes + .map(|bytes| { + let page = ProgressPage::from_bytes(&bytes)?; + self.check_scope(page.scope())?; + if page.digest()? != digest { + return Err(OperationError::Conflict.into()); + } + Ok(page) + }) + .transpose() + } + pub fn enrollment(&self, key: Digest) -> JournalResult> { + let bytes = self + .tx + .query_row( + "SELECT body FROM enrollments WHERE key=?1", + [key.as_bytes().as_slice()], + |row| blob(row, 0, MAX_RECORD_BYTES), + ) + .optional()?; + bytes + .map(|bytes| { + let record = EnrollmentRecord::from_bytes(&bytes)?; + self.check_scope(record.spec().scope)?; + if record.spec().key()? != key { + return Err(OperationError::Conflict.into()); + } + Ok(record) + }) + .transpose() + } + pub fn write_enrollment(&self, record: &EnrollmentRecord) -> JournalResult<()> { + self.check_scope(record.spec().scope)?; + self.tx.execute("INSERT INTO enrollments(key,body) VALUES (?1,?2) ON CONFLICT(key) DO UPDATE SET body=excluded.body", params![record.spec().key()?.as_bytes().as_slice(), record.to_bytes()?])?; + self.advance_registry() + } + pub fn check_version(&self, expected: RegistryVersion) -> JournalResult<()> { + self.check_scope(expected.scope())?; + if self.snapshot()?.registry() != expected { + return Err(OperationError::Conflict.into()); + } + Ok(()) + } +} diff --git a/crates/cellule-host/minion/journal/schema.sql b/crates/cellule-host/minion/journal/schema.sql new file mode 100644 index 00000000..e91bba86 --- /dev/null +++ b/crates/cellule-host/minion/journal/schema.sql @@ -0,0 +1,64 @@ +-- Embedding-owned reference schema. Runtime wire bytes remain canonical. +CREATE TABLE IF NOT EXISTS state ( + singleton INTEGER PRIMARY KEY CHECK (singleton = 1), + format INTEGER NOT NULL CHECK (format = 1), + scope BLOB NOT NULL CHECK (length(scope) = 48), + profile BLOB NOT NULL CHECK (length(profile) = 32), + head BLOB NOT NULL CHECK (length(head) <= 65536), + registry BLOB NOT NULL CHECK (length(registry) <= 65536) +); +CREATE TABLE IF NOT EXISTS intents (key BLOB PRIMARY KEY CHECK (length(key)=16), body BLOB NOT NULL CHECK (length(body)<=65536)); +CREATE TABLE IF NOT EXISTS operations (key BLOB PRIMARY KEY CHECK (length(key)=16), request BLOB NOT NULL CHECK (length(request)<=65536), body BLOB NOT NULL CHECK (length(body)<=65536)); +CREATE TABLE IF NOT EXISTS progress (key BLOB PRIMARY KEY CHECK (length(key)=32), body BLOB NOT NULL CHECK (length(body)<=1048576)); +CREATE TABLE IF NOT EXISTS enrollments (key BLOB PRIMARY KEY CHECK (length(key)=32), body BLOB NOT NULL CHECK (length(body)<=65536)); +CREATE TABLE IF NOT EXISTS actions ( + key BLOB NOT NULL CHECK (length(key)=32), node BLOB NOT NULL CHECK (length(node)=16), session BLOB NOT NULL CHECK (length(session)=16), + operation BLOB NOT NULL CHECK (length(operation)=16), sequence BLOB NOT NULL CHECK (length(sequence) IN (0,8)), effect INTEGER NOT NULL, + accepted BLOB NOT NULL CHECK (length(accepted)<=65536), result BLOB CHECK (length(result)<=65536), + PRIMARY KEY(key,node,session) +); +CREATE UNIQUE INDEX IF NOT EXISTS movement_index ON actions(operation,sequence,effect,node,session) WHERE length(sequence)=8; +CREATE TABLE IF NOT EXISTS bases ( + key BLOB NOT NULL, node BLOB NOT NULL, session BLOB NOT NULL, kind INTEGER NOT NULL CHECK (kind IN (1,2,3)), + body BLOB NOT NULL CHECK (length(body)<=65536), PRIMARY KEY(key,node,session,kind), + FOREIGN KEY(key,node,session) REFERENCES actions(key,node,session) +); + +-- Role evidence shares registry revisions and the same accepted backend jobs. +CREATE TABLE IF NOT EXISTS reader_evacuations ( + key BLOB PRIMARY KEY CHECK(length(key)=32), body BLOB NOT NULL CHECK(length(body)<=1048576) +); +CREATE TABLE IF NOT EXISTS reader_evacuation_pages ( + key BLOB PRIMARY KEY CHECK(length(key)=32), body BLOB NOT NULL CHECK(length(body)<=65536) +); +CREATE TABLE IF NOT EXISTS latest_reader_evacuations ( + operation BLOB NOT NULL CHECK(length(operation)=16), original BLOB NOT NULL CHECK(length(original)=32), + witness BLOB NOT NULL REFERENCES reader_evacuations(key), PRIMARY KEY(operation,original) +); +CREATE TABLE IF NOT EXISTS follower_policy ( + singleton INTEGER PRIMARY KEY CHECK(singleton = 1), + body BLOB NOT NULL CHECK(length(body) <= 65536) +); +CREATE TABLE IF NOT EXISTS follower_evacuations ( + key BLOB PRIMARY KEY CHECK(length(key) = 32), + body BLOB NOT NULL CHECK(length(body) <= 1048576) +); +CREATE TABLE IF NOT EXISTS latest_follower_evacuations ( + operation BLOB NOT NULL CHECK(length(operation) = 16), + original BLOB NOT NULL CHECK(length(original) = 32), + witness BLOB NOT NULL REFERENCES follower_evacuations(key), + PRIMARY KEY(operation, original) +); +-- One immutable original ownership set per operation/process request. Pages and +-- this pointer commit in the same transaction as the registry advance. +CREATE TABLE IF NOT EXISTS original_writers ( + operation BLOB NOT NULL CHECK(length(operation)=16), + process BLOB NOT NULL CHECK(length(process)=32), + key BLOB NOT NULL UNIQUE CHECK(length(key)=32), + body BLOB NOT NULL CHECK(length(body)<=65536), + PRIMARY KEY(operation,process) +); +CREATE TABLE IF NOT EXISTS original_writer_pages ( + key BLOB PRIMARY KEY CHECK(length(key)=32), + body BLOB NOT NULL CHECK(length(body)<=1048576) +); diff --git a/crates/cellule-host/minion/journal/tests.rs b/crates/cellule-host/minion/journal/tests.rs new file mode 100644 index 00000000..616d0b44 --- /dev/null +++ b/crates/cellule-host/minion/journal/tests.rs @@ -0,0 +1,2127 @@ +use super::*; +use cellule_runtime::control::{Control, ControlState, Owner, RootRef}; +use cellule_runtime::identity::{ApplicationId, CellTarget, IncarnationId, NamespaceId, TenantId}; + +fn scope() -> FleetScope { + FleetScope { + fleet: Digest::from_bytes([200; 32]), + application: ApplicationId::from_bytes([3; 16]), + } +} +fn endpoint(n: u8) -> EnrollmentEndpoint { + EnrollmentEndpoint { + node: NodeId::from_bytes([n; 16]), + session: SessionId::from_bytes([n + 10; 16]), + intent_revision: 1, + } +} +fn intent(n: u8) -> NodeIntent { + NodeIntent::initial(scope(), endpoint(n).node, endpoint(n).session).unwrap() +} +fn spec(n: u64) -> MoveAttemptSpec { + MoveAttemptSpec { + id: AttemptId { + operation: OperationId::from_bytes([1; 16]).unwrap(), + sequence: n, + }, + target: CellTarget::new( + TenantId::from_bytes([4; 16]), + scope().application, + NamespaceId::from_bytes([5; 16]), + &n.to_be_bytes(), + ) + .unwrap(), + incarnation: IncarnationId::from_bytes([6; 16]), + source_node: endpoint(1).node, + source: endpoint(1).session, + generation: 7, + source_epoch: 4, + destination_node: endpoint(2).node, + destination: endpoint(2).session, + cost: TransferCost { + memory_bytes: 65536, + disk_bytes: 4096, + file_descriptors: 8, + job_credits: 1, + }, + snapshot_digest: Digest::from_bytes([7; 32]), + deadline_ms: 20_000, + } +} +fn position() -> PublishedPosition { + PublishedPosition { + incarnation: spec(1).incarnation, + epoch: 4, + root: RootRef { + digest: Digest::from_bytes([8; 32]), + txid: 9, + checksum: u64::MAX, + commit_sequence: 9, + }, + } +} +fn control(idle: bool) -> Control { + let mut control = Control::initial( + spec(1).target.cell_id(), + spec(1).incarnation, + Owner { + session: spec(1).source, + endpoint: "https://source.internal:8789".into(), + }, + Digest::from_bytes([9; 32]), + 1, + ) + .unwrap(); + control.epoch = position().epoch; + control.revision = 9; + control.progress = 9; + control.root = Some(position().root); + control.state = if idle { + ControlState::Idle + } else { + ControlState::Serving + }; + if idle { + control.owner = None; + } + control +} +fn request(node: u8, id: u8) -> MaintenanceOperation { + MaintenanceOperation::new( + OperationId::from_bytes([id; 16]).unwrap(), + Digest::from_bytes([id + 30; 32]), + endpoint(node).node, + endpoint(node).session, + 2, + 0, + 20_000, + ) + .unwrap() +} +fn enrollment(id: u8, target: u8) -> EnrollmentSpec { + EnrollmentSpec { + scope: scope(), + request: Digest::from_bytes([id; 32]), + role: EnrollmentRole::Follower { log_epoch: 7 }, + source: Some(endpoint(3)), + target: endpoint(target), + } +} +fn result( + accepted: &AcceptedFleetAction, + outcome: FleetOutcome, + now_ms: i64, +) -> FleetActionOutcome { + FleetActionOutcome { + scope: scope(), + action_key: accepted.action().key().unwrap(), + node: accepted.node(), + session: accepted.session(), + observed_at_ms: now_ms, + outcome, + } +} +fn lose(journal: &SqliteJournal) { + journal + .inner + .lose_commit_reply + .store(true, std::sync::atomic::Ordering::SeqCst); +} + +struct Fixture { + _directory: tempfile::TempDir, + path: PathBuf, + journal: SqliteJournal, +} + +#[tokio::test] +async fn atomic_absence_resolution_rejects_accepted_release_and_preserves_original() { + let fixture = Fixture::new().await; + let delayed = fixture.releasing().await; + let current = fixture.journal.load_snapshot(scope()).await.unwrap(); + let resolved = fixture + .journal + .compare_exchange( + ¤t, + 1, + 1, + &JournalTransition::ResolveUnaccepted { + id: spec(1).id, + effect: MovementAction::Release, + }, + ) + .await + .unwrap(); + assert_eq!(resolved.head().revision(), current.head().revision() + 1); + assert_eq!(resolved.head().reserved_restore_bytes(), 4096); + assert!( + fixture + .journal + .accept_action(&delayed, spec(1).source_node, spec(1).source, 1) + .await + .is_err() + ); + let action = resolved + .head() + .movement_action(spec(1).id, MovementAction::Release, 1) + .unwrap(); + let accepted = fixture + .journal + .accept_action(&action, spec(1).source_node, spec(1).source, 1) + .await + .unwrap(); + assert!(matches!(accepted, FleetActionAcceptance::New(_))); + let error = fixture + .journal + .compare_exchange( + &resolved, + 1, + 1, + &JournalTransition::ResolveUnaccepted { + id: spec(1).id, + effect: MovementAction::Release, + }, + ) + .await + .unwrap_err(); + assert!(matches!( + error.downcast_ref::(), + Some(OperationError::Busy) + )); + assert_eq!( + fixture.journal.load_snapshot(scope()).await.unwrap(), + resolved + ); + assert!(matches!( + fixture + .journal + .load_movement_action( + scope(), + spec(1).id, + MovementAction::Release, + spec(1).source_node, + spec(1).source + ) + .await + .unwrap(), + Some(FleetActionAcceptance::Existing { result: None, .. }) + )); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn independent_acceptance_and_absence_cas_have_one_linearized_winner() { + let fixture = Fixture::new().await; + let action = fixture.releasing().await; + let other = fixture.client().await; + let current = fixture.journal.load_snapshot(scope()).await.unwrap(); + let transition = JournalTransition::ResolveUnaccepted { + id: spec(1).id, + effect: MovementAction::Release, + }; + let (accepted, resolved) = tokio::join!( + fixture + .journal + .accept_action(&action, spec(1).source_node, spec(1).source, 1), + other.compare_exchange(¤t, 1, 1, &transition) + ); + match (accepted, resolved) { + (Ok(FleetActionAcceptance::New(_)), Err(error)) => { + assert!(matches!( + error.downcast_ref::(), + Some(OperationError::Busy) + )); + assert_eq!( + fixture.journal.load_snapshot(scope()).await.unwrap(), + current + ); + } + (Err(_), Ok(resolved)) => { + assert_eq!(resolved.head().revision(), current.head().revision() + 1); + assert!( + fixture + .journal + .load_movement_action( + scope(), + spec(1).id, + MovementAction::Release, + spec(1).source_node, + spec(1).source + ) + .await + .unwrap() + .is_none() + ); + } + _ => panic!("acceptance/absence race did not linearize"), + } + assert_eq!( + other + .load_snapshot(scope()) + .await + .unwrap() + .head() + .reserved_restore_bytes(), + 4096 + ); + other.close().await.unwrap(); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn lost_absence_cas_reply_survives_reconstruction_and_fences_delayed_envelope() { + let fixture = Fixture::new().await; + let delayed = fixture.releasing().await; + let current = fixture.journal.load_snapshot(scope()).await.unwrap(); + lose(&fixture.journal); + assert!( + fixture + .journal + .compare_exchange( + ¤t, + 1, + 1, + &JournalTransition::ResolveUnaccepted { + id: spec(1).id, + effect: MovementAction::Release + } + ) + .await + .is_err() + ); + fixture.journal.close().await.unwrap(); + let reopened = fixture.client().await; + let resolved = reopened.load_snapshot(scope()).await.unwrap(); + assert_eq!(resolved.head().revision(), current.head().revision() + 1); + assert_eq!(resolved.head().reserved_restore_bytes(), 4096); + assert!( + reopened + .accept_action(&delayed, spec(1).source_node, spec(1).source, 1) + .await + .is_err() + ); + let fresh = resolved + .head() + .movement_action(spec(1).id, MovementAction::Release, 1) + .unwrap(); + assert_eq!(fresh.key().unwrap(), delayed.key().unwrap()); + assert!(matches!( + reopened + .accept_action(&fresh, spec(1).source_node, spec(1).source, 1) + .await + .unwrap(), + FleetActionAcceptance::New(_) + )); + reopened.close().await.unwrap(); +} +impl Fixture { + async fn new() -> Self { + let directory = tempfile::tempdir().unwrap(); + let path = directory.path().join("fleet.sqlite"); + let journal = SqliteJournal::open(path.clone(), scope(), FleetProfile::default(), 0) + .await + .unwrap(); + for n in 1..=3 { + journal.register_initial_intent(&intent(n)).await.unwrap(); + } + let registry = journal.load_snapshot(scope()).await.unwrap().registry(); + let registry = journal.bootstrap_registry(registry).await.unwrap(); + journal.set_scheduling(registry, true).await.unwrap(); + journal + .claim_controller(scope(), 0, SessionId::from_bytes([206; 16]), 0) + .await + .unwrap(); + Self { + _directory: directory, + path, + journal, + } + } + async fn client(&self) -> SqliteJournal { + SqliteJournal::open(self.path.clone(), scope(), FleetProfile::default(), 0) + .await + .unwrap() + } + async fn client_error(&self) -> JournalError { + match SqliteJournal::open(self.path.clone(), scope(), FleetProfile::default(), 0).await { + Err(error) => error, + Ok(journal) => { + journal.close().await.unwrap(); + panic!("corrupt journal unexpectedly reopened") + } + } + } + async fn transition(&self, transition: JournalTransition) -> FleetJournalSnapshot { + let current = self.journal.load_snapshot(scope()).await.unwrap(); + self.journal + .compare_exchange( + ¤t, + current.head().controller().unwrap().epoch, + 0, + &transition, + ) + .await + .unwrap() + } + async fn event(&self, event: AttemptEvent) -> FleetJournalSnapshot { + self.transition(JournalTransition::Attempt { + id: spec(1).id, + event, + }) + .await + } + async fn preparing(&self) -> FleetAction { + self.transition(JournalTransition::Allocate(spec(1))).await; + self.event(AttemptEvent::BeginPrepare) + .await + .head() + .movement_action(spec(1).id, MovementAction::Prepare, 0) + .unwrap() + } + async fn releasing(&self) -> FleetAction { + self.preparing().await; + self.event(AttemptEvent::Reserved(ReceiverReservation { + session: spec(1).destination, + expires_at_ms: 20_000, + })) + .await; + self.event(AttemptEvent::BeginRelease) + .await + .head() + .movement_action(spec(1).id, MovementAction::Release, 0) + .unwrap() + } + async fn activating(&self) -> AcceptedFleetAction { + self.releasing().await; + self.event(AttemptEvent::Released(position())).await; + let head = self.event(AttemptEvent::BeginActivate).await; + let action = head + .head() + .movement_action(spec(1).id, MovementAction::Activate, 0) + .unwrap(); + match self + .journal + .accept_action(&action, spec(1).destination_node, spec(1).destination, 0) + .await + .unwrap() + { + FleetActionAcceptance::New(accepted) => accepted, + _ => panic!("not first acceptance"), + } + } +} + +#[tokio::test] +async fn reconstruction_preserves_scope_profile_header_and_stopped_policy() { + let fixture = Fixture::new().await; + let before = fixture.journal.load_snapshot(scope()).await.unwrap(); + let stopped = fixture + .journal + .set_scheduling(before.registry(), false) + .await + .unwrap(); + fixture.journal.close().await.unwrap(); + let restarted = fixture.client().await; + let loaded = restarted.load_snapshot(scope()).await.unwrap(); + assert_eq!(loaded.head(), before.head()); + assert_eq!(loaded.registry(), stopped); + let foreign = FleetScope { + fleet: Digest::from_bytes([199; 32]), + ..scope() + }; + assert!( + SqliteJournal::open(fixture.path.clone(), foreign, FleetProfile::default(), 0) + .await + .is_err() + ); + assert!( + SqliteJournal::open( + fixture.path.clone(), + scope(), + FleetProfile { + max_inflight: 1, + ..FleetProfile::default() + }, + 0 + ) + .await + .is_err() + ); + assert!(restarted.load_snapshot(foreign).await.is_err()); + restarted.close().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn independently_opened_clients_linearize_competing_controller_and_allocation_cas() { + let fixture = Fixture::new().await; + let other = fixture.client().await; + let current = fixture.journal.load_snapshot(scope()).await.unwrap(); + let (a, b) = tokio::join!( + fixture.journal.claim_controller( + scope(), + current.head().revision(), + SessionId::from_bytes([207; 16]), + 30_000 + ), + other.claim_controller( + scope(), + current.head().revision(), + SessionId::from_bytes([208; 16]), + 30_000 + ) + ); + assert_ne!(a.is_ok(), b.is_ok()); + let current = fixture.journal.load_snapshot(scope()).await.unwrap(); + assert_eq!(current.head().controller().unwrap().epoch, 2); + let mut a_spec = spec(1); + a_spec.deadline_ms = 50_000; + let mut b_spec = a_spec.clone(); + b_spec.target = spec(2).target; + let a = JournalTransition::Allocate(a_spec); + let b = JournalTransition::Allocate(b_spec); + let (a, b) = tokio::join!( + fixture.journal.compare_exchange(¤t, 2, 30_000, &a), + other.compare_exchange(¤t, 2, 30_000, &b) + ); + assert_ne!(a.is_ok(), b.is_ok()); + let loaded = other.load_snapshot(scope()).await.unwrap(); + assert_eq!(loaded.head().attempts().len(), 1); + assert_eq!(loaded.head().reserved_restore_bytes(), 4096); + assert!( + other + .compare_exchange( + &loaded, + 1, + 30_000, + &JournalTransition::Attempt { + id: spec(1).id, + event: AttemptEvent::BeginPrepare + } + ) + .await + .is_err() + ); + other.close().await.unwrap(); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn expired_controller_and_lost_allocation_reply_preserve_both_permits_after_restart() { + let fixture = Fixture::new().await; + let current = fixture.journal.load_snapshot(scope()).await.unwrap(); + lose(&fixture.journal); + assert!( + fixture + .journal + .compare_exchange(¤t, 1, 0, &JournalTransition::Allocate(spec(1))) + .await + .is_err() + ); + fixture.event(AttemptEvent::BeginPrepare).await; + fixture.event(AttemptEvent::OutcomeUnknown).await; + fixture + .transition(JournalTransition::Allocate(spec(2))) + .await; + fixture.journal.close().await.unwrap(); + let restarted = fixture.client().await; + let current = restarted.load_snapshot(scope()).await.unwrap(); + let current = restarted + .claim_controller( + scope(), + current.head().revision(), + SessionId::from_bytes([209; 16]), + 30_000, + ) + .await + .unwrap(); + assert_eq!(current.head().attempts().len(), 2); + assert_eq!(current.head().reserved_restore_bytes(), 8192); + let mut third = spec(3); + third.deadline_ms = 50_000; + assert!(matches!( + restarted + .compare_exchange(¤t, 2, 30_000, &JournalTransition::Allocate(third)) + .await + .err() + .unwrap() + .downcast_ref::(), + Some(OperationError::Budget) + )); + let stopped = restarted + .set_scheduling(current.registry(), false) + .await + .unwrap(); + let current = restarted.load_snapshot(scope()).await.unwrap(); + assert_eq!(current.registry(), stopped); + assert_eq!(current.head().reserved_restore_bytes(), 8192); + assert!(matches!( + restarted + .compare_exchange(¤t, 2, 30_000, &JournalTransition::Allocate(spec(3))) + .await + .err() + .unwrap() + .downcast_ref::(), + Some(OperationError::Stopped) + )); + restarted.close().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn concurrent_first_action_acceptance_is_one_effect_and_one_durable_original() { + let fixture = Fixture::new().await; + let other = fixture.client().await; + let action = fixture.preparing().await; + let (a, b) = tokio::join!( + fixture + .journal + .accept_action(&action, spec(1).destination_node, spec(1).destination, 0), + other.accept_action(&action, spec(1).destination_node, spec(1).destination, 0) + ); + let (a, b) = (a.unwrap(), b.unwrap()); + let original = match (a, b) { + ( + FleetActionAcceptance::New(a), + FleetActionAcceptance::Existing { + accepted: b, + result: None, + }, + ) + | ( + FleetActionAcceptance::Existing { + accepted: a, + result: None, + }, + FleetActionAcceptance::New(b), + ) => { + assert_eq!(a, b); + a + } + _ => panic!("not exactly one first acceptance"), + }; + let reserved = result( + &original, + FleetOutcome::Reserved(ReceiverReservation { + session: spec(1).destination, + expires_at_ms: 20_000, + }), + 1, + ); + lose(&fixture.journal); + assert!( + fixture + .journal + .publish_action_result(&original, &reserved) + .await + .is_err() + ); + fixture.journal.close().await.unwrap(); + let restarted = fixture.client().await; + match restarted + .accept_action( + &action, + spec(1).destination_node, + spec(1).destination, + 50_000, + ) + .await + .unwrap() + { + FleetActionAcceptance::Existing { + accepted, + result: Some(result), + } => { + assert_eq!(accepted, original); + assert_eq!(*result, reserved); + } + _ => panic!("lost accepted result"), + } + let changed = result( + &original, + FleetOutcome::Blocked(DrainBlocker::ReceiverCapacity), + 2, + ); + assert!( + restarted + .publish_action_result(&original, &changed) + .await + .is_err() + ); + other.close().await.unwrap(); + restarted.close().await.unwrap(); +} + +#[tokio::test] +async fn source_result_and_endpoint_specific_inspections_survive_controller_reconstruction() { + let fixture = Fixture::new().await; + let action = fixture.releasing().await; + let original = match fixture + .journal + .accept_action(&action, spec(1).source_node, spec(1).source, 0) + .await + .unwrap() + { + FleetActionAcceptance::New(original) => original, + _ => panic!("not new"), + }; + let released = result(&original, FleetOutcome::Released(position()), 1); + lose(&fixture.journal); + assert!( + fixture + .journal + .publish_action_result(&original, &released) + .await + .is_err() + ); + let current = fixture.event(AttemptEvent::OutcomeUnknown).await; + let inspection = current + .head() + .movement_action(spec(1).id, MovementAction::Inspect, 0) + .unwrap(); + assert!(matches!( + fixture + .journal + .accept_action(&inspection, spec(1).source_node, spec(1).source, 0) + .await + .unwrap(), + FleetActionAcceptance::New(_) + )); + assert!(matches!( + fixture + .journal + .accept_action( + &inspection, + spec(1).destination_node, + spec(1).destination, + 0 + ) + .await + .unwrap(), + FleetActionAcceptance::New(_) + )); + fixture.journal.close().await.unwrap(); + let restarted = fixture.client().await; + match restarted + .load_movement_action( + scope(), + spec(1).id, + MovementAction::Release, + spec(1).source_node, + spec(1).source, + ) + .await + .unwrap() + .unwrap() + { + FleetActionAcceptance::Existing { + accepted, + result: Some(result), + } => { + assert_eq!(accepted, original); + assert_eq!(*result, released); + } + _ => panic!("lost source release"), + } + assert_eq!( + restarted + .load_snapshot(scope()) + .await + .unwrap() + .head() + .reserved_restore_bytes(), + 4096 + ); + restarted.close().await.unwrap(); +} + +#[tokio::test] +async fn acquisition_basis_keeps_original_input_and_time_after_lost_commit_reply() { + let fixture = Fixture::new().await; + let accepted = fixture.activating().await; + let basis = AcquisitionBasis::new(accepted.clone(), control(true), 1).unwrap(); + let activated = ActivationEvidence { + node: spec(1).destination_node, + session: spec(1).destination, + position: PublishedPosition { + epoch: 5, + ..position() + }, + }; + let outcome = result(&accepted, FleetOutcome::Activated(activated), 3); + assert!( + fixture + .journal + .publish_action_result(&accepted, &outcome) + .await + .is_err() + ); + lose(&fixture.journal); + assert!( + fixture + .journal + .record_acquisition_basis(&basis) + .await + .is_err() + ); + let retry = AcquisitionBasis::new(accepted.clone(), control(true), 2).unwrap(); + assert_eq!( + fixture + .journal + .record_acquisition_basis(&retry) + .await + .unwrap(), + basis + ); + let mut changed = control(true); + changed.revision += 1; + assert!( + fixture + .journal + .record_acquisition_basis(&AcquisitionBasis::new(accepted.clone(), changed, 2).unwrap()) + .await + .is_err() + ); + fixture + .journal + .publish_action_result(&accepted, &outcome) + .await + .unwrap(); + fixture.journal.close().await.unwrap(); + let restarted = fixture.client().await; + assert_eq!( + restarted.load_acquisition_basis(&accepted).await.unwrap(), + Some(basis) + ); + restarted.close().await.unwrap(); +} + +#[tokio::test] +async fn recovery_records_reconstruct_exact_original_basis_position_and_result() { + use cellule_runtime::node::{ + NODE_LOG_PROTOCOL_VERSION, NodeAdvertisement, NodeCapacity, NodeDirectory, + NodeFailureDomain, + }; + let fixture = Fixture::new().await; + fixture.releasing().await; + let head = fixture.event(AttemptEvent::BeginRecover).await; + let action = head + .head() + .movement_action(spec(1).id, MovementAction::Recover, 0) + .unwrap(); + let accepted = match fixture + .journal + .accept_action(&action, spec(1).destination_node, spec(1).destination, 0) + .await + .unwrap() + { + FleetActionAcceptance::New(a) => a, + _ => panic!("not new"), + }; + let layout = cellule_runtime::ltx::CellStorageLayout::new( + cellule_store::Store::new(Arc::new(object_store::memory::InMemory::new())), + object_store::path::Path::from("fleet-reference-recovery"), + *scope().application.as_bytes(), + ); + let image = Digest::from_bytes([90; 32]); + let release = Digest::from_bytes([91; 32]); + let key = ed25519_dalek::SigningKey::from_bytes(&[92; 32]); + let directory = NodeDirectory::new(layout, scope().fleet, image, release); + let signed = |n, issued, expires| { + NodeAdvertisement::sign( + endpoint(n).node, + endpoint(n).session, + "https://recovery.internal:8789".into(), + scope().fleet, + Digest::from_bytes([93; 32]), + image, + release, + &key, + 1, + issued, + expires, + vec![Digest::from_bytes([9; 32])], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + free_memory_bytes: 128 << 20, + free_disk_bytes: 8 << 30, + follower_free_bytes: 8 << 30, + follower_retained_bytes: 0, + job_credits: 1, + log_protocol: NODE_LOG_PROTOCOL_VERSION, + }, + ) + .unwrap() + }; + directory.create(signed(1, 1, 9), 1).await.unwrap(); + directory.create(signed(2, 10, 10_010), 10).await.unwrap(); + let proof = directory + .claim_expired_for_takeover(spec(1).source, spec(1).destination, 10) + .await + .unwrap(); + let basis = RecoveryBasis::new(&accepted, control(false), proof, 10).unwrap(); + lose(&fixture.journal); + assert!( + fixture + .journal + .record_recovery_basis(&accepted, &basis) + .await + .is_err() + ); + let retry = RecoveryBasis::new(&accepted, control(false), proof, 11).unwrap(); + assert_eq!( + fixture + .journal + .record_recovery_basis(&accepted, &retry) + .await + .unwrap(), + basis + ); + let restored = basis + .control() + .takeover(Owner { + session: spec(1).destination, + endpoint: "https://receiver.internal:8789".into(), + }) + .unwrap(); + let evidence = RecoveryEvidence::new(basis.clone(), restored.clone(), 11).unwrap(); + let recovered = RecoveredActivation { + recovery: evidence.clone(), + serving: ActivationEvidence { + node: spec(1).destination_node, + session: spec(1).destination, + position: evidence.position().unwrap(), + }, + }; + let outcome = result(&accepted, FleetOutcome::Recovered(Box::new(recovered)), 12); + assert!( + fixture + .journal + .publish_action_result(&accepted, &outcome) + .await + .is_err() + ); + lose(&fixture.journal); + assert!( + fixture + .journal + .record_recovery_evidence(&accepted, &evidence) + .await + .is_err() + ); + assert_eq!( + fixture + .journal + .record_recovery_evidence( + &accepted, + &RecoveryEvidence::new(basis.clone(), restored, 12).unwrap() + ) + .await + .unwrap(), + evidence + ); + fixture + .journal + .publish_action_result(&accepted, &outcome) + .await + .unwrap(); + fixture.journal.close().await.unwrap(); + let restarted = fixture.client().await; + assert_eq!( + restarted.load_recovery_basis(&accepted).await.unwrap(), + Some(basis) + ); + assert_eq!( + restarted.load_recovery_evidence(&accepted).await.unwrap(), + Some(evidence) + ); + match restarted + .load_movement_action( + scope(), + spec(1).id, + MovementAction::Recover, + spec(1).destination_node, + spec(1).destination, + ) + .await + .unwrap() + .unwrap() + { + FleetActionAcceptance::Existing { + result: Some(saved), + .. + } => assert_eq!(*saved, outcome), + _ => panic!("not retained"), + } + restarted.close().await.unwrap(); +} + +#[tokio::test] +async fn maintenance_cordon_enrollment_and_original_request_survive_lost_reply_and_reboot() { + let fixture = Fixture::new().await; + let spec = enrollment(40, 1); + let admitted = match fixture.journal.accept_enrollment(&spec, 1).await.unwrap() { + FleetEnrollmentAcceptance::New(r) => r, + _ => panic!("not new"), + }; + let current = fixture.journal.load_snapshot(scope()).await.unwrap(); + lose(&fixture.journal); + assert!( + fixture + .journal + .compare_exchange( + ¤t, + 1, + 0, + &JournalTransition::BeginMaintenance(request(1, 1)) + ) + .await + .is_err() + ); + fixture.journal.close().await.unwrap(); + let restarted = fixture.client().await; + let current = restarted.load_snapshot(scope()).await.unwrap(); + let page = restarted + .intents_page(current.registry(), None, 128) + .await + .unwrap(); + assert_eq!(page.entries()[0].mode(), NodeMode::Draining); + assert_eq!(page.entries()[0].revision(), 2); + assert!(restarted.register_initial_intent(&intent(1)).await.is_err()); + assert!( + matches!(restarted.accept_enrollment(&spec,50_000).await.unwrap(),FleetEnrollmentAcceptance::Existing(original) if original==admitted) + ); + let mut new = spec.clone(); + new.request = Digest::from_bytes([41; 32]); + new.target.intent_revision = 2; + assert!(restarted.accept_enrollment(&new, 2).await.is_err()); + let same = restarted + .compare_exchange( + ¤t, + 1, + 0, + &JournalTransition::BeginMaintenance(request(1, 1)), + ) + .await + .unwrap(); + assert_eq!(same, current); + assert!( + restarted + .compare_exchange( + ¤t, + 99, + 0, + &JournalTransition::BeginMaintenance(request(1, 1)) + ) + .await + .is_err() + ); + let deadline = restarted + .compare_exchange( + ¤t, + 1, + 1, + &JournalTransition::Maintenance(MaintenanceEvent::ExtendDeadline(25_000)), + ) + .await + .unwrap(); + assert_eq!(deadline.head().maintenance().unwrap().intent_revision(), 3); + let duplicate = restarted + .compare_exchange( + &deadline, + 1, + 1, + &JournalTransition::BeginMaintenance(request(1, 1)), + ) + .await + .unwrap(); + assert_eq!(duplicate, deadline); + let changed = MaintenanceOperation::new( + request(1, 1).id(), + Digest::from_bytes([99; 32]), + endpoint(1).node, + endpoint(1).session, + 2, + 0, + 20_000, + ) + .unwrap(); + assert!( + restarted + .compare_exchange( + &deadline, + 1, + 1, + &JournalTransition::BeginMaintenance(changed) + ) + .await + .is_err() + ); + let reboot = restarted + .compare_exchange( + &deadline, + 1, + 2, + &JournalTransition::Maintenance(MaintenanceEvent::SessionReplaced( + SessionId::from_bytes([99; 16]), + )), + ) + .await + .unwrap(); + let page = restarted + .intents_page(reboot.registry(), None, 128) + .await + .unwrap(); + assert_eq!(page.entries()[0].session(), SessionId::from_bytes([99; 16])); + assert_eq!(page.entries()[0].revision(), 4); + assert_eq!(page.entries()[0].mode(), NodeMode::Draining); + let obligations = restarted + .enrollments_page(reboot.registry(), None, 128) + .await + .unwrap(); + assert_eq!(obligations.entries(), &[admitted.clone()]); + assert!(obligations.entries()[0].unresolved()); + let completed = restarted + .publish_enrollment_result( + &admitted, + EnrollmentEvent::Established(Digest::from_bytes([42; 32])), + 3, + ) + .await + .unwrap(); + assert_eq!(completed.accepted_at_ms(), 1); + restarted.close().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn enrollment_racing_cordon_is_admitted_before_it_or_definitely_refused() { + let fixture = Fixture::new().await; + let other = fixture.client().await; + let expected = fixture.journal.load_snapshot(scope()).await.unwrap(); + let spec = enrollment(43, 1); + let transition = JournalTransition::BeginMaintenance(request(1, 1)); + let (enrolled, cordoned) = tokio::join!( + fixture.journal.accept_enrollment(&spec, 1), + other.compare_exchange(&expected, 1, 0, &transition) + ); + assert_ne!(enrolled.is_ok(), cordoned.is_ok()); + if let Ok(FleetEnrollmentAcceptance::New(record)) = enrolled { + let current = other.load_snapshot(scope()).await.unwrap(); + other + .compare_exchange(¤t, 1, 0, &transition) + .await + .unwrap(); + assert_eq!( + other + .load_enrollment(scope(), spec.key().unwrap()) + .await + .unwrap(), + Some(record) + ); + } else { + assert!( + fixture + .journal + .load_enrollment(scope(), spec.key().unwrap()) + .await + .unwrap() + .is_none() + ); + } + let mut late = spec; + late.request = Digest::from_bytes([44; 32]); + late.target.intent_revision = 2; + assert!(fixture.journal.accept_enrollment(&late, 2).await.is_err()); + other.close().await.unwrap(); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn terminal_history_and_permit_retirement_commit_atomically_across_reconstruction() { + let fixture = Fixture::new().await; + let stale = fixture.journal.load_snapshot(scope()).await.unwrap(); + fixture + .transition(JournalTransition::Allocate(spec(1))) + .await; + fixture.event(AttemptEvent::BeginCancel).await; + let current = fixture.event(AttemptEvent::Cancelled).await; + let progress = current.head().retirement_page(&[spec(1).id]).unwrap(); + assert!( + fixture + .journal + .compare_exchange( + &stale, + 1, + 0, + &JournalTransition::Retire { + progress: progress.clone() + } + ) + .await + .is_err() + ); + assert!( + fixture + .journal + .load_progress(scope(), progress.digest().unwrap()) + .await + .unwrap() + .is_none() + ); + assert_eq!( + fixture + .journal + .load_snapshot(scope()) + .await + .unwrap() + .head() + .reserved_restore_bytes(), + 4096 + ); + lose(&fixture.journal); + assert!( + fixture + .journal + .compare_exchange( + ¤t, + 1, + 0, + &JournalTransition::Retire { + progress: progress.clone() + } + ) + .await + .is_err() + ); + fixture.journal.close().await.unwrap(); + let restarted = fixture.client().await; + let head = restarted.load_snapshot(scope()).await.unwrap(); + assert!(head.head().attempts().is_empty()); + assert_eq!(head.head().reserved_restore_bytes(), 0); + assert_eq!( + head.head().progress().unwrap().digest, + progress.digest().unwrap() + ); + assert_eq!( + restarted + .load_progress(scope(), progress.digest().unwrap()) + .await + .unwrap(), + Some(progress) + ); + restarted.close().await.unwrap(); +} + +#[tokio::test] +async fn later_operations_retain_old_cordons_and_return_to_service_does_not_rewrite_head() { + let fixture = Fixture::new().await; + fixture + .transition(JournalTransition::BeginMaintenance(request(1, 1))) + .await; + fixture + .transition(JournalTransition::Maintenance(MaintenanceEvent::Cordoned)) + .await; + fixture + .transition(JournalTransition::Maintenance( + MaintenanceEvent::BeginEvacuation, + )) + .await; + let proof = DrainEvidence { + node: endpoint(1).node, + session: endpoint(1).session, + remaining_cells: 0, + unresolved_attempts: 0, + relocated: true, + readers_settled: true, + followers_settled: true, + facilities_closed: true, + stopped: true, + withdrawn: true, + }; + fixture + .transition(JournalTransition::Maintenance( + MaintenanceEvent::ReadyToClose(proof), + )) + .await; + fixture + .transition(JournalTransition::Maintenance(MaintenanceEvent::Stopped( + proof, + ))) + .await; + let completed = fixture + .journal + .load_operation(scope(), request(1, 1).id()) + .await + .unwrap() + .unwrap(); + let later = fixture + .transition(JournalTransition::BeginMaintenance(request(2, 2))) + .await; + fixture.journal.close().await.unwrap(); + let restarted = fixture.client().await; + let page = restarted + .intents_page(later.registry(), None, 128) + .await + .unwrap(); + assert_eq!(page.entries()[0].mode(), NodeMode::Draining); + assert_eq!(page.entries()[1].mode(), NodeMode::Draining); + assert_eq!( + restarted + .load_operation(scope(), request(1, 1).id()) + .await + .unwrap(), + Some(completed) + ); + let returned = restarted + .return_to_service( + scope(), + endpoint(1).node, + request(1, 1).id(), + SessionId::from_bytes([41; 16]), + 3, + ) + .await + .unwrap(); + assert_eq!(returned.mode(), NodeMode::Active); + let again = restarted + .return_to_service( + scope(), + endpoint(1).node, + request(1, 1).id(), + SessionId::from_bytes([41; 16]), + 3, + ) + .await + .unwrap(); + assert_eq!(again, returned); + let current = restarted.load_snapshot(scope()).await.unwrap(); + assert_eq!(current.head(), later.head()); + let mut move_spec = spec(1); + move_spec.source_node = endpoint(3).node; + move_spec.source = endpoint(3).session; + move_spec.destination_node = endpoint(1).node; + move_spec.destination = returned.session(); + restarted + .compare_exchange(¤t, 1, 0, &JournalTransition::Allocate(move_spec)) + .await + .unwrap(); + restarted.close().await.unwrap(); +} + +#[tokio::test] +async fn bounded_scans_reject_changed_versions_and_keep_pending_failed_boots() { + let fixture = Fixture::new().await; + let a = match fixture + .journal + .accept_enrollment(&enrollment(45, 1), 1) + .await + .unwrap() + { + FleetEnrollmentAcceptance::New(r) => r, + _ => panic!("not new"), + }; + let b = match fixture + .journal + .accept_enrollment(&enrollment(46, 2), 1) + .await + .unwrap() + { + FleetEnrollmentAcceptance::New(r) => r, + _ => panic!("not new"), + }; + let snapshot = fixture.journal.load_snapshot(scope()).await.unwrap(); + let first = fixture + .journal + .intents_page(snapshot.registry(), None, 1) + .await + .unwrap(); + assert_eq!(first.entries(), &[intent(1)]); + assert_eq!(first.next(), Some(endpoint(1).node)); + let second = fixture + .journal + .intents_page(snapshot.registry(), first.next(), 1) + .await + .unwrap(); + assert_eq!(second.entries(), &[intent(2)]); + let enrolled = fixture + .journal + .enrollments_page(snapshot.registry(), None, 1) + .await + .unwrap(); + assert_eq!(enrolled.entries().len(), 1); + assert!(enrolled.next().is_some()); + let last = fixture + .journal + .enrollments_page(snapshot.registry(), enrolled.next(), 1) + .await + .unwrap(); + assert!(last.next().is_none()); + assert_eq!(last.entries().len(), 1); + let new = fixture + .journal + .rebind_active_intent(&intent(3), SessionId::from_bytes([47; 16]), 2) + .await + .unwrap(); + assert_ne!(new.session(), endpoint(3).session); + assert!( + fixture + .journal + .intents_page(snapshot.registry(), second.next(), 1) + .await + .is_err() + ); + assert!( + fixture + .journal + .enrollments_page(snapshot.registry(), None, 1) + .await + .is_err() + ); + let current = fixture.journal.load_snapshot(scope()).await.unwrap(); + let rows = fixture + .journal + .enrollments_page(current.registry(), None, 128) + .await + .unwrap(); + assert!(rows.entries().contains(&a)); + assert!(rows.entries().contains(&b)); + assert!(rows.entries().iter().all(|r| r.unresolved())); + for limit in [0, MAX_PAGE_ENTRIES + 1] { + assert!( + fixture + .journal + .intents_page(current.registry(), None, limit) + .await + .is_err() + ); + assert!( + fixture + .journal + .enrollments_page(current.registry(), None, limit) + .await + .is_err() + ); + } + assert!( + fixture + .journal + .intents_page(current.registry(), Some(NodeId::from_bytes([99; 16])), 1) + .await + .is_err() + ); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn dropped_transaction_waiter_cannot_cancel_commit_and_close_joins_owned_job() { + let fixture = Fixture::new().await; + let journal = fixture.journal.clone(); + let (entered_tx, entered_rx) = tokio::sync::oneshot::channel(); + let (resume_tx, resume_rx) = std::sync::mpsc::channel(); + let writer = tokio::spawn(async move { + journal + .run(move |db| { + db.write_intent(&intent(4))?; + entered_tx.send(()).unwrap(); + resume_rx.recv().unwrap(); + Ok(()) + }) + .await + }); + entered_rx.await.unwrap(); + writer.abort(); + let closer = fixture.journal.clone(); + let closing = tokio::spawn(async move { closer.close().await }); + tokio::task::yield_now().await; + assert!(!closing.is_finished()); + assert_eq!(fixture.journal.inner.activity.lock().unwrap().pending, 1); + resume_tx.send(()).unwrap(); + closing.await.unwrap().unwrap(); + assert_eq!( + fixture.journal.inner.slots.available_permits(), + MAX_PENDING_JOBS + ); + assert!(fixture.journal.inner.connection.lock().unwrap().is_none()); + assert!(matches!( + fixture + .journal + .load_snapshot(scope()) + .await + .err() + .unwrap() + .downcast_ref::(), + Some(cellule_runtime::Error::RuntimeClosed) + )); + fixture.journal.close().await.unwrap(); + let restarted = fixture.client().await; + let snapshot = restarted.load_snapshot(scope()).await.unwrap(); + assert!( + restarted + .intents_page(snapshot.registry(), None, 128) + .await + .unwrap() + .entries() + .contains(&intent(4)) + ); + restarted.close().await.unwrap(); +} + +#[tokio::test] +async fn lost_enrollment_replies_preserve_pending_and_immutable_settlement_after_restart() { + let fixture = Fixture::new().await; + let spec = enrollment(48, 1); + lose(&fixture.journal); + assert!(fixture.journal.accept_enrollment(&spec, 1).await.is_err()); + fixture.journal.close().await.unwrap(); + let restarted = fixture.client().await; + let original = match restarted.accept_enrollment(&spec, 30_000).await.unwrap() { + FleetEnrollmentAcceptance::Existing(record) => record, + _ => panic!("lost reply must not create another enrollment"), + }; + assert_eq!(original.status(), EnrollmentStatus::Pending); + assert_eq!(original.accepted_at_ms(), 1); + assert_eq!(original.updated_at_ms(), 1); + let mut changed = spec.clone(); + changed.role = EnrollmentRole::Follower { log_epoch: 8 }; + assert!(restarted.accept_enrollment(&changed, 30_000).await.is_err()); + let event = EnrollmentEvent::Refused(Digest::from_bytes([49; 32])); + lose(&restarted); + assert!( + restarted + .publish_enrollment_result(&original, event, 30_000) + .await + .is_err() + ); + let settled = restarted + .publish_enrollment_result(&original, event, 30_001) + .await + .unwrap(); + assert_eq!(settled.status(), EnrollmentStatus::Refused); + assert_eq!(settled.updated_at_ms(), 30_000); + assert!(!settled.unresolved()); + assert!( + restarted + .publish_enrollment_result( + &original, + EnrollmentEvent::Established(Digest::from_bytes([50; 32])), + 30_002 + ) + .await + .is_err() + ); + restarted.close().await.unwrap(); + let reconstructed = fixture.client().await; + assert_eq!( + reconstructed + .load_enrollment(scope(), spec.key().unwrap()) + .await + .unwrap(), + Some(settled) + ); + reconstructed.close().await.unwrap(); +} + +#[tokio::test] +async fn unexecuted_refusal_fences_delayed_acceptance_and_survives_client_restart() { + let fixture = Fixture::new().await; + let spec = enrollment(88, 1); + let proof = Digest::from_bytes([89; 32]); + let before = fixture + .journal + .load_snapshot(scope()) + .await + .unwrap() + .registry(); + lose(&fixture.journal); + assert!( + fixture + .journal + .refuse_unexecuted_enrollment(&spec, proof, 10) + .await + .is_err() + ); + fixture.journal.close().await.unwrap(); + let restarted = fixture.client().await; + let original = restarted + .refuse_unexecuted_enrollment(&spec, proof, 20) + .await + .unwrap(); + assert_eq!(original.status(), EnrollmentStatus::Refused); + assert_eq!(original.accepted_at_ms(), 10); + let after = restarted.load_snapshot(scope()).await.unwrap().registry(); + assert_eq!(after.revision(), before.revision() + 1); + match restarted.accept_enrollment(&spec, 30).await.unwrap() { + FleetEnrollmentAcceptance::Existing(observed) => assert_eq!(observed, original), + FleetEnrollmentAcceptance::New(_) => panic!("delayed acceptance must replay exclusion"), + } + let mut changed = spec.clone(); + changed.role = EnrollmentRole::Follower { log_epoch: 8 }; + assert!( + restarted + .refuse_unexecuted_enrollment(&changed, proof, 40) + .await + .is_err() + ); + assert!( + restarted + .refuse_unexecuted_enrollment(&spec, Digest::from_bytes([90; 32]), 40) + .await + .is_err() + ); + assert_eq!( + restarted.load_snapshot(scope()).await.unwrap().registry(), + after + ); + restarted.close().await.unwrap(); +} + +#[tokio::test] +async fn concurrent_acceptance_and_unexecuted_refusal_share_one_transaction_domain() { + let fixture = Fixture::new().await; + let other = fixture.client().await; + let spec = enrollment(91, 1); + let proof = Digest::from_bytes([92; 32]); + let (accepted, refused) = tokio::join!( + fixture.journal.accept_enrollment(&spec, 10), + other.refuse_unexecuted_enrollment(&spec, proof, 11) + ); + accepted.unwrap(); + let refused = refused.unwrap(); + assert_eq!(refused.status(), EnrollmentStatus::Refused); + assert_eq!( + fixture + .journal + .load_enrollment(scope(), spec.key().unwrap()) + .await + .unwrap(), + Some(refused) + ); + let live = enrollment(93, 2); + let original = match fixture.journal.accept_enrollment(&live, 20).await.unwrap() { + FleetEnrollmentAcceptance::New(record) => record, + _ => panic!("new request"), + }; + fixture + .journal + .publish_enrollment_result(&original, EnrollmentEvent::Established(proof), 21) + .await + .unwrap(); + assert!( + other + .refuse_unexecuted_enrollment(&live, proof, 22) + .await + .is_err() + ); + other.close().await.unwrap(); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn failed_transaction_rolls_back_head_intent_and_registry_together() { + let fixture = Fixture::new().await; + let before = fixture.journal.load_snapshot(scope()).await.unwrap(); + let captured = before.clone(); + let error = fixture + .journal + .run(move |db| -> JournalResult<()> { + db.write_intent(&intent(4))?; + let next = captured.head().transition( + db.profile, + captured.head().revision(), + 1, + 0, + JournalTransition::Allocate(spec(1)), + )?; + db.set_head(&next)?; + Err(std::io::Error::other("injected failure before commit").into()) + }) + .await + .unwrap_err(); + assert_eq!( + error.downcast_ref::().unwrap().to_string(), + "injected failure before commit" + ); + assert_eq!( + fixture.journal.load_snapshot(scope()).await.unwrap(), + before + ); + assert_eq!( + fixture + .journal + .intents_page(before.registry(), None, 128) + .await + .unwrap() + .entries(), + &[intent(1), intent(2), intent(3)] + ); + fixture.journal.close().await.unwrap(); + let restarted = fixture.client().await; + assert_eq!(restarted.load_snapshot(scope()).await.unwrap(), before); + restarted.close().await.unwrap(); +} + +#[tokio::test] +async fn malformed_head_and_missing_committed_references_fail_reconstruction() { + let fixture = Fixture::new().await; + let saved = fixture.journal.load_snapshot(scope()).await.unwrap(); + fixture + .journal + .run(|db| { + db.tx.execute("UPDATE state SET head=?1", [vec![0u8]])?; + Ok(()) + }) + .await + .unwrap(); + let error = fixture.journal.load_snapshot(scope()).await.unwrap_err(); + assert!(matches!( + error.downcast_ref::(), + Some(OperationError::Codec(_)) + )); + assert!(std::error::Error::source(error.as_ref()).is_some()); + fixture.journal.close().await.unwrap(); + let error = fixture.client_error().await; + assert!(matches!( + error.downcast_ref::(), + Some(OperationError::Codec(_)) + )); + + // Repair only this deliberately corrupted fixture through SQLite, then + // remove a referenced operation. The adapter must reject either root. + let connection = Connection::open(&fixture.path).unwrap(); + connection + .execute( + "UPDATE state SET head=?1", + [saved.head().to_bytes().unwrap()], + ) + .unwrap(); + connection.close().unwrap(); + let restarted = fixture.client().await; + let current = restarted.load_snapshot(scope()).await.unwrap(); + restarted + .compare_exchange( + ¤t, + 1, + 0, + &JournalTransition::BeginMaintenance(request(1, 52)), + ) + .await + .unwrap(); + restarted + .run(|db| { + db.tx.execute("DELETE FROM operations", [])?; + Ok(()) + }) + .await + .unwrap(); + assert!(restarted.load_snapshot(scope()).await.is_err()); + restarted.close().await.unwrap(); + assert!(matches!( + fixture + .client_error() + .await + .downcast_ref::(), + Some(OperationError::Invalid(_)) + )); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn job_admission_is_bounded_and_close_joins_every_accepted_transaction() { + let fixture = Fixture::new().await; + let journal = fixture.journal.clone(); + let (entered_tx, entered_rx) = tokio::sync::oneshot::channel(); + let (resume_tx, resume_rx) = std::sync::mpsc::channel(); + let first = tokio::spawn(async move { + journal + .run(move |_| { + entered_tx.send(()).unwrap(); + resume_rx.recv().unwrap(); + Ok(()) + }) + .await + }); + entered_rx.await.unwrap(); + let mut accepted = Vec::new(); + for _ in 1..MAX_PENDING_JOBS { + let journal = fixture.journal.clone(); + accepted.push(tokio::spawn(async move { journal.run(|_| Ok(())).await })); + } + tokio::time::timeout(Duration::from_secs(5), async { + loop { + if fixture.journal.inner.activity.lock().unwrap().pending == MAX_PENDING_JOBS { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + let error = fixture.journal.run(|_| Ok(())).await.unwrap_err(); + assert!(matches!( + error.downcast_ref::(), + Some(cellule_runtime::Error::Capacity("reference journal jobs")) + )); + let journal = fixture.journal.clone(); + let closing = tokio::spawn(async move { journal.close().await }); + tokio::time::timeout(Duration::from_secs(5), async { + loop { + if !fixture.journal.inner.activity.lock().unwrap().accepting { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + assert!(!closing.is_finished()); + resume_tx.send(()).unwrap(); + first.await.unwrap().unwrap(); + for job in accepted { + job.await.unwrap().unwrap(); + } + closing.await.unwrap().unwrap(); + assert_eq!(fixture.journal.inner.activity.lock().unwrap().pending, 0); + assert_eq!( + fixture.journal.inner.slots.available_permits(), + MAX_PENDING_JOBS + ); + assert!(fixture.journal.inner.connection.lock().unwrap().is_none()); +} + +#[tokio::test] +async fn fresh_inspection_authorization_checks_current_head_and_boot_without_writes() { + let fixture = Fixture::new().await; + fixture.preparing().await; + let before = fixture.journal.load_snapshot(scope()).await.unwrap(); + let action = before + .head() + .movement_action(spec(1).id, MovementAction::Inspect, 0) + .unwrap(); + let request = FleetInspectionRequest::new( + action, + before.registry(), + Digest::from_bytes([60; 32]), + spec(1).destination_node, + spec(1).destination, + 5_000, + ) + .unwrap(); + fixture + .journal + .authorize_inspection(&request, 1) + .await + .unwrap(); + assert_eq!( + fixture.journal.load_snapshot(scope()).await.unwrap(), + before + ); + assert!( + fixture + .journal + .load_movement_action( + scope(), + spec(1).id, + MovementAction::Inspect, + spec(1).destination_node, + spec(1).destination + ) + .await + .unwrap() + .is_none() + ); + // A current head is insufficient if the physical endpoint has a newer boot. + fixture + .journal + .rebind_active_intent(&intent(2), SessionId::from_bytes([61; 16]), 2) + .await + .unwrap(); + assert!( + fixture + .journal + .authorize_inspection(&request, 2) + .await + .is_err() + ); + let changed = fixture.journal.load_snapshot(scope()).await.unwrap(); + assert_eq!(changed.head(), before.head()); + let old_boot = FleetInspectionRequest::new( + request.action().clone(), + changed.registry(), + Digest::from_bytes([62; 32]), + request.node(), + request.session(), + request.deadline_ms(), + ) + .unwrap(); + assert!( + fixture + .journal + .authorize_inspection(&old_boot, 2) + .await + .is_err() + ); + fixture + .journal + .claim_controller( + scope(), + before.head().revision(), + before.head().controller().unwrap().claimant, + 2, + ) + .await + .unwrap(); + assert!( + fixture + .journal + .authorize_inspection(&request, 2) + .await + .is_err() + ); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn native_snapshot_authorization_rechecks_full_barrier_and_exact_boot_without_writes() { + let fixture = Fixture::new().await; + let client = fixture.client().await; + let before = fixture.journal.load_snapshot(scope()).await.unwrap(); + let request = FleetSnapshotRequest::new( + before.clone(), + Digest::from_bytes([63; 32]), + endpoint(2).node, + endpoint(2).session, + FleetSnapshotSubject::Cells(None), + 128, + 0, + 5_000, + ) + .unwrap(); + client.authorize_snapshot(&request, 1).await.unwrap(); + assert_eq!( + fixture.journal.load_snapshot(scope()).await.unwrap(), + before + ); + assert!(client.authorize_snapshot(&request, 5_000).await.is_err()); + fixture + .journal + .rebind_active_intent(&intent(2), SessionId::from_bytes([64; 16]), 2) + .await + .unwrap(); + assert!(client.authorize_snapshot(&request, 2).await.is_err()); + let changed = client.load_snapshot(scope()).await.unwrap(); + assert_eq!(changed.head(), before.head()); + let old_boot = FleetSnapshotRequest::new( + changed.clone(), + Digest::from_bytes([65; 32]), + request.node(), + request.session(), + FleetSnapshotSubject::Host, + 1, + 0, + 5_000, + ) + .unwrap(); + assert!(client.authorize_snapshot(&old_boot, 2).await.is_err()); + let fresh = FleetSnapshotRequest::new( + changed, + Digest::from_bytes([66; 32]), + endpoint(2).node, + SessionId::from_bytes([64; 16]), + FleetSnapshotSubject::Host, + 1, + 0, + 5_000, + ) + .unwrap(); + client.authorize_snapshot(&fresh, 2).await.unwrap(); + fixture + .journal + .claim_controller( + scope(), + before.head().revision(), + before.head().controller().unwrap().claimant, + 2, + ) + .await + .unwrap(); + assert!(client.authorize_snapshot(&fresh, 2).await.is_err()); + client.close().await.unwrap(); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn complete_roster_traverses_multiple_pages_and_retains_every_status() { + let fixture = Fixture::new().await; + for n in 4..=132 { + fixture + .journal + .register_initial_intent(&intent(n)) + .await + .unwrap(); + } + let mut records = Vec::new(); + for n in 1..=130 { + let FleetEnrollmentAcceptance::New(mut record) = fixture + .journal + .accept_enrollment(&enrollment(n, 1), 1) + .await + .unwrap() + else { + panic!("duplicate fixture enrollment") + }; + match n % 4 { + 0 => { + record = fixture + .journal + .publish_enrollment_result( + &record, + EnrollmentEvent::Refused(Digest::from_bytes([210; 32])), + 2, + ) + .await + .unwrap() + } + 1 => {} + _ => { + record = fixture + .journal + .publish_enrollment_result( + &record, + EnrollmentEvent::Established(Digest::from_bytes([211; 32])), + 2, + ) + .await + .unwrap(); + if n % 4 == 3 { + record = fixture + .journal + .publish_enrollment_result( + &record, + EnrollmentEvent::Retired(Digest::from_bytes([212; 32])), + 3, + ) + .await + .unwrap(); + } + } + } + records.push(record); + } + records.sort_by_key(|record| *record.spec().key().unwrap().as_bytes()); + let snapshot = fixture.journal.load_snapshot(scope()).await.unwrap(); + let deadline = tokio::time::Instant::now() + std::time::Duration::from_secs(3); + let roster = FleetRoster::collect(&fixture.journal, &snapshot, deadline) + .await + .unwrap(); + assert_eq!(roster.snapshot(), &snapshot); + assert_eq!(roster.intents().len(), 132); + assert_eq!(roster.enrollments(), records); + assert_eq!( + roster.required_boots(), + vec![ + FleetRosterBoot { + node: endpoint(1).node, + session: endpoint(1).session + }, + FleetRosterBoot { + node: endpoint(3).node, + session: endpoint(3).session + } + ] + ); + roster.confirm(&fixture.journal, deadline).await.unwrap(); + let digest = roster.digest().unwrap(); + fixture.journal.close().await.unwrap(); + let reopened = fixture.client().await; + let again = FleetRoster::collect(&reopened, &snapshot, deadline) + .await + .unwrap(); + assert_eq!(again.enrollments(), records); + assert_eq!(again.digest().unwrap(), digest); + reopened.close().await.unwrap(); +} + +#[tokio::test] +async fn roster_barrier_rejects_independent_enrollment_commit_and_original_reply_loss() { + let fixture = Fixture::new().await; + let snapshot = fixture.journal.load_snapshot(scope()).await.unwrap(); + let deadline = tokio::time::Instant::now() + std::time::Duration::from_secs(3); + let roster = FleetRoster::collect(&fixture.journal, &snapshot, deadline) + .await + .unwrap(); + let client = fixture.client().await; + lose(&client); + let request = enrollment(233, 1); + assert!(client.accept_enrollment(&request, 1).await.is_err()); + let current = fixture.journal.load_snapshot(scope()).await.unwrap(); + assert_eq!(current.head(), snapshot.head()); + assert_ne!(current.registry(), snapshot.registry()); + assert!(roster.confirm(&fixture.journal, deadline).await.is_err()); + assert!( + FleetRoster::collect(&fixture.journal, &snapshot, deadline) + .await + .is_err() + ); + let fresh = FleetRoster::collect(&fixture.journal, ¤t, deadline) + .await + .unwrap(); + assert_eq!(fresh.enrollments().len(), 1); + assert_eq!(fresh.enrollments()[0].spec(), &request); + assert_eq!(fresh.enrollments()[0].status(), EnrollmentStatus::Pending); + assert_eq!(fresh.required_boots().len(), 2); + client.close().await.unwrap(); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn roster_barrier_rechecks_controller_head_even_with_unchanged_registry() { + let fixture = Fixture::new().await; + let snapshot = fixture.journal.load_snapshot(scope()).await.unwrap(); + let deadline = tokio::time::Instant::now() + std::time::Duration::from_secs(3); + let roster = FleetRoster::collect(&fixture.journal, &snapshot, deadline) + .await + .unwrap(); + let renewed = fixture + .journal + .claim_controller( + scope(), + snapshot.head().revision(), + snapshot.head().controller().unwrap().claimant, + 1, + ) + .await + .unwrap(); + assert_eq!(renewed.registry(), snapshot.registry()); + assert_ne!(renewed.head(), snapshot.head()); + assert!(roster.confirm(&fixture.journal, deadline).await.is_err()); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn unbootstrapped_roster_and_expired_collection_cannot_claim_coverage() { + let directory = tempfile::tempdir().unwrap(); + let journal = SqliteJournal::open( + directory.path().join("fleet.sqlite"), + scope(), + FleetProfile::default(), + 0, + ) + .await + .unwrap(); + let snapshot = journal.load_snapshot(scope()).await.unwrap(); + let roster = FleetRoster::collect( + &journal, + &snapshot, + tokio::time::Instant::now() + std::time::Duration::from_secs(3), + ) + .await + .unwrap(); + assert!(roster.enrollments().is_empty()); + assert!(roster.snapshot().registry().bootstrap_revision().is_none()); + assert!(!roster.covers_advertisements(&[], 0).unwrap()); + journal.close().await.unwrap(); + assert!(matches!( + FleetRoster::collect(&journal, &snapshot, tokio::time::Instant::now()).await, + Err(cellule_runtime::Error::Node( + "fleet roster collection deadline elapsed" + )) + )); +} diff --git a/crates/cellule-host/minion/journal/writer_inventory.rs b/crates/cellule-host/minion/journal/writer_inventory.rs new file mode 100644 index 00000000..b7a68f07 --- /dev/null +++ b/crates/cellule-host/minion/journal/writer_inventory.rs @@ -0,0 +1,167 @@ +use super::*; + +impl FleetOriginalWriterJournal for SqliteJournal { + fn persist_original_writers<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + record: &'a OriginalWriterInventoryRecord, + pages: &'a [OriginalWriterInventoryPage], + now_ms: i64, + ) -> FleetAdapterFuture<'a, OriginalWriterInventoryRecord> { + Box::pin(async move { + record.validate_pages(pages)?; + let expected = expected.clone(); + let record = record.clone(); + let pages = pages.to_vec(); + let stored = self + .run(move |db| { + let basis = record.basis(); + db.check_scope(basis.registry.scope())?; + let digest = record.digest()?; + if let Some(original) = + db.original_writers(basis.operation.id(), basis.process_request)? + { + if original != record { + return Err(OperationError::Conflict.into()); + } + for page in &pages { + if db.original_writer_page(page.digest()?)?.as_ref() != Some(page) { + return Err(OperationError::Conflict.into()); + } + } + // Immutable original set: no superseding pointer, timestamp + // refresh, provider retry or registry change on exact replay. + return Ok(original); + } + let current = db.snapshot()?; + let lease = current.head().controller().ok_or(OperationError::Fenced)?; + let intent = db.required_intent(basis.operation.node())?; + if current != expected + || basis.registry != expected.registry() + || basis.head_digest + != Digest::from_bytes( + *blake3::hash(&expected.head().to_bytes()?).as_bytes(), + ) + || current.head().maintenance() != Some(&basis.operation) + || current.registry().bootstrap_revision().is_none() + || basis.operation.phase() == MaintenancePhase::Completed + || intent.mode() != NodeMode::Draining + || intent.session() != basis.operation.session() + || intent.revision() != basis.operation.intent_revision() + || now_ms >= lease.expires_at_ms + || now_ms >= basis.operation.deadline_ms() + || now_ms < basis.interval.1 + || now_ms - basis.interval.0 > 30_000 + || db.enrollment(basis.boot.spec().key()?)?.as_ref() != Some(&basis.boot) + { + return Err(OperationError::Conflict.into()); + } + for page in &pages { + let key = page.digest()?; + if let Some(original) = db.original_writer_page(key)? { + if original != *page { + return Err(OperationError::Conflict.into()); + } + } else { + db.tx.execute( + "INSERT INTO original_writer_pages(key,body) VALUES(?1,?2)", + params![key.as_bytes().as_slice(), page.to_bytes()?], + )?; + } + } + db.tx.execute( + "INSERT INTO original_writers(operation,process,key,body) VALUES(?1,?2,?3,?4)", + params![ + basis.operation.id().as_bytes().as_slice(), + basis.process_request.as_bytes().as_slice(), + digest.as_bytes().as_slice(), + record.to_bytes()? + ], + )?; + db.advance_registry()?; + Ok(record) + }) + .await?; + #[cfg(test)] + self.original_writer_reply().await; + Ok(stored) + }) + } + fn original_writers<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + operation: OperationId, + process_request: Digest, + ) -> FleetAdapterFuture<'a, Option> { + let expected = expected.clone(); + Box::pin(self.run(move |db| { + db.check_scope(expected.head().scope())?; + if db.snapshot()? != expected { + return Err(OperationError::Conflict.into()); + } + db.original_writers(operation, process_request) + })) + } + fn original_writer_page( + &self, + scope: FleetScope, + digest: Digest, + ) -> FleetAdapterFuture<'_, Option> { + Box::pin(self.run(move |db| { + db.check_scope(scope)?; + db.original_writer_page(digest) + })) + } +} +impl Db<'_> { + fn original_writers( + &self, + operation: OperationId, + process_request: Digest, + ) -> JournalResult> { + let row = self + .tx + .query_row( + "SELECT key,body FROM original_writers WHERE operation=?1 AND process=?2", + params![ + operation.as_bytes().as_slice(), + process_request.as_bytes().as_slice() + ], + |row| Ok((blob(row, 0, 32)?, blob(row, 1, MAX_RECORD_BYTES)?)), + ) + .optional()?; + row.map(|(key, body)| -> JournalResult<_> { + let record = OriginalWriterInventoryRecord::from_bytes(&body)?; + self.check_scope(record.basis().registry.scope())?; + if record.basis().operation.id() != operation + || record.basis().process_request != process_request + || record.digest()?.as_bytes().as_slice() != key.as_slice() + { + return Err(OperationError::Conflict.into()); + } + Ok(record) + }) + .transpose() + } + fn original_writer_page( + &self, + digest: Digest, + ) -> JournalResult> { + let body = self + .tx + .query_row( + "SELECT body FROM original_writer_pages WHERE key=?1", + [digest.as_bytes().as_slice()], + |row| blob(row, 0, MAX_PAGE_BYTES), + ) + .optional()?; + body.map(|body| -> JournalResult<_> { + let page = OriginalWriterInventoryPage::from_bytes(&body)?; + if page.digest()? != digest { + return Err(OperationError::Conflict.into()); + } + Ok(page) + }) + .transpose() + } +} diff --git a/crates/cellule-host/minion/main.rs b/crates/cellule-host/minion/main.rs new file mode 100644 index 00000000..7746777a --- /dev/null +++ b/crates/cellule-host/minion/main.rs @@ -0,0 +1,83 @@ +//! Reference embedding with a durable journal and real leased-node overload. + +mod journal; +#[cfg(test)] +mod reconciler_tests; +mod scenario; + +use cellule_host::fleet::FleetJournal; +use cellule_runtime::fleet::operations::{FleetProfile, FleetScope}; +use cellule_runtime::identity::{ApplicationId, Digest}; +use journal::{JournalResult, SqliteJournal}; + +#[tokio::main] +async fn main() -> JournalResult<()> { + let mut args = std::env::args_os().skip(1); + let command = args.next(); + if matches!(command.as_deref(), Some(value) if value == "overload" || value == "controller-restart" || value == "balance") + && args.next().is_none() + { + let summary = if command.as_deref() == Some(std::ffi::OsStr::new("controller-restart")) { + scenario::controller_restart().await? + } else if command.as_deref() == Some(std::ffi::OsStr::new("balance")) { + scenario::count_balance().await? + } else { + scenario::overload().await? + }; + println!( + "released={} activated={} retired={} receipt_checks={} max_inflight={} max_restore_bytes={} joined_nodes={} boot_retirements={} receiver_nodes={} lost_release_replies={} controller_epoch={} expired_receiver_cleanups={} blocker_count={} final_counts={:?}", + summary.released, + summary.activated, + summary.retired, + summary.receipt_checks, + summary.max_inflight, + summary.max_restore_bytes, + summary.joined_nodes, + summary.boot_retirements, + summary.receiver_nodes, + summary.lost_release_replies, + summary.controller_epoch, + summary.expired_receiver_cleanups, + summary.blockers.len(), + summary.final_counts + ); + println!("blockers={:?}", summary.blockers); + return Ok(()); + } + let database = args.next(); + if command.as_deref() != Some(std::ffi::OsStr::new("inspect-journal")) + || database.is_none() + || args.next().is_some() + { + return Err(std::io::Error::other( + "usage: fleet_operations overload | controller-restart | balance | inspect-journal ", + ) + .into()); + } + let database = database.ok_or_else(|| std::io::Error::other("database path is absent"))?; + let scope = FleetScope { + fleet: Digest::from_bytes([200; 32]), + application: ApplicationId::from_bytes([3; 16]), + }; + let now_ms = i64::try_from( + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH)? + .as_millis(), + )?; + let journal = + SqliteJournal::open(database.into(), scope, FleetProfile::default(), now_ms).await?; + let result = journal.load_snapshot(scope).await; + let close = journal.close().await; + let snapshot = result?; + close?; + println!( + "journal revision={} registry revision={} bootstrapped={} scheduling={} unresolved={} restore_bytes={}", + snapshot.head().revision(), + snapshot.registry().revision(), + snapshot.registry().bootstrap_revision().is_some(), + snapshot.registry().scheduling_enabled(), + snapshot.head().attempts().len(), + snapshot.head().reserved_restore_bytes() + ); + Ok(()) +} diff --git a/crates/cellule-host/minion/reconciler_tests.rs b/crates/cellule-host/minion/reconciler_tests.rs new file mode 100644 index 00000000..45f9a2d0 --- /dev/null +++ b/crates/cellule-host/minion/reconciler_tests.rs @@ -0,0 +1,1776 @@ +//! Public driver sequencing with a real durable journal and synthetic effects. +//! These are adapter/driver tests, not leased-node or restoration qualification. + +use super::journal::SqliteJournal; +use cellule_host::fleet::*; +use cellule_runtime::cell::{actor::OwnedCellObservation, catalog::CatalogRole}; +use cellule_runtime::control::RootRef; +use cellule_runtime::fleet::operations::*; +use cellule_runtime::identity::*; +use cellule_runtime::node::*; +use ed25519_dalek::SigningKey; +use std::sync::{ + Arc, + atomic::{AtomicBool, AtomicUsize, Ordering}, +}; +use tokio::time::{Duration, Instant}; + +const NOW: i64 = 1_000_000; +fn scope() -> FleetScope { + FleetScope { + fleet: Digest::from_bytes([200; 32]), + application: ApplicationId::from_bytes([3; 16]), + } +} +fn node(n: u8) -> NodeId { + NodeId::from_bytes([n; 16]) +} +fn session(n: u8) -> SessionId { + SessionId::from_bytes([n + 10; 16]) +} +fn target(n: u8) -> CellTarget { + CellTarget::new( + TenantId::from_bytes([4; 16]), + scope().application, + NamespaceId::from_bytes([5; 16]), + &[n], + ) + .unwrap() +} +fn position(epoch: u64) -> PublishedPosition { + PublishedPosition { + incarnation: IncarnationId::from_bytes([6; 16]), + epoch, + root: RootRef { + digest: Digest::from_bytes([8; 32]), + txid: 9, + checksum: u64::MAX, + commit_sequence: 9, + }, + } +} + +struct Observer { + complete: bool, + omit_last_cell: AtomicBool, + capture_mutation: std::sync::Mutex>>, + calls: AtomicUsize, + pressured: bool, + draining: AtomicBool, + cell_state: std::sync::Mutex, +} + +#[derive(Clone, Copy)] +enum CellState { + Settled, + Busy, + Blob, + RoleBlocked, + NoCost, + NoPosition, +} +impl FleetObserver for Observer { + fn observe<'a>( + &'a self, + roster: &'a FleetRoster, + now_ms: i64, + _: Instant, + ) -> FleetAdapterFuture<'a, FleetObservation> { + Box::pin(async move { + let expected = roster.snapshot(); + let mutation = self.capture_mutation.lock().unwrap().take(); + if let Some(journal) = mutation { + journal + .accept_enrollment( + &EnrollmentSpec { + scope: scope(), + request: Digest::from_bytes([229; 32]), + role: EnrollmentRole::Follower { log_epoch: 99 }, + source: Some(EnrollmentEndpoint { + node: node(3), + session: session(3), + intent_revision: 1, + }), + target: EnrollmentEndpoint { + node: node(2), + session: session(2), + intent_revision: 1, + }, + }, + now_ms, + ) + .await?; + } + self.calls.fetch_add(1, Ordering::SeqCst); + let cell_state = *self.cell_state.lock().unwrap(); + let mut nodes = Vec::new(); + for n in 1..=3 { + let key = SigningKey::from_bytes(&[n; 32]); + let ad = NodeAdvertisement::sign( + node(n), + session(n), + format!("https://node-{n}.internal:8789"), + scope().fleet, + Digest::from_bytes([30; 32]), + Digest::from_bytes([31; 32]), + Digest::from_bytes([32; 32]), + &key, + 1, + now_ms, + now_ms + 30_000, + vec![Digest::from_bytes([9; 32])], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + free_memory_bytes: 1 << 30, + free_disk_bytes: 1 << 30, + job_credits: 16, + log_protocol: 1, + ..NodeCapacity::default() + }, + )? + .with_operational_placement( + NodePlacementCapacity { + memory_capacity_bytes: 2 << 30, + disk_capacity_bytes: 2 << 30, + active_cells: if n == 1 { 6 } else { 0 }, + max_active_cells: 16, + running_jobs: 0, + job_capacity: 16, + ..NodePlacementCapacity::default() + }, + NodeOperationalSample { + mode: if n == 1 && self.draining.load(Ordering::SeqCst) { + NodeMode::Draining + } else { + NodeMode::Active + }, + pressure: if n == 1 && self.pressured { + NodePressure::Shedding + } else { + NodePressure::Normal + }, + sequence: now_ms as u64, + observed_at_ms: now_ms, + }, + &key, + )?; + nodes.push(ad); + } + let mut cells = (1..=6) + .map(|n| { + let mut owned = FleetOwnedCell { + node: node(1), + session: session(1), + observation: OwnedCellObservation { + target: target(n), + generation: u64::from(n), + incarnation: position(1).incarnation, + code: Digest::from_bytes([9; 32]), + schema: 1, + role: CatalogRole::Sql, + resident_since_ms: now_ms - 120_000, + last_used_ms: now_ms - 120_000, + position: Some(position(1)), + cost: Some(TransferCost { + memory_bytes: 65536, + disk_bytes: 4096, + file_descriptors: 8, + job_credits: 1, + }), + maintenance_cost: Some(TransferCost { + memory_bytes: 65536, + disk_bytes: 8192, + file_descriptors: 8, + job_credits: 1, + }), + database_bytes: Some(4096), + sampled_at_ms: Some(now_ms), + stable_observations: 2, + work_blocker: None, + quiescing: false, + maintenance_work: None, + blockers: Vec::new(), + }, + }; + if !matches!(cell_state, CellState::Settled) { + let row = &mut owned.observation; + row.cost = None; + row.sampled_at_ms = None; + row.database_bytes = None; + row.stable_observations = 0; + row.blockers = + vec![DrainBlocker::BusyExecution, DrainBlocker::UnknownInventory]; + row.work_blocker = Some( + cellule_runtime::primitives::maintenance::TransferWorkClass::Queue, + ); + match cell_state { + CellState::Blob => row.role = CatalogRole::Blob, + CellState::RoleBlocked => { + row.blockers.push(DrainBlocker::FollowerObligation) + } + CellState::NoCost => row.maintenance_cost = None, + CellState::NoPosition => row.position = None, + CellState::Settled | CellState::Busy => {} + } + } + owned + }) + .collect::>(); + if self.omit_last_cell.load(Ordering::SeqCst) { + cells.pop(); + } + Ok(FleetObservation::new( + scope(), + expected.registry(), + 1, + now_ms, + now_ms, + self.complete, + nodes, + cells, + )?) + }) + } +} + +struct Transport { + journal: Arc, + lose_release: AtomicBool, + lose_cordon: AtomicBool, + unknown_cordon: AtomicBool, + unconfirmed_cordon: AtomicBool, + block_after_acceptance: AtomicBool, + block_before_acceptance: AtomicBool, + blocked_before_acceptance: AtomicUsize, + block_first_acceptance: AtomicBool, + reject_prepare: bool, + inspected: AtomicUsize, + dispatched: AtomicUsize, + released: AtomicUsize, + maintenance_released: AtomicUsize, + cordoned: AtomicUsize, +} +impl FleetTransport for Transport { + fn dispatch<'a>( + &'a self, + action: &'a FleetAction, + _: Instant, + ) -> FleetAdapterFuture<'a, Arc> { + Box::pin(async move { + if matches!(action.kind(), FleetActionKind::Maintenance { .. }) { + return self.dispatch_cordon(action).await; + } + let FleetActionKind::Movement { + action: effect, + attempt, + } = action.kind() + else { + panic!(); + }; + let spec = attempt.spec(); + let (origin, boot) = if effect.is_source_release() { + (spec.source_node, spec.source) + } else { + (spec.destination_node, spec.destination) + }; + // The real journal rejects dispatch before its phase CAS. Repeated + // dispatch returns the same committed synthetic effect record. + if self.block_before_acceptance.load(Ordering::SeqCst) { + self.blocked_before_acceptance + .fetch_add(1, Ordering::SeqCst); + std::future::pending::<()>().await; + } + let accepted = match self + .journal + .accept_action(action, origin, boot, action.issued_at_ms()) + .await? + { + FleetActionAcceptance::New(accepted) => accepted, + FleetActionAcceptance::Existing { + accepted, + result: Some(outcome), + } => { + return Ok(Arc::new(FleetActionCompletion { + accepted, + outcome: *outcome, + committed: true, + execution_error: None, + journal_error: None, + })); + } + FleetActionAcceptance::Existing { .. } => { + panic!("synthetic action must have retained outcome") + } + }; + self.dispatched.fetch_add(1, Ordering::SeqCst); + if self.block_after_acceptance.load(Ordering::SeqCst) + || self.block_first_acceptance.swap(false, Ordering::SeqCst) + { + // Model an accepted endpoint whose effect has no confirmed + // result. Dropping this transport leaves its durable acceptance. + std::future::pending::<()>().await; + } + let outcome = match effect { + MovementAction::Prepare if self.reject_prepare => { + FleetOutcome::Rejected(DrainBlocker::ReceiverCapacity) + } + MovementAction::Prepare => FleetOutcome::Reserved(ReceiverReservation { + session: boot, + expires_at_ms: spec.deadline_ms, + }), + MovementAction::Release | MovementAction::ReleaseMaintenance => { + if *effect == MovementAction::ReleaseMaintenance { + self.maintenance_released.fetch_add(1, Ordering::SeqCst); + } + self.released.fetch_add(1, Ordering::SeqCst); + FleetOutcome::Released(position(spec.source_epoch)) + } + MovementAction::Activate => { + let mut idle = cellule_runtime::control::Control::initial( + spec.target.cell_id(), + spec.incarnation, + cellule_runtime::control::Owner { + session: spec.source, + endpoint: "https://source.internal:8789".into(), + }, + Digest::from_bytes([9; 32]), + 1, + )?; + idle.state = cellule_runtime::control::ControlState::Idle; + idle.owner = None; + idle.epoch = spec.source_epoch; + idle.root = Some(position(spec.source_epoch).root); + idle.revision = 9; + idle.progress = 9; + let basis = + AcquisitionBasis::new(accepted.clone(), idle, action.issued_at_ms())?; + self.journal.record_acquisition_basis(&basis).await?; + FleetOutcome::Activated(ActivationEvidence { + node: origin, + session: boot, + position: position(spec.source_epoch + 1), + }) + } + MovementAction::Cancel => FleetOutcome::ReceiverCleaned, + _ => panic!("unsupported synthetic effect"), + }; + let outcome = FleetActionOutcome { + scope: scope(), + action_key: action.key()?, + node: origin, + session: boot, + observed_at_ms: action.issued_at_ms(), + outcome, + }; + self.journal + .publish_action_result(&accepted, &outcome) + .await?; + if effect.is_source_release() && self.lose_release.swap(false, Ordering::SeqCst) { + return Err(std::io::Error::other( + "injected lost release response after durable result", + ) + .into()); + } + Ok(Arc::new(FleetActionCompletion { + accepted, + outcome, + committed: true, + execution_error: None, + journal_error: None, + })) + }) + } + fn inspect<'a>( + &'a self, + request: &'a FleetInspectionRequest, + _: Instant, + ) -> FleetAdapterFuture<'a, Arc> { + Box::pin(async move { + self.journal + .authorize_inspection(request, request.action().issued_at_ms()) + .await?; + self.inspected.fetch_add(1, Ordering::SeqCst); + let FleetActionKind::Movement { attempt, .. } = request.action().kind() else { + panic!(); + }; + let effect = match attempt.phase() { + AttemptPhase::Preparing => MovementAction::Prepare, + AttemptPhase::Releasing => MovementAction::Release, + AttemptPhase::MaintenanceReleasing => MovementAction::ReleaseMaintenance, + AttemptPhase::Activating | AttemptPhase::Activated => MovementAction::Activate, + AttemptPhase::Cancelling | AttemptPhase::CleaningReceiver => MovementAction::Cancel, + _ => panic!("unsupported synthetic inspection"), + }; + let outcome = match self + .journal + .load_movement_action( + scope(), + attempt.spec().id, + effect, + request.node(), + request.session(), + ) + .await? + { + Some(FleetActionAcceptance::Existing { + result: Some(result), + .. + }) => result.outcome.clone(), + None => FleetOutcome::Unknown, + _ => panic!("synthetic action result absent"), + }; + let at = request.action().issued_at_ms(); + Ok(Arc::new(FleetInspectionObservation::new( + request.clone(), + at, + FleetActionOutcome { + scope: scope(), + action_key: request.action().key()?, + node: request.node(), + session: request.session(), + observed_at_ms: at, + outcome, + }, + )?)) + }) + } +} + +impl Transport { + async fn dispatch_cordon( + &self, + action: &FleetAction, + ) -> std::result::Result, Box> + { + let FleetActionKind::Maintenance { + action: MaintenanceAction::Cordon, + operation, + } = action.kind() + else { + panic!("synthetic transport only implements cordon"); + }; + let accepted = match self + .journal + .accept_action( + action, + operation.node(), + operation.session(), + action.issued_at_ms(), + ) + .await? + { + FleetActionAcceptance::New(accepted) => { + self.dispatched.fetch_add(1, Ordering::SeqCst); + accepted + } + FleetActionAcceptance::Existing { + accepted, + result: Some(outcome), + } if !matches!(outcome.outcome, FleetOutcome::Unknown) => { + return Ok(Arc::new(FleetActionCompletion { + accepted, + outcome: *outcome, + committed: true, + execution_error: None, + journal_error: None, + })); + } + FleetActionAcceptance::Existing { accepted, .. } => accepted, + }; + if self.block_after_acceptance.load(Ordering::SeqCst) + || self.block_first_acceptance.swap(false, Ordering::SeqCst) + { + std::future::pending::<()>().await; + } + let unknown = self.unknown_cordon.swap(false, Ordering::SeqCst); + if !unknown { + self.cordoned.fetch_add(1, Ordering::SeqCst); + } + let outcome = FleetActionOutcome { + scope: scope(), + action_key: action.key()?, + node: operation.node(), + session: operation.session(), + observed_at_ms: action.issued_at_ms(), + outcome: if unknown { + FleetOutcome::Unknown + } else { + FleetOutcome::Cordoned + }, + }; + self.journal + .publish_action_result(&accepted, &outcome) + .await?; + if self.lose_cordon.swap(false, Ordering::SeqCst) { + return Err(std::io::Error::other( + "injected lost cordon response after durable result", + ) + .into()); + } + let unconfirmed = self.unconfirmed_cordon.swap(false, Ordering::SeqCst); + Ok(Arc::new(FleetActionCompletion { + accepted, + outcome, + committed: !unconfirmed, + execution_error: unknown.then(|| { + Arc::new(cellule_runtime::Error::Control( + "injected unresolved cordon execution", + )) + }), + journal_error: unconfirmed.then(|| { + Arc::new(cellule_runtime::Error::Facility { + name: "injected-cordon-publication", + source: Box::new(std::io::Error::other("injected unconfirmed result")), + }) + }), + })) + } +} + +struct Fixture { + _root: tempfile::TempDir, + journal: Arc, + observer: Arc, + transport: Arc, +} +impl Fixture { + async fn new(complete: bool, pressured: bool, reject_prepare: bool) -> Self { + let root = tempfile::tempdir().unwrap(); + let journal = Arc::new( + SqliteJournal::open( + root.path().join("fleet.sqlite"), + scope(), + FleetProfile::default(), + 0, + ) + .await + .unwrap(), + ); + for n in 1..=3 { + journal + .register_initial_intent( + &NodeIntent::initial(scope(), node(n), session(n)).unwrap(), + ) + .await + .unwrap(); + // Synthetic driver evidence: these tests exercise orchestration, + // not the native boot authority qualified by scenario/startup. + let boot = EnrollmentSpec { + scope: scope(), + request: Digest::from_bytes([n + 128; 32]), + role: EnrollmentRole::Node { + mode: NodeMode::Active, + }, + source: None, + target: EnrollmentEndpoint { + node: node(n), + session: session(n), + intent_revision: 1, + }, + }; + let FleetEnrollmentAcceptance::New(record) = + journal.accept_enrollment(&boot, 0).await.unwrap() + else { + panic!("boot already accepted") + }; + journal + .publish_enrollment_result( + &record, + EnrollmentEvent::Established(Digest::from_bytes([n + 160; 32])), + 0, + ) + .await + .unwrap(); + } + let version = journal.load_snapshot(scope()).await.unwrap().registry(); + let version = journal.bootstrap_registry(version).await.unwrap(); + journal.set_scheduling(version, true).await.unwrap(); + let observer = Arc::new(Observer { + complete, + omit_last_cell: AtomicBool::new(false), + capture_mutation: std::sync::Mutex::new(None), + calls: AtomicUsize::new(0), + pressured, + draining: AtomicBool::new(false), + cell_state: std::sync::Mutex::new(CellState::Settled), + }); + let transport = Arc::new(Transport { + journal: journal.clone(), + lose_release: AtomicBool::new(false), + lose_cordon: AtomicBool::new(false), + unknown_cordon: AtomicBool::new(false), + unconfirmed_cordon: AtomicBool::new(false), + block_after_acceptance: AtomicBool::new(false), + block_before_acceptance: AtomicBool::new(false), + blocked_before_acceptance: AtomicUsize::new(0), + block_first_acceptance: AtomicBool::new(false), + reject_prepare, + inspected: AtomicUsize::new(0), + dispatched: AtomicUsize::new(0), + released: AtomicUsize::new(0), + maintenance_released: AtomicUsize::new(0), + cordoned: AtomicUsize::new(0), + }); + Self { + _root: root, + journal, + observer, + transport, + } + } + fn driver(&self, claimant: u8) -> FleetReconciler { + FleetReconciler::new( + scope(), + SessionId::from_bytes([claimant; 16]), + FleetProfile::default(), + self.journal.clone(), + self.observer.clone(), + self.transport.clone(), + ) + .unwrap() + } + async fn step(&self, driver: &FleetReconciler, index: i64) -> FleetReconcileReport { + driver + .reconcile_once( + || Ok(NOW + index * 100), + Instant::now() + Duration::from_secs(3), + ) + .await + .unwrap() + } + async fn stop(&self) { + let version = self + .journal + .load_snapshot(scope()) + .await + .unwrap() + .registry(); + self.journal.set_scheduling(version, false).await.unwrap(); + } + + async fn request_maintenance(&self, deadline_ms: i64) -> MaintenanceOperation { + let snapshot = self.journal.load_snapshot(scope()).await.unwrap(); + let snapshot = self + .journal + .claim_controller( + scope(), + snapshot.head().revision(), + SessionId::from_bytes([206; 16]), + NOW, + ) + .await + .unwrap(); + let operation = MaintenanceOperation::new( + OperationId::from_bytes([207; 16]).unwrap(), + Digest::from_bytes([208; 32]), + node(1), + session(1), + 2, + NOW, + deadline_ms, + ) + .unwrap(); + self.journal + .compare_exchange( + &snapshot, + snapshot.head().controller().unwrap().epoch, + NOW, + &JournalTransition::BeginMaintenance(operation.clone()), + ) + .await + .unwrap(); + operation + } +} + +#[tokio::test] +async fn maintenance_cordon_precedes_partial_normal_pressure_evacuation() { + let fixture = Fixture::new(false, false, false).await; + let operation = fixture.request_maintenance(NOW + 20_000).await; + let driver = fixture.driver(206); + let cordoned = fixture.step(&driver, 1).await; + assert_eq!( + cordoned.snapshot.head().maintenance().unwrap().phase(), + MaintenancePhase::Cordoned + ); + assert_eq!(cordoned.allocated, 0); + assert_eq!(cordoned.dispatched, 1); + assert!(cordoned.maintenance_failure.is_none()); + let evacuation = fixture.step(&driver, 2).await; + assert_eq!( + evacuation.snapshot.head().maintenance().unwrap().phase(), + MaintenancePhase::Evacuating + ); + assert_eq!(evacuation.allocated, 2); + for attempt in evacuation.snapshot.head().attempts() { + assert_eq!(attempt.spec().id.operation, operation.id()); + assert_eq!(attempt.spec().source_node, node(1)); + assert_ne!(attempt.spec().destination_node, node(1)); + assert_eq!(attempt.spec().deadline_ms, operation.deadline_ms()); + } + fixture.stop().await; + for index in 3..=6 { + fixture.step(&driver, index).await; + } + let retained = fixture.step(&driver, 7).await; + assert!(retained.snapshot.head().attempts().is_empty()); + assert_eq!( + retained.snapshot.head().maintenance().unwrap().phase(), + MaintenancePhase::Evacuating + ); + assert!( + retained + .blockers + .contains(&DrainBlocker::IncompleteObservation) + ); + assert_eq!(fixture.transport.cordoned.load(Ordering::SeqCst), 1); + assert_eq!( + fixture + .transport + .maintenance_released + .load(Ordering::SeqCst), + 2 + ); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn stopped_optional_scheduling_still_cordons_and_retains_maintenance() { + let fixture = Fixture::new(false, false, false).await; + fixture.request_maintenance(NOW + 20_000).await; + fixture.stop().await; + let driver = fixture.driver(206); + let cordoned = fixture.step(&driver, 1).await; + assert_eq!( + cordoned.snapshot.head().maintenance().unwrap().phase(), + MaintenancePhase::Cordoned + ); + let evacuation = fixture.step(&driver, 2).await; + assert_eq!( + evacuation.snapshot.head().maintenance().unwrap().phase(), + MaintenancePhase::Evacuating + ); + assert_eq!(evacuation.allocated, 0); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn busy_maintenance_plans_peak_cost_and_adopts_lost_explicit_release() { + let fixture = Fixture::new(false, false, false).await; + *fixture.observer.cell_state.lock().unwrap() = CellState::Busy; + fixture.request_maintenance(NOW + 20_000).await; + let driver = fixture.driver(206); + let cordoned = fixture.step(&driver, 1).await; + assert_eq!(cordoned.allocated, 0); + let planned = fixture.step(&driver, 2).await; + assert_eq!(planned.allocated, 2); + assert_eq!(planned.snapshot.head().reserved_restore_bytes(), 16_384); + let specs = planned + .snapshot + .head() + .attempts() + .iter() + .map(|attempt| attempt.spec().clone()) + .collect::>(); + assert!(specs.iter().all(|spec| spec.cost.disk_bytes == 8192)); + fixture.stop().await; + fixture.step(&driver, 3).await; + fixture.transport.lose_release.store(true, Ordering::SeqCst); + let unknown = fixture.step(&driver, 4).await; + assert_eq!(unknown.released, 1); + assert_eq!(unknown.failures.len(), 1); + assert!(unknown.snapshot.head().attempts().iter().any(|attempt| { + attempt.phase() == AttemptPhase::MaintenanceReleasing + && attempt.released().is_none() + && unknown + .snapshot + .head() + .retirement_page(&[attempt.spec().id]) + .is_err() + })); + assert_eq!(unknown.snapshot.head().reserved_restore_bytes(), 16_384); + let reconstructed = fixture.driver(206); + for index in 5..=8 { + fixture.step(&reconstructed, index).await; + } + assert_eq!( + fixture + .transport + .maintenance_released + .load(Ordering::SeqCst), + 2 + ); + assert_eq!(fixture.transport.released.load(Ordering::SeqCst), 2); + let snapshot = fixture.journal.load_snapshot(scope()).await.unwrap(); + assert!(snapshot.head().attempts().is_empty()); + assert_eq!(snapshot.head().reserved_restore_bytes(), 0); + assert_eq!( + snapshot.head().maintenance().unwrap().phase(), + MaintenancePhase::Evacuating + ); + for spec in specs { + let record = fixture + .journal + .load_movement_action( + scope(), + spec.id, + MovementAction::ReleaseMaintenance, + spec.source_node, + spec.source, + ) + .await + .unwrap() + .unwrap(); + assert!( + matches!(record, FleetActionAcceptance::Existing { result: Some(ref result), .. } + if matches!(result.outcome, FleetOutcome::Released(_))) + ); + assert!( + fixture + .journal + .load_movement_action( + scope(), + spec.id, + MovementAction::Release, + spec.source_node, + spec.source + ) + .await + .unwrap() + .is_none() + ); + } + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn draining_advertisement_or_pressure_alone_cannot_plan_busy_work() { + for draining in [false, true] { + let fixture = Fixture::new(true, true, false).await; + *fixture.observer.cell_state.lock().unwrap() = CellState::Busy; + fixture.observer.draining.store(draining, Ordering::SeqCst); + let report = fixture.step(&fixture.driver(206), 1).await; + assert_eq!(report.allocated, 0); + assert!(report.snapshot.head().attempts().is_empty()); + assert_eq!(fixture.transport.released.load(Ordering::SeqCst), 0); + fixture.journal.close().await.unwrap(); + } +} + +#[tokio::test] +async fn maintenance_planning_keeps_blob_role_and_missing_evidence_blocked() { + for state in [ + CellState::Blob, + CellState::RoleBlocked, + CellState::NoCost, + CellState::NoPosition, + ] { + let fixture = Fixture::new(false, false, false).await; + *fixture.observer.cell_state.lock().unwrap() = state; + fixture.request_maintenance(NOW + 20_000).await; + let driver = fixture.driver(206); + fixture.step(&driver, 1).await; + let report = fixture.step(&driver, 2).await; + assert_eq!(report.allocated, 0); + assert!(report.snapshot.head().attempts().is_empty()); + assert_eq!(fixture.transport.released.load(Ordering::SeqCst), 0); + fixture.journal.close().await.unwrap(); + } +} + +#[tokio::test] +async fn replacement_controller_adopts_lost_cordon_reply_without_repeating_effect() { + let fixture = Fixture::new(false, false, false).await; + let operation = fixture.request_maintenance(NOW + 60_000).await; + fixture.stop().await; + fixture.transport.lose_cordon.store(true, Ordering::SeqCst); + let lost = fixture.step(&fixture.driver(206), 1).await; + assert!(lost.maintenance_failure.is_some()); + assert!(lost.failures.is_empty()); + assert_eq!( + lost.snapshot.head().maintenance().unwrap().phase(), + MaintenancePhase::Requested + ); + assert_eq!( + lost.next_wake_at_ms, + NOW + 100 + FleetProfile::default().reconcile_interval_ms + ); + let replaced = fixture.step(&fixture.driver(209), 310).await; + assert_eq!(replaced.snapshot.head().controller().unwrap().epoch, 2); + assert_eq!( + replaced.snapshot.head().maintenance().unwrap().id(), + operation.id() + ); + assert_eq!( + replaced.snapshot.head().maintenance().unwrap().phase(), + MaintenancePhase::Cordoned + ); + assert!(replaced.maintenance_failure.is_none()); + assert_eq!(fixture.transport.cordoned.load(Ordering::SeqCst), 1); + assert_eq!(fixture.transport.dispatched.load(Ordering::SeqCst), 1); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn timed_out_cordon_retains_acceptance_and_retries_without_false_evacuation() { + let fixture = Fixture::new(false, false, false).await; + fixture.request_maintenance(NOW + 20_000).await; + fixture.stop().await; + fixture + .transport + .block_first_acceptance + .store(true, Ordering::SeqCst); + let driver = fixture.driver(206); + let report = first_acceptance_deadline_at( + &driver, + &fixture.transport.dispatched, + Duration::from_millis(100), + 2, + ) + .await; + assert!(report.maintenance_failure.is_some()); + let cellule_runtime::Error::Facility { name, source } = + report.maintenance_failure.as_ref().unwrap().as_ref() + else { + panic!("cordon timeout source lost"); + }; + assert_eq!(*name, "fleet-controller-deadline"); + assert!(source.is::()); + assert_eq!(report.dispatched, 1); + assert_eq!( + report.snapshot.head().maintenance().unwrap().phase(), + MaintenancePhase::Requested + ); + assert_eq!(fixture.transport.cordoned.load(Ordering::SeqCst), 0); + let next = fixture.step(&fixture.driver(206), 2).await; + assert_eq!( + next.snapshot.head().maintenance().unwrap().phase(), + MaintenancePhase::Cordoned + ); + assert_eq!(fixture.transport.cordoned.load(Ordering::SeqCst), 1); + assert_eq!(fixture.transport.dispatched.load(Ordering::SeqCst), 1); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn unconfirmed_or_unknown_cordon_retains_phase_and_original_error() { + for unknown in [false, true] { + let fixture = Fixture::new(false, false, false).await; + fixture.request_maintenance(NOW + 20_000).await; + fixture.stop().await; + if unknown { + fixture + .transport + .unknown_cordon + .store(true, Ordering::SeqCst); + } else { + fixture + .transport + .unconfirmed_cordon + .store(true, Ordering::SeqCst); + } + let driver = fixture.driver(206); + let report = fixture.step(&driver, 1).await; + assert_eq!( + report.snapshot.head().maintenance().unwrap().phase(), + MaintenancePhase::Requested + ); + assert_eq!(report.allocated, 0); + assert_eq!( + report.next_wake_at_ms, + NOW + 100 + FleetProfile::default().reconcile_interval_ms + ); + let error = report.maintenance_failure.unwrap(); + if unknown { + assert!(matches!( + error.as_ref(), + cellule_runtime::Error::Control("injected unresolved cordon execution") + )); + assert!(report.blockers.contains(&DrainBlocker::OutcomeUnknown)); + } else { + let cellule_runtime::Error::Facility { name, source } = error.as_ref() else { + panic!("publication source lost"); + }; + assert_eq!(*name, "injected-cordon-publication"); + assert!(source.is::()); + assert!(report.blockers.contains(&DrainBlocker::PendingPublication)); + } + let replay = fixture.step(&fixture.driver(206), 2).await; + assert_eq!( + replay.snapshot.head().maintenance().unwrap().phase(), + MaintenancePhase::Cordoned + ); + assert!(replay.maintenance_failure.is_none()); + assert_eq!(fixture.transport.cordoned.load(Ordering::SeqCst), 1); + assert_eq!(fixture.transport.dispatched.load(Ordering::SeqCst), 1); + fixture.journal.close().await.unwrap(); + } +} + +#[tokio::test] +async fn expired_maintenance_still_closes_admission_but_allocates_no_new_moves() { + let fixture = Fixture::new(false, false, false).await; + fixture.request_maintenance(NOW + 100).await; + let driver = fixture.driver(206); + let cordoned = fixture.step(&driver, 2).await; + assert_eq!( + cordoned.snapshot.head().maintenance().unwrap().phase(), + MaintenancePhase::Cordoned + ); + assert!(cordoned.blockers.contains(&DrainBlocker::Deadline)); + let evacuation = fixture.step(&driver, 3).await; + assert_eq!(evacuation.allocated, 0); + assert!(evacuation.blockers.contains(&DrainBlocker::Deadline)); + assert_eq!( + evacuation.snapshot.head().maintenance().unwrap().phase(), + MaintenancePhase::Evacuating + ); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn public_reconciler_drives_durable_phases_fresh_activation_cleanup_and_retirement() { + let fixture = Fixture::new(true, false, false).await; + let driver = fixture.driver(206); + let first = fixture.step(&driver, 0).await; + assert_eq!(first.allocated, 2); + assert_eq!(first.dispatched, 0); + assert!(first.next_wake_at_ms < NOW + 100); + assert_eq!(first.snapshot.head().reserved_restore_bytes(), 8192); + fixture.stop().await; + assert_eq!(fixture.step(&driver, 1).await.dispatched, 2); // prepare + let released = fixture.step(&driver, 2).await; + assert_eq!(released.dispatched, 2); + assert_eq!(released.released, 2); + assert_eq!(released.activated, 0); + let activated = fixture.step(&driver, 3).await; + assert_eq!(activated.inspected, 2); + assert_eq!(activated.activated, 2); + assert_eq!(activated.released, 0); + assert!( + activated + .snapshot + .head() + .attempts() + .iter() + .all(|a| a.phase() == AttemptPhase::Activated) + ); + assert_eq!(fixture.step(&driver, 4).await.dispatched, 2); // independent resource proof + let retired = fixture.step(&fixture.driver(206), 5).await; + assert_eq!(retired.retired, 2); + assert_eq!(retired.inspected, 2); + assert_eq!(retired.activated, 0); + assert!(retired.snapshot.head().attempts().is_empty()); + let page = fixture + .journal + .load_progress(scope(), retired.snapshot.head().progress().unwrap().digest) + .await + .unwrap() + .unwrap(); + let latest = fixture + .journal + .last_movement_at(&retired.snapshot) + .await + .unwrap() + .unwrap(); + assert!((NOW + 300..NOW + 400).contains(&latest)); + for entry in page.entries() { + assert_eq!( + fixture + .journal + .last_moved_at( + &retired.snapshot, + entry.spec().target.cell_id(), + entry.spec().incarnation + ) + .await + .unwrap(), + entry.completed_at_ms() + ); + } + assert_eq!(fixture.observer.calls.load(Ordering::SeqCst), 1); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn reconstructed_reconciler_consumes_lost_release_reply_without_repeating_effect() { + let fixture = Fixture::new(true, false, false).await; + let driver = fixture.driver(206); + fixture.step(&driver, 0).await; + fixture.stop().await; + fixture.step(&driver, 1).await; + fixture.transport.lose_release.store(true, Ordering::SeqCst); + let report = driver + .reconcile_once(|| Ok(NOW + 200), Instant::now() + Duration::from_secs(3)) + .await + .unwrap(); + assert_eq!(report.failures.len(), 1); + assert!( + report.failures[0] + .error + .to_string() + .contains("fleet-transport") + ); + assert_eq!(report.released, 1); // Healthy sibling advances after the lost reply. + let snapshot = fixture.journal.load_snapshot(scope()).await.unwrap(); + assert_eq!(snapshot.head().attempts().len(), 2); + assert_eq!(snapshot.head().reserved_restore_bytes(), 8192); + let next = fixture.step(&fixture.driver(206), 3).await; + assert_eq!(next.inspected, 2); + assert_eq!(next.dispatched, 1); + assert!( + next.snapshot + .head() + .attempts() + .iter() + .all(|a| matches!(a.phase(), AttemptPhase::Released | AttemptPhase::Activated)) + ); + assert_eq!(fixture.transport.dispatched.load(Ordering::SeqCst), 5); + assert_eq!(fixture.transport.released.load(Ordering::SeqCst), 2); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn partial_roster_blocks_optional_count_moves_but_allows_pressure_relief() { + let partial = Fixture::new(false, false, false).await; + let report = partial.step(&partial.driver(206), 0).await; + assert_eq!(report.allocated, 0); + assert!( + report + .blockers + .contains(&DrainBlocker::IncompleteObservation) + ); + partial.journal.close().await.unwrap(); + let urgent = Fixture::new(false, true, false).await; + assert_eq!(urgent.step(&urgent.driver(206), 0).await.allocated, 2); + urgent.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn definite_prepare_refusal_cancels_without_release_and_cancellation_is_not_movement() { + let fixture = Fixture::new(true, false, true).await; + let driver = fixture.driver(206); + fixture.step(&driver, 0).await; + fixture.stop().await; + let report = fixture.step(&driver, 1).await; + assert!( + report + .snapshot + .head() + .attempts() + .iter() + .all(|a| a.phase() == AttemptPhase::Cancelled) + ); + assert_eq!(report.cancelled, 2); + assert!(report.blockers.contains(&DrainBlocker::ReceiverCapacity)); + assert!(report.next_wake_at_ms >= NOW + 100 + FleetProfile::default().reconcile_interval_ms); + let retired = fixture.step(&driver, 2).await; + assert_eq!(retired.retired, 2); + assert_eq!(fixture.transport.dispatched.load(Ordering::SeqCst), 2); + assert_eq!( + fixture + .journal + .last_movement_at(&retired.snapshot) + .await + .unwrap(), + None + ); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn competing_controllers_are_fenced_and_stale_history_barriers_conflict() { + let fixture = Fixture::new(true, false, false).await; + let report = fixture.step(&fixture.driver(206), 0).await; + assert!( + fixture + .driver(207) + .reconcile_once(|| Ok(NOW + 100), Instant::now() + Duration::from_secs(3)) + .await + .is_err() + ); + fixture.stop().await; + assert!( + fixture + .journal + .last_movement_at(&report.snapshot) + .await + .is_err() + ); + assert_eq!( + fixture + .journal + .load_snapshot(scope()) + .await + .unwrap() + .head() + .attempts() + .len(), + 2 + ); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn racing_public_drivers_allocate_one_shared_batch() { + let fixture = Fixture::new(true, false, false).await; + let a = fixture.driver(206); + let b = fixture.driver(207); + let (a, b) = tokio::join!( + a.reconcile_once(|| Ok(NOW), Instant::now() + Duration::from_secs(3)), + b.reconcile_once(|| Ok(NOW), Instant::now() + Duration::from_secs(3)), + ); + assert_ne!(a.is_ok(), b.is_ok()); + let retained = fixture.journal.load_snapshot(scope()).await.unwrap(); + assert_eq!(retained.head().attempts().len(), 2); + assert_eq!(retained.head().reserved_restore_bytes(), 8192); + assert_eq!(fixture.transport.dispatched.load(Ordering::SeqCst), 0); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn expired_never_dispatched_plan_cancels_and_retires_without_remote_effect() { + let fixture = Fixture::new(true, false, false).await; + let driver = fixture.driver(206); + fixture.step(&driver, 0).await; + fixture.stop().await; + let cancelled = fixture.step(&fixture.driver(207), 310).await; + assert!( + cancelled + .snapshot + .head() + .attempts() + .iter() + .all(|a| a.phase() == AttemptPhase::Cancelled) + ); + assert_eq!(cancelled.snapshot.head().controller().unwrap().epoch, 2); + assert_eq!(cancelled.dispatched, 0); + let retired = fixture.step(&fixture.driver(207), 311).await; + assert_eq!(retired.retired, 2); + assert_eq!(fixture.transport.dispatched.load(Ordering::SeqCst), 0); + fixture.journal.close().await.unwrap(); +} + +// These adapter models test a retained accepted endpoint, rather than SQLite +// setup latency. Advance the original timer only after that endpoint accepted. +async fn first_acceptance_deadline_at( + driver: &FleetReconciler, + accepted: &AtomicUsize, + duration: Duration, + shares: u32, +) -> FleetReconcileReport { + tokio::time::pause(); + assert_eq!(accepted.load(Ordering::SeqCst), 0); + let deadline = Instant::now() + duration; + let mut pass = Box::pin(driver.reconcile_once(|| Ok(NOW + 100), deadline)); + let wall_deadline = std::time::Instant::now() + Duration::from_secs(3); + loop { + assert!(futures_util::poll!(pass.as_mut()).is_pending()); + if accepted.load(Ordering::SeqCst) == 1 { + break; + } + assert!( + std::time::Instant::now() < wall_deadline, + "first endpoint acceptance not reached" + ); + tokio::task::yield_now().await; + } + let budget = deadline.saturating_duration_since(Instant::now()) / shares; + tokio::time::advance(budget + Duration::from_millis(1)).await; + // Keep the executor runnable while the retained journal worker and healthy + // sibling return. Idle-clock advancement must not create a second fault. + let wall_deadline = std::time::Instant::now() + Duration::from_secs(3); + let result = loop { + if let std::task::Poll::Ready(result) = futures_util::poll!(pass.as_mut()) { + break result; + } + assert!( + std::time::Instant::now() < wall_deadline, + "first-endpoint deadline did not settle its report" + ); + tokio::task::yield_now().await; + }; + tokio::time::resume(); + result.unwrap() +} + +// Fault timing is relative to an observed protocol boundary. Freeze only this +// current-thread model clock after setup; native/process SLO clocks are unchanged. +async fn transport_deadline_at( + driver: &FleetReconciler, + reached: &AtomicUsize, +) -> FleetReconcileReport { + tokio::time::pause(); + assert_eq!(reached.load(Ordering::SeqCst), 0); + let deadline = Instant::now() + Duration::from_millis(150); + let mut pass = Box::pin(driver.reconcile_once(|| Ok(NOW + 100), deadline)); + // The driver visits two endpoints sequentially, keeping one share for + // planning. Freeze I/O time and expire each original partition only after + // its selected fault boundary; one millisecond covers timer quantization. + for (boundary, shares) in [(1, 3), (2, 2)] { + let budget = deadline.saturating_duration_since(Instant::now()) / shares; + let wall_deadline = std::time::Instant::now() + Duration::from_secs(3); + loop { + assert!(futures_util::poll!(pass.as_mut()).is_pending()); + if reached.load(Ordering::SeqCst) == boundary { + break; + } + assert!( + std::time::Instant::now() < wall_deadline, + "transport fault boundary not reached" + ); + // Keep the model runnable while SQLite's original worker completes; + // the paused deadline cannot auto-advance ahead of the selected fault. + tokio::task::yield_now().await; + } + tokio::time::advance(budget + Duration::from_millis(1)).await; + } + let report = pass.await.unwrap(); + tokio::time::resume(); + report +} + +#[tokio::test] +async fn transport_deadline_preserves_accepted_phase_permits_and_original_timeout() { + let fixture = Fixture::new(true, false, false).await; + let driver = fixture.driver(206); + fixture.step(&driver, 0).await; + fixture.stop().await; + fixture + .transport + .block_after_acceptance + .store(true, Ordering::SeqCst); + let report = transport_deadline_at(&driver, &fixture.transport.dispatched).await; + assert_eq!(report.failures.len(), 2); + for failure in &report.failures { + let cellule_runtime::Error::Facility { name, source } = failure.error.as_ref() else { + panic!("timeout source lost"); + }; + assert_eq!(*name, "fleet-controller-deadline"); + assert!( + source + .downcast_ref::() + .is_some() + ); + } + let retained = fixture.journal.load_snapshot(scope()).await.unwrap(); + assert_eq!(retained.head().attempts().len(), 2); + assert_eq!(retained.head().reserved_restore_bytes(), 8192); + assert_eq!( + retained.head().attempts()[0].phase(), + AttemptPhase::Preparing + ); + let attempt = &retained.head().attempts()[0]; + assert!(matches!( + fixture + .journal + .load_movement_action( + scope(), + attempt.spec().id, + MovementAction::Prepare, + attempt.spec().destination_node, + attempt.spec().destination + ) + .await + .unwrap(), + Some(FleetActionAcceptance::Existing { result: None, .. }) + )); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn transport_deadline_before_acceptance_retains_permits_without_fabricating_an_action() { + let fixture = Fixture::new(true, false, false).await; + let driver = fixture.driver(206); + fixture.step(&driver, 0).await; + fixture.stop().await; + fixture + .transport + .block_before_acceptance + .store(true, Ordering::SeqCst); + let report = transport_deadline_at(&driver, &fixture.transport.blocked_before_acceptance).await; + assert_eq!(report.failures.len(), 2); + for failure in &report.failures { + let cellule_runtime::Error::Facility { name, source } = failure.error.as_ref() else { + panic!("timeout source lost"); + }; + assert_eq!(*name, "fleet-controller-deadline"); + assert!(source.is::()); + } + let retained = fixture.journal.load_snapshot(scope()).await.unwrap(); + assert_eq!(retained.head().attempts().len(), 2); + assert_eq!(retained.head().reserved_restore_bytes(), 8192); + for attempt in retained.head().attempts() { + assert_eq!(attempt.phase(), AttemptPhase::Preparing); + assert!( + fixture + .journal + .load_movement_action( + scope(), + attempt.spec().id, + MovementAction::Prepare, + attempt.spec().destination_node, + attempt.spec().destination, + ) + .await + .unwrap() + .is_none() + ); + } + assert_eq!(fixture.transport.dispatched.load(Ordering::SeqCst), 0); + assert_eq!(fixture.transport.released.load(Ordering::SeqCst), 0); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn unavailable_first_endpoint_leaves_deadline_for_healthy_sibling() { + let fixture = Fixture::new(true, false, false).await; + let driver = fixture.driver(206); + fixture.step(&driver, 0).await; + fixture.stop().await; + fixture + .transport + .block_first_acceptance + .store(true, Ordering::SeqCst); + let report = first_acceptance_deadline_at( + &driver, + &fixture.transport.dispatched, + Duration::from_millis(300), + 3, + ) + .await; + assert_eq!(report.failures.len(), 1); + assert_eq!(report.dispatched, 2); + assert_eq!(report.snapshot.head().reserved_restore_bytes(), 8192); + assert_eq!( + report.snapshot.head().attempts()[0].phase(), + AttemptPhase::Preparing + ); + assert_eq!( + report.snapshot.head().attempts()[1].phase(), + AttemptPhase::Reserved + ); + assert_eq!( + report.failures[0].attempt, + report.snapshot.head().attempts()[0].spec().id + ); + assert!(report.blockers.contains(&DrainBlocker::OutcomeUnknown)); + assert_eq!( + report.next_wake_at_ms, + NOW + 100 + FleetProfile::default().reconcile_interval_ms + ); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn regressing_application_clock_fails_before_controller_claim() { + let fixture = Fixture::new(true, false, false).await; + let reads = AtomicUsize::new(0); + let error = fixture + .driver(206) + .reconcile_once( + || Ok(NOW - reads.fetch_add(1, Ordering::SeqCst) as i64), + Instant::now() + Duration::from_secs(3), + ) + .await + .unwrap_err(); + assert!(matches!( + error, + cellule_runtime::Error::Control("fleet controller clock regressed") + )); + let snapshot = fixture.journal.load_snapshot(scope()).await.unwrap(); + assert!(snapshot.head().controller().is_none()); + assert!(snapshot.head().attempts().is_empty()); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn pressure_relief_uses_remaining_shared_permit_without_forgetting_existing_receive() { + let fixture = Fixture::new(false, true, false).await; + let driver = fixture.driver(206); + let initial = fixture.step(&driver, 0).await; + let id = initial.snapshot.head().attempts()[0].spec().id; + let mut snapshot = initial.snapshot; + for event in [AttemptEvent::BeginCancel, AttemptEvent::Cancelled] { + snapshot = fixture + .journal + .compare_exchange( + &snapshot, + 1, + NOW + 50, + &JournalTransition::Attempt { id, event }, + ) + .await + .unwrap(); + } + let progress = snapshot.head().retirement_page(&[id]).unwrap(); + fixture + .journal + .compare_exchange( + &snapshot, + 1, + NOW + 50, + &JournalTransition::Retire { progress }, + ) + .await + .unwrap(); + let next = fixture.step(&driver, 1).await; + assert_eq!(next.dispatched, 1); // finish preparation of the prior receive + assert_eq!(next.allocated, 1); // only the remaining shared slot + assert_eq!(next.snapshot.head().attempts().len(), 2); + assert_eq!(next.snapshot.head().reserved_restore_bytes(), 8192); + let cells = next + .snapshot + .head() + .attempts() + .iter() + .map(|a| a.spec().target.cell_id()) + .collect::>(); + assert_ne!(cells[0], cells[1]); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn unknown_without_acceptance_requires_atomic_absence_cas_before_retry() { + let fixture = Fixture::new(true, false, false).await; + let driver = fixture.driver(206); + let initial = fixture.step(&driver, 0).await; + let id = initial.snapshot.head().attempts()[0].spec().id; + fixture.stop().await; + let mut snapshot = fixture.journal.load_snapshot(scope()).await.unwrap(); + for event in [AttemptEvent::BeginPrepare, AttemptEvent::OutcomeUnknown] { + snapshot = fixture + .journal + .compare_exchange( + &snapshot, + 1, + NOW + 50, + &JournalTransition::Attempt { id, event }, + ) + .await + .unwrap(); + } + let report = fixture.step(&driver, 1).await; + assert_eq!(report.inspected, 1); + assert_eq!(report.dispatched, 2); // absence CAS rearms the first; the other is Planned + assert_eq!(report.retired, 0); + let unknown = report + .snapshot + .head() + .attempts() + .iter() + .find(|attempt| attempt.spec().id == id) + .unwrap(); + assert_eq!(unknown.phase(), AttemptPhase::Reserved); + assert_eq!(unknown.blocker(), None); + assert_eq!(report.snapshot.head().reserved_restore_bytes(), 8192); + assert!( + fixture + .journal + .load_movement_action( + scope(), + id, + MovementAction::Prepare, + unknown.spec().destination_node, + unknown.spec().destination + ) + .await + .unwrap() + .is_some() + ); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn count_balancing_rejects_unknown_live_boot_despite_adapter_completeness() { + let fixture = Fixture::new(true, false, false).await; + let key = EnrollmentSpec { + scope: scope(), + request: Digest::from_bytes([131; 32]), + role: EnrollmentRole::Node { + mode: NodeMode::Active, + }, + source: None, + target: EnrollmentEndpoint { + node: node(3), + session: session(3), + intent_revision: 1, + }, + } + .key() + .unwrap(); + let record = fixture + .journal + .load_enrollment(scope(), key) + .await + .unwrap() + .unwrap(); + fixture + .journal + .publish_enrollment_result( + &record, + EnrollmentEvent::Retired(Digest::from_bytes([230; 32])), + 1, + ) + .await + .unwrap(); + let report = fixture.step(&fixture.driver(206), 0).await; + assert_eq!(fixture.observer.calls.load(Ordering::SeqCst), 1); + assert_eq!(report.allocated, 0); + assert!( + report + .blockers + .contains(&DrainBlocker::IncompleteObservation) + ); + assert!(report.snapshot.head().attempts().is_empty()); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn missing_failed_boot_blocks_count_balancing_without_expiring_its_obligation() { + let fixture = Fixture::new(true, false, false).await; + fixture + .journal + .register_initial_intent(&NodeIntent::initial(scope(), node(4), session(4)).unwrap()) + .await + .unwrap(); + let request = EnrollmentSpec { + scope: scope(), + request: Digest::from_bytes([232; 32]), + role: EnrollmentRole::Node { + mode: NodeMode::Active, + }, + source: None, + target: EnrollmentEndpoint { + node: node(4), + session: session(4), + intent_revision: 1, + }, + }; + let FleetEnrollmentAcceptance::New(record) = fixture + .journal + .accept_enrollment(&request, 0) + .await + .unwrap() + else { + panic!("not a new boot") + }; + fixture + .journal + .publish_enrollment_result( + &record, + EnrollmentEvent::Established(Digest::from_bytes([231; 32])), + 0, + ) + .await + .unwrap(); + let report = fixture.step(&fixture.driver(206), 0).await; + assert_eq!(report.allocated, 0); + assert!( + report + .blockers + .contains(&DrainBlocker::IncompleteObservation) + ); + let retained = fixture + .journal + .load_enrollment(scope(), request.key().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(retained.status(), EnrollmentStatus::Established); + assert_eq!(retained.updated_at_ms(), 0); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn enrollment_during_capture_invalidates_planning_before_allocation() { + let fixture = Fixture::new(true, false, false).await; + *fixture.observer.capture_mutation.lock().unwrap() = Some(fixture.journal.clone()); + let result = fixture + .driver(206) + .reconcile_once(|| Ok(NOW), Instant::now() + Duration::from_secs(3)) + .await; + assert!( + matches!(result, Err(cellule_runtime::Error::FleetOperation(error)) if matches!(*error, OperationError::Conflict)) + ); + let current = fixture.journal.load_snapshot(scope()).await.unwrap(); + assert!(current.head().attempts().is_empty()); + assert_eq!(fixture.transport.dispatched.load(Ordering::SeqCst), 0); + let roster = FleetRoster::collect( + fixture.journal.as_ref(), + ¤t, + Instant::now() + Duration::from_secs(3), + ) + .await + .unwrap(); + let enrolled = roster + .enrollments() + .iter() + .find(|record| record.spec().request == Digest::from_bytes([229; 32])) + .unwrap(); + assert_eq!(enrolled.status(), EnrollmentStatus::Pending); + assert!(enrolled.unresolved()); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn complete_flag_cannot_hide_an_owned_or_transitioning_cell_from_signed_counts() { + let fixture = Fixture::new(true, false, false).await; + fixture + .observer + .omit_last_cell + .store(true, Ordering::SeqCst); + let report = fixture.step(&fixture.driver(206), 0).await; + assert_eq!(report.allocated, 0); + assert!( + report + .blockers + .contains(&DrainBlocker::IncompleteObservation) + ); + assert!(report.snapshot.head().attempts().is_empty()); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn unknown_enrollment_disables_counts_and_keeps_pressure_relief_available() { + for pressured in [false, true] { + let fixture = Fixture::new(true, pressured, false).await; + let request = EnrollmentSpec { + scope: scope(), + request: Digest::from_bytes([235; 32]), + role: EnrollmentRole::Follower { log_epoch: 99 }, + source: Some(EnrollmentEndpoint { + node: node(3), + session: session(3), + intent_revision: 1, + }), + target: EnrollmentEndpoint { + node: node(2), + session: session(2), + intent_revision: 1, + }, + }; + fixture + .journal + .accept_enrollment(&request, 0) + .await + .unwrap(); + let report = fixture.step(&fixture.driver(206), 0).await; + assert_eq!(report.allocated, if pressured { 2 } else { 0 }); + assert!( + report + .blockers + .contains(&DrainBlocker::IncompleteObservation) + ); + let record = fixture + .journal + .load_enrollment(scope(), request.key().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(record.status(), EnrollmentStatus::Pending); + assert_eq!(record.updated_at_ms(), 0); + fixture.journal.close().await.unwrap(); + } +} diff --git a/crates/cellule-host/minion/scenario/adapters.rs b/crates/cellule-host/minion/scenario/adapters.rs new file mode 100644 index 00000000..c5e806b9 --- /dev/null +++ b/crates/cellule-host/minion/scenario/adapters.rs @@ -0,0 +1,145 @@ +use super::*; +use cellule_host::fleet::*; +use cellule_runtime::fleet::operations::*; + +pub(super) struct Cells { + pub records: Arc>, + pub local: usize, + pub root: PathBuf, +} +impl FleetCellProvider for Cells { + fn cell_inputs<'a>( + &'a self, + spec: &'a MoveAttemptSpec, + ) -> FleetAdapterFuture<'a, FleetCellInputs> { + Box::pin(async move { + let record = self + .records + .get(&spec.target.cell_id()) + .ok_or_else(|| invalid("unknown example Cell"))?; + if record.target != spec.target || record.incarnation != spec.incarnation { + return Err(invalid("example Cell binding mismatch")); + } + Ok(FleetCellInputs { + catalog: record.catalog.clone(), + replica: record.replica.clone(), + authority: record.authority.clone(), + destination: self.root.join(format!( + "node-{}-{:?}.sqlite", + self.local, + spec.target.cell_id() + )), + owner: owner(self.local), + }) + }) + } + fn recovery_inputs<'a>( + &'a self, + _: &'a MoveAttemptSpec, + ) -> FleetAdapterFuture<'a, FleetRecoveryInputs> { + Box::pin(async { + Err(invalid( + "this clean-movement scenario has no failed-session recovery proof", + )) + }) + } +} + +/// Fixed three-boot in-process transport. Trusted composition pins identities; +/// production endpoints must provide equivalent authentication independently. +pub(super) struct LocalFleet { + pub nodes: Vec>, + pub journal: Arc, + pub boots: Vec, + pub records: Arc>, + pub capture_sequence: std::sync::atomic::AtomicU64, + pub lose_release_replies: bool, + pub lost_release_replies: std::sync::atomic::AtomicUsize, + pub expired_receiver_cleanups: std::sync::atomic::AtomicUsize, +} +impl LocalFleet { + fn endpoint(&self, physical: NodeId, boot: SessionId) -> JournalResult<&Arc> { + let index = (0..3) + .find(|n| node_id(*n) == physical && session(*n) == boot) + .ok_or_else(|| invalid("unrecognized example boot endpoint"))?; + self.nodes + .get(index) + .ok_or_else(|| invalid("missing example node")) + } +} +impl FleetTransport for LocalFleet { + fn dispatch<'a>( + &'a self, + action: &'a FleetAction, + _: Instant, + ) -> FleetAdapterFuture<'a, Arc> { + Box::pin(async move { + let FleetActionKind::Movement { + action: effect, + attempt, + } = action.kind() + else { + return Err(invalid("non-movement example action")); + }; + let spec = attempt.spec(); + let (node, boot) = if effect.is_source_release() { + (spec.source_node, spec.source) + } else { + (spec.destination_node, spec.destination) + }; + let completion = self + .endpoint(node, boot)? + .apply_fleet_action(action.clone(), clock()?) + .await + .map_err(|error| Box::new(error) as super::JournalError)?; + let now = clock()?; + if *effect == MovementAction::Cancel + && attempt.phase() == AttemptPhase::CleaningReceiver + && attempt + .reservation() + .is_some_and(|r| r.expires_at_ms <= now) + && completion.committed + && matches!(completion.outcome.outcome, FleetOutcome::ReceiverCleaned) + { + self.expired_receiver_cleanups + .fetch_add(1, std::sync::atomic::Ordering::SeqCst); + } + if self.lose_release_replies && effect.is_source_release() { + if !completion.committed + || !matches!(completion.outcome.outcome, FleetOutcome::Released(_)) + { + return Err(std::io::Error::other(format!( + "release failed before injected reply loss: {completion:?}" + )) + .into()); + } + self.lost_release_replies + .fetch_add(1, std::sync::atomic::Ordering::SeqCst); + return Err(invalid("injected loss after committed source release")); + } + Ok(completion) + }) + } + fn inspect<'a>( + &'a self, + request: &'a FleetInspectionRequest, + _: Instant, + ) -> FleetAdapterFuture<'a, Arc> { + Box::pin(async move { + self.endpoint(request.node(), request.session())? + .inspect_fleet_action(request.clone()) + .await + .map_err(|error| Box::new(error) as super::JournalError) + }) + } +} +impl FleetObserver for LocalFleet { + fn observe<'a>( + &'a self, + roster: &'a FleetRoster, + _: i64, + deadline: Instant, + ) -> FleetAdapterFuture<'a, FleetObservation> { + Box::pin(super::observation::observe(self, roster, deadline)) + } +} diff --git a/crates/cellule-host/minion/scenario/application.rs b/crates/cellule-host/minion/scenario/application.rs new file mode 100644 index 00000000..e7b6ae75 --- /dev/null +++ b/crates/cellule-host/minion/scenario/application.rs @@ -0,0 +1,104 @@ +use cellule_app::{ApplicationBuilder, CellType, CompiledApplication}; +use cellule_runtime::{ + cell::catalog::CatalogRole, + identity::{Digest, NamespaceId}, + registry::{ + BuildDescriptor, CellModule, MigrationDescriptor, ModuleDescriptor, NamespaceDescriptor, + OperationDescriptor, Query, QueryContext, RegistryBuilder, + }, +}; +use std::sync::Arc; + +pub(super) const NAMESPACE: NamespaceId = NamespaceId::from_bytes([2; 16]); +struct Module; +impl CellModule for Module { + const NAME: &'static str = "fleet-example"; + fn descriptor(&self) -> &'static ModuleDescriptor { + static DESCRIPTOR: ModuleDescriptor = ModuleDescriptor { + name: "fleet-example", + source_digest: Digest::from_bytes([1; 32]), + retained_codes: &[], + schema_min: 1, + schema_max: 1, + migrations: &[MigrationDescriptor { + version: 1, + sql: "-- host migration v1", + digest: Digest::from_bytes([ + 0xd7, 0x41, 0xcb, 0x18, 0xae, 0xd4, 0x80, 0xb0, 0xe1, 0x55, 0x8e, 0x34, 0x5a, + 0x6b, 0xef, 0xf5, 0xe1, 0x60, 0x80, 0x59, 0x06, 0xba, 0xfe, 0x75, 0xff, 0x9f, + 0xa0, 0x7d, 0x10, 0xe7, 0x77, 0xbf, + ]), + }], + commands: &[], + queries: &[OperationDescriptor { + id: 1, + codec_version: 1, + schema_min: 1, + schema_max: 1, + input_limit: 8, + output_limit: 8, + }], + workflow_definitions: &[], + activity_types: &[], + namespaces: &[NamespaceDescriptor { + id: NAMESPACE, + name: "fleet-example", + role: CatalogRole::Sql, + shards: 1, + effect_targets: &[], + dead_letter: None, + }], + }; + &DESCRIPTOR + } + fn register(self, registry: &mut RegistryBuilder) -> cellule_runtime::Result<()> { + registry.bind_query::() + } +} +pub(super) fn compile() -> cellule_runtime::Result> { + let mut builder = ApplicationBuilder::new( + "fleet-example", + BuildDescriptor { + source_revision: "fleet-example".into(), + cargo_lock_digest: Digest::from_bytes([7; 32]), + }, + )?; + builder.register(Module)?; + builder.cell_type(CellType::new( + "fleet-example", + "fleet-example", + NAMESPACE, + CatalogRole::Sql, + 1, + )?)?; + Ok(Arc::new(builder.finish()?)) +} + +/// Receipt-bound counter read through the same typed reader path as applications. +pub(super) struct ReadValue; +impl Query for ReadValue { + const MODULE: &'static str = "fleet-example"; + const ID: u32 = 1; + const CODEC_VERSION: u32 = 1; + type Input = u64; + type Output = i64; + fn execute(context: &mut QueryContext<'_>, _: u64) -> cellule_runtime::Result { + use cellule_runtime::primitives::sql::{SqlBatch, SqlStatement, SqlValue}; + let sets = context.sql(&SqlBatch { + statements: vec![SqlStatement { + sql: "SELECT value FROM counter".into(), + parameters: vec![], + }], + })?; + match sets + .first() + .and_then(|set| set.rows.first()) + .and_then(|row| row.first()) + { + Some(SqlValue::Integer(value)) => Ok(*value), + _ => Err(cellule_runtime::Error::Command( + "fleet example counter is missing", + )), + } + } +} diff --git a/crates/cellule-host/minion/scenario/balance/mod.rs b/crates/cellule-host/minion/scenario/balance/mod.rs new file mode 100644 index 00000000..0f5842ec --- /dev/null +++ b/crates/cellule-host/minion/scenario/balance/mod.rs @@ -0,0 +1,183 @@ +//! Caller-driven, real-time count convergence; no residence or lease clock shim. + +use super::*; +use cellule_host::fleet::FleetRoster; + +pub(super) async fn run( + root: &tempfile::TempDir, + journal: Arc, + nodes: &mut Vec>, + boots: &mut Vec, + profile: FleetProfile, +) -> JournalResult { + // This scenario spans real residence and several batches. Its newly issued + // command receipts remain resolvable for five minutes; other profiles keep + // their original one-minute receipts and timing bounds. + let (records, acknowledged) = initialize(root, &journal, nodes, boots, 300_000).await?; + let fleet = Arc::new(adapters::LocalFleet { + nodes: nodes.clone(), + boots: boots.clone(), + records: records.clone(), + journal: journal.clone(), + capture_sequence: std::sync::atomic::AtomicU64::new(0), + lose_release_replies: false, + lost_release_replies: std::sync::atomic::AtomicUsize::new(0), + expired_receiver_cleanups: std::sync::atomic::AtomicUsize::new(0), + }); + let driver = FleetReconciler::new( + scope(), + SessionId::from_bytes([206; 16]), + profile, + journal.clone(), + fleet.clone(), + fleet.clone(), + )?; + let mut summary = ScenarioSummary { + released: 0, + activated: 0, + retired: 0, + receipt_checks: 0, + max_inflight: 0, + max_restore_bytes: 0, + joined_nodes: 0, + boot_retirements: 0, + receiver_nodes: 0, + lost_release_replies: 0, + controller_epoch: 0, + expired_receiver_cleanups: 0, + blockers: Vec::new(), + final_counts: [0; 3], + }; + let mut specs = HashMap::new(); + let deadline = Instant::now() + Duration::from_secs(120); + loop { + if Instant::now() >= deadline { + return Err(std::io::Error::other(format!( + "count convergence deadline: summary={summary:?} attempts={:?}", + journal.load_snapshot(scope()).await?.head().attempts() + )) + .into()); + } + let report = driver + .reconcile_once(clock, deadline.min(Instant::now() + Duration::from_secs(5))) + .await?; + if let Some(failure) = report.failures.first() { + return Err(Box::new(Arc::clone(&failure.error)) as JournalError); + } + for attempt in report.snapshot.head().attempts() { + let spec = attempt.spec(); + if let Some(original) = specs.insert(spec.target.cell_id(), spec.clone()) + && original != *spec + { + return Err(invalid( + "count balancing repeated a Cell during its cooldown", + )); + } + } + summary.released += report.released; + summary.activated += report.activated; + summary.retired += report.retired; + summary.max_inflight = summary + .max_inflight + .max(report.snapshot.head().attempts().len()); + summary.max_restore_bytes = summary + .max_restore_bytes + .max(report.snapshot.head().reserved_restore_bytes()); + summary.controller_epoch = report + .snapshot + .head() + .controller() + .ok_or_else(|| invalid("count controller absent"))? + .epoch; + for blocker in report.blockers { + if !summary.blockers.contains(&blocker) { + summary.blockers.push(blocker); + } + } + if summary.retired == 8 && report.snapshot.head().attempts().is_empty() { + // Keep scheduling enabled while checking equilibrium. Stopping + // first would make a no-oscillation assertion vacuous. + let mut complete_samples = 0; + while complete_samples < 2 { + if Instant::now() >= deadline { + return Err(invalid( + "count equilibrium lacked two complete fresh samples", + )); + } + let stable = driver + .reconcile_once(clock, deadline.min(Instant::now() + Duration::from_secs(5))) + .await?; + if stable.allocated != 0 + || !stable.failures.is_empty() + || !stable.snapshot.head().attempts().is_empty() + { + return Err(std::io::Error::other(format!( + "count equilibrium did not remain settled with scheduling enabled: {stable:?}" + )).into()); + } + // Exact authority renewal or an actor transition can invalidate + // an otherwise settled interval. The driver correctly refuses + // count planning for it. Require two *consecutive* complete, + // post-batch samples within the original convergence deadline; + // any new allocation or native failure still fails immediately. + if stable.blockers.iter().any(|blocker| { + matches!( + blocker, + cellule_runtime::fleet::operations::DrainBlocker::IncompleteObservation + | cellule_runtime::fleet::operations::DrainBlocker::StaleObservation + ) + }) { + complete_samples = 0; + tokio::time::sleep(Duration::from_millis(250)).await; + } else { + complete_samples += 1; + } + } + let settled = journal.load_snapshot(scope()).await?; + journal.set_scheduling(settled.registry(), false).await?; + loop { + if Instant::now() >= deadline { + return Err(invalid("final count coverage did not become complete")); + } + let snapshot = journal.load_snapshot(scope()).await?; + let roster = FleetRoster::collect(journal.as_ref(), &snapshot, deadline).await?; + if let Some(counts) = + observation::complete_counts(&fleet, &roster, deadline).await? + { + summary.final_counts = counts; + break; + } + tokio::time::sleep(Duration::from_millis(250)).await; + } + break; + } + // Each pass drives canonical signed boot renewal. Waiting for actual + // residence never suspends the leases for a minute or invents samples. + tokio::time::sleep(Duration::from_millis(250)).await; + } + if summary.final_counts != [4, 4, 4] + || summary.released != 8 + || summary.activated != 8 + || summary.retired != 8 + || specs.len() != 8 + || summary.max_inflight > profile.max_inflight + || summary.max_restore_bytes > profile.max_restore_bytes + { + return Err( + std::io::Error::other(format!("count convergence differs: {summary:?}")).into(), + ); + } + for spec in specs.values() { + verify_movement(&fleet, &records, &acknowledged, spec).await?; + summary.receipt_checks += 1; + } + summary.receiver_nodes = specs + .values() + .map(|spec| spec.destination) + .collect::>() + .len(); + Ok(summary) +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-host/minion/scenario/balance/tests.rs b/crates/cellule-host/minion/scenario/balance/tests.rs new file mode 100644 index 00000000..3421c870 --- /dev/null +++ b/crates/cellule-host/minion/scenario/balance/tests.rs @@ -0,0 +1,19 @@ +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn real_residence_and_post_batch_samples_converge_without_oscillation() { + let started = tokio::time::Instant::now(); + let summary = super::super::count_balance().await.unwrap(); + assert!(started.elapsed() >= std::time::Duration::from_secs(60)); + assert_eq!(summary.final_counts, [4, 4, 4]); + assert_eq!(summary.released, 8); + assert_eq!(summary.activated, 8); + assert_eq!(summary.retired, 8); + assert_eq!(summary.receipt_checks, 8); + assert_eq!(summary.max_inflight, 2); + assert!(summary.max_restore_bytes > 0 && summary.max_restore_bytes <= 8 << 30); + assert_eq!(summary.joined_nodes, 3); + assert_eq!(summary.boot_retirements, 3); + assert_eq!(summary.receiver_nodes, 2); + assert_eq!(summary.controller_epoch, 1); + // Transitional batches can be incomplete. The scenario requires complete + // final counts and two complete, scheduling-enabled equilibrium passes. +} diff --git a/crates/cellule-host/minion/scenario/follower_tests/aggregate.rs b/crates/cellule-host/minion/scenario/follower_tests/aggregate.rs new file mode 100644 index 00000000..28706ac0 --- /dev/null +++ b/crates/cellule-host/minion/scenario/follower_tests/aggregate.rs @@ -0,0 +1,436 @@ +//! Full native capture with managed follower producers and persisted tails. +use super::*; +use cellule_host::fleet::{ + FleetFollowerReferences, FleetNodeInventory, FleetNodeInventoryScan, FleetRoster, + FleetSnapshotRequest, FleetSnapshotSubject, +}; +use cellule_runtime::follower::FollowerLaneState; + +pub(super) async fn roster(fixture: &ManagedFixture) -> FleetRoster { + let snapshot = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + FleetRoster::collect( + fixture.native.journal.as_ref(), + &snapshot, + Instant::now() + Duration::from_secs(5), + ) + .await + .unwrap() +} + +pub(super) async fn page( + fixture: &ManagedFixture, + roster: &FleetRoster, + index: usize, + subject: FleetSnapshotSubject, + sequence: &mut u64, +) -> Arc { + *sequence += 1; + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.managed-follower-inventory-test.v1\0"); + hash.update(&sequence.to_be_bytes()); + hash.update(session(index).as_bytes()); + let now = clock().unwrap(); + let request = FleetSnapshotRequest::new( + roster.snapshot().clone(), + Digest::from_bytes(*hash.finalize().as_bytes()), + node_id(index), + session(index), + subject, + 1, + now, + (now + 5_000).min(roster.snapshot().head().controller().unwrap().expires_at_ms), + ) + .unwrap(); + fixture.nodes[index].fleet_snapshot(request).await.unwrap() +} + +pub(super) async fn collect( + fixture: &ManagedFixture, + roster: &FleetRoster, + index: usize, + sequence: &mut u64, +) -> FleetNodeInventory { + let mut scan = FleetNodeInventoryScan::new(roster, node_id(index), session(index)).unwrap(); + while let Some(subject) = scan.next_subject().unwrap() { + let response = page(fixture, roster, index, subject, sequence).await; + scan.accept(response.request(), &response, clock().unwrap()) + .unwrap(); + } + scan.finish().unwrap() +} + +pub(super) async fn references( + fixture: &ManagedFixture, + roster: &FleetRoster, + index: usize, +) -> FleetFollowerReferences { + FleetFollowerReferences::collect( + &fixture.native.directory, + roster, + node_id(index), + 1, + Instant::now() + Duration::from_secs(5), + clock, + ) + .await + .unwrap() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn managed_aggregate_matches_original_producer_and_foreign_persisted_tails() { + let fixture = ManagedFixture::new().await; + let roster = roster(&fixture).await; + let mut sequence = 0; + let mut inventories = Vec::new(); + let mut authorities = Vec::new(); + for index in 0..3 { + let inventory = collect(&fixture, &roster, index, &mut sequence).await; + inventory.validate_enrollments(&roster).unwrap(); + let logs = references(&fixture, &roster, index).await; + logs.validate_enrollments(&roster).unwrap(); + if index == 0 { + assert_eq!(inventory.cells().len(), 1); + assert!(inventory.bindings().follower_producer); + assert!(inventory.bindings().durability_supervisor); + assert_eq!(inventory.node_log(), Some((session(0), node_id(0), 1))); + assert_eq!(inventory.follower_enrollments().len(), 1); + let original = fixture + .native + .node + .follower_enrollment_completion(1) + .unwrap() + .unwrap(); + let observed = &inventory.follower_enrollments()[0]; + assert_eq!( + observed.attempt, + original.attempt.evidence_digest().unwrap() + ); + assert!(observed.delivered && observed.native_started); + for (member, original) in observed.members.iter().zip(&original.members) { + assert_eq!(member.spec, original.spec); + assert_eq!(member.accepted, original.accepted); + assert_eq!(member.event, original.event); + assert!(member.published); + } + assert!(logs.entries().is_empty()); + } else { + // Zero writers does not remove the other owner's durable obligation. + assert!(inventory.cells().is_empty()); + assert_eq!(inventory.follower_store_state(), Some((1, 0))); + assert_eq!(inventory.follower_lanes().len(), 1); + let lane = &inventory.follower_lanes()[0]; + assert_eq!( + (lane.leader, lane.epoch, lane.state), + (session(0), 1, FollowerLaneState::Open) + ); + assert_eq!(logs.entries().len(), 1); + let authority = &logs.entries()[0]; + assert_eq!(authority.leader, lane.leader); + assert_eq!(authority.log.epoch(), lane.epoch); + assert!(authority.log.active()); + assert!(authority.log.members().contains(&node_id(index))); + } + inventories.push(inventory); + authorities.push(logs); + } + for authority in &mut authorities { + authority + .recheck( + &fixture.native.directory, + &roster, + 1, + Instant::now() + Duration::from_secs(5), + clock, + ) + .await + .unwrap(); + } + for (index, inventory) in inventories.iter_mut().enumerate() { + let mut recheck = inventory.recheck(); + while let Some(subject) = recheck.next_subject().unwrap() { + let response = page(&fixture, &roster, index, subject, &mut sequence).await; + recheck + .accept(response.request(), &response, clock().unwrap()) + .unwrap(); + } + recheck.finish().unwrap(); + } + roster + .confirm( + fixture.native.journal.as_ref(), + Instant::now() + Duration::from_secs(5), + ) + .await + .unwrap(); + let value = fixture + .handle + .query(64, 64, |connection| { + let value: i64 = + connection.query_row("SELECT value FROM counter", [], |row| row.get(0))?; + Ok(value.to_be_bytes().to_vec()) + }) + .await + .unwrap(); + assert_eq!(value, 29i64.to_be_bytes()); + drop(inventories); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn managed_rotation_failure_preserves_original_error_and_foreign_obligation() { + let fixture = ManagedFixture::new().await; + fixture.handle.drain().await.unwrap(); + fixture + .native + .transport + .lose_retire + .store(true, Ordering::Release); + let (entered, resume) = fixture.native.transport.pause_retirement_retry(); + let request = fixture.native.node.request_node_log_rotation(1).unwrap(); + captured(entered).await; + let roster = roster(&fixture).await; + let mut sequence = 0; + let source = collect(&fixture, &roster, 0, &mut sequence).await; + source.validate_enrollments(&roster).unwrap(); + let original = fixture + .native + .node + .follower_enrollment_completion(1) + .unwrap() + .unwrap(); + let observed = &source.follower_enrollments()[0]; + assert!(Arc::ptr_eq( + observed.execution_error.as_ref().unwrap(), + original.execution_error.as_ref().unwrap() + )); + assert!(Arc::ptr_eq( + observed.retirement.as_ref().unwrap(), + original.retirement.as_ref().unwrap() + )); + assert!(!observed.native_closed); + assert!(observed.retirement.as_ref().unwrap().confirmed().is_err()); + let receiver = collect(&fixture, &roster, 1, &mut sequence).await; + receiver.validate_enrollments(&roster).unwrap(); + assert_eq!( + receiver.follower_lanes()[0].state, + FollowerLaneState::Retired + ); + let mut logs = references(&fixture, &roster, 1).await; + logs.validate_enrollments(&roster).unwrap(); + assert_eq!(logs.entries().len(), 1); + assert_eq!(fixture.native.authority.attempts.load(Ordering::Acquire), 0); + let interval = logs.interval(); + fixture + .native + .transport + .lose_retire + .store(false, Ordering::Release); + resume.send(()).unwrap(); + until(|| request.observe().unwrap().phase() == cellule_host::NodeLogRotationPhase::Recruiting) + .await; + assert!(request.observe().unwrap().retirement().is_some()); + // This fixed fixture has no spare ensemble. Retirement does not invent a + // replacement or erase the retained canonical binding while recruitment waits. + assert_eq!( + fixture + .native + .node + .runtime() + .node_durability() + .unwrap() + .1 + .identity() + .unwrap(), + (session(0), node_id(0), 1) + ); + assert!( + logs.recheck( + &fixture.native.directory, + &roster, + 1, + Instant::now() + Duration::from_secs(5), + clock + ) + .await + .is_err() + ); + assert_eq!(logs.interval(), interval); + assert_eq!(logs.entries().len(), 1); + let after_roster = self::roster(&fixture).await; + let after = references(&fixture, &after_roster, 1).await; + assert!(after.entries().is_empty()); + let inventory = collect(&fixture, &after_roster, 1, &mut sequence).await; + inventory.validate_enrollments(&after_roster).unwrap(); + assert_eq!(inventory.follower_store_state(), Some((0, 0))); + assert_eq!( + inventory.follower_lanes()[0].state, + FollowerLaneState::Retired + ); + drop(request); + drop(source); + drop(receiver); + drop(inventory); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn authority_recheck_rejects_real_object_coverage_even_with_unchanged_topology() { + let fixture = ManagedFixture::new().await; + let roster = roster(&fixture).await; + let mut logs = references(&fixture, &roster, 1).await; + let before = fixture + .native + .directory + .follower_logs_page(node_id(1), None, 1, clock().unwrap()) + .await + .unwrap(); + let original = logs.entries()[0].log.tiered_through(); + let now = clock().unwrap(); + fixture + .handle + .execute( + MutationIdentity { + request_id: RequestId::from_bytes([2; 16]), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }, + Digest::from_bytes([2; 32]), + now, + 64, + 64, + |tx| { + tx.execute_batch("UPDATE counter SET value = 41")?; + Ok(HandlerOutcome::Success(41i64.to_be_bytes().to_vec())) + }, + ) + .await + .unwrap(); + // Canonical drain publishes the accepted mutation and closes its SQLite + // owner. No invented coverage watermark is written by this test. + fixture.handle.drain().await.unwrap(); + let after = fixture + .native + .directory + .follower_logs_page(node_id(1), None, 1, clock().unwrap()) + .await + .unwrap(); + assert_eq!(before.topology(), after.topology()); + assert!(after.entries()[0].log.tiered_through() > original); + let interval = logs.interval(); + assert!(matches!( + logs.recheck( + &fixture.native.directory, + &roster, + 1, + Instant::now() + Duration::from_secs(5), + clock + ) + .await, + Err(Error::Node("authoritative follower inventory changed")) + )); + assert_eq!(logs.interval(), interval); + assert_eq!(logs.entries()[0].log.tiered_through(), original); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn authority_scan_rejects_deadlines_regressed_clocks_and_changed_liveness() { + let fixture = ManagedFixture::new().await; + let roster = roster(&fixture).await; + for limit in [0, 129] { + assert!( + FleetFollowerReferences::collect( + &fixture.native.directory, + &roster, + node_id(1), + limit, + Instant::now() + Duration::from_secs(3), + clock + ) + .await + .is_err() + ); + } + assert!(matches!( + FleetFollowerReferences::collect( + &fixture.native.directory, + &roster, + node_id(1), + 1, + Instant::now(), + || panic!("expired scan called the clock") + ) + .await, + Err(Error::Deadline) + )); + let now = clock().unwrap(); + let mut calls = 0; + assert!( + FleetFollowerReferences::collect( + &fixture.native.directory, + &roster, + node_id(1), + 1, + Instant::now() + Duration::from_secs(3), + || { + calls += 1; + Ok(now - i64::from(calls > 1)) + } + ) + .await + .is_err() + ); + let owner = fixture + .native + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .unwrap(); + let expiry = owner.advertisement().expires_at_ms(); + // Controlled directory time exercises expiry without fencing or restamping + // the native processes, which are still joined at their actual clock. + let mut logs = FleetFollowerReferences::collect( + &fixture.native.directory, + &roster, + node_id(1), + 1, + Instant::now() + Duration::from_secs(3), + || Ok(expiry - 1), + ) + .await + .unwrap(); + assert_eq!( + logs.entries()[0].leader_state, + cellule_runtime::node::LogLeaderState::Live + ); + let before = fixture + .native + .directory + .follower_logs_page(node_id(1), None, 1, expiry - 1) + .await + .unwrap(); + let after = fixture + .native + .directory + .follower_logs_page(node_id(1), None, 1, expiry + 1) + .await + .unwrap(); + assert_eq!(before.topology(), after.topology()); + assert_eq!( + after.entries()[0].leader_state, + cellule_runtime::node::LogLeaderState::Expired + ); + assert!(matches!( + logs.recheck( + &fixture.native.directory, + &roster, + 1, + Instant::now() + Duration::from_secs(3), + || Ok(expiry + 1) + ) + .await, + Err(Error::Node("authoritative follower inventory changed")) + )); + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/follower_tests/coverage.rs b/crates/cellule-host/minion/scenario/follower_tests/coverage.rs new file mode 100644 index 00000000..9940486e --- /dev/null +++ b/crates/cellule-host/minion/scenario/follower_tests/coverage.rs @@ -0,0 +1,276 @@ +//! Complete cross-node graph, including enrolled lanes before their first append. +use super::*; +use cellule_host::fleet::{ + FleetFollowerReferences, FleetNodeInventory, FleetRoleCoverage, FleetRoster, +}; + +pub(super) async fn captures( + fixture: &ManagedFixture, + roster: &FleetRoster, +) -> (Vec, Vec, u64) { + let mut sequence = 0; + let mut native = Vec::new(); + let mut foreign = Vec::new(); + for index in 0..fixture.nodes.len() { + native.push(aggregate::collect(fixture, roster, index, &mut sequence).await); + foreign.push(aggregate::references(fixture, roster, index).await); + } + (native, foreign, sequence) +} + +pub(super) async fn native_rechecks( + fixture: &ManagedFixture, + roster: &FleetRoster, + native: &mut [FleetNodeInventory], + sequence: &mut u64, +) { + for (index, inventory) in native.iter_mut().enumerate() { + let mut recheck = inventory.recheck(); + while let Some(subject) = recheck.next_subject().unwrap() { + let page = aggregate::page(fixture, roster, index, subject, sequence).await; + recheck + .accept(page.request(), &page, clock().unwrap()) + .unwrap(); + } + recheck.finish().unwrap(); + } +} + +pub(super) async fn foreign_rechecks( + fixture: &ManagedFixture, + roster: &FleetRoster, + foreign: &mut [FleetFollowerReferences], +) { + for references in foreign { + references + .recheck( + &fixture.native.directory, + roster, + 1, + Instant::now() + Duration::from_secs(5), + clock, + ) + .await + .unwrap(); + } +} + +pub(super) fn check( + roster: &FleetRoster, + native: &[FleetNodeInventory], + foreign: &[FleetFollowerReferences], +) -> cellule_runtime::Result { + FleetRoleCoverage::check( + roster, + &native.iter().collect::>(), + &foreign.iter().collect::>(), + clock().unwrap(), + ) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn full_role_coverage_requires_global_rechecks_and_is_order_independent() { + let fixture = ManagedFixture::new().await; + let roster = aggregate::roster(&fixture).await; + let (mut native, mut foreign, mut sequence) = captures(&fixture, &roster).await; + assert!(check(&roster, &native, &foreign).is_err()); + tokio::time::sleep(Duration::from_millis(2)).await; + native_rechecks(&fixture, &roster, &mut native, &mut sequence).await; + assert!(check(&roster, &native, &foreign).is_err()); + foreign_rechecks(&fixture, &roster, &mut foreign).await; + let coverage = check(&roster, &native, &foreign).unwrap(); + roster + .confirm( + fixture.native.journal.as_ref(), + Instant::now() + Duration::from_secs(5), + ) + .await + .unwrap(); + assert_eq!(coverage.native_boots(), 3); + assert_eq!(coverage.physical_nodes(), 3); + assert_eq!(coverage.pending_enrollments(), 0); + assert_eq!(coverage.snapshot(), roster.snapshot()); + assert_eq!(coverage.roster_digest(), roster.digest().unwrap()); + native.reverse(); + foreign.reverse(); + assert_eq!( + check(&roster, &native, &foreign).unwrap().digest(), + coverage.digest() + ); + assert!(check(&roster, &native[..2], &foreign).is_err()); + assert!(check(&roster, &native, &foreign[..2]).is_err()); + assert!( + FleetRoleCoverage::check( + &roster, + &[&native[0], &native[0], &native[2]], + &foreign.iter().collect::>(), + clock().unwrap() + ) + .is_err() + ); + assert!( + FleetRoleCoverage::check( + &roster, + &native.iter().collect::>(), + &[&foreign[0], &foreign[0], &foreign[2]], + clock().unwrap() + ) + .is_err() + ); + drop((native, foreign, coverage)); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn dropped_global_native_recheck_cannot_reuse_an_earlier_confirmation() { + let fixture = ManagedFixture::new().await; + let roster = aggregate::roster(&fixture).await; + let (mut native, mut foreign, mut sequence) = captures(&fixture, &roster).await; + native_rechecks(&fixture, &roster, &mut native, &mut sequence).await; + foreign_rechecks(&fixture, &roster, &mut foreign).await; + assert!(check(&roster, &native, &foreign).is_ok()); + let original = native[1].interval(); + drop(native[1].recheck()); + assert_eq!(native[1].interval(), original); + assert!(check(&roster, &native, &foreign).is_err()); + tokio::time::sleep(Duration::from_millis(2)).await; + native_rechecks(&fixture, &roster, &mut native, &mut sequence).await; + // Foreign checks must follow this latest complete native round. + assert!(check(&roster, &native, &foreign).is_err()); + foreign_rechecks(&fixture, &roster, &mut foreign).await; + assert!(check(&roster, &native, &foreign).is_ok()); + drop((native, foreign)); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn failed_foreign_recheck_preserves_rows_and_invalidates_coverage() { + let fixture = ManagedFixture::new().await; + let roster = aggregate::roster(&fixture).await; + let (mut native, mut foreign, mut sequence) = captures(&fixture, &roster).await; + native_rechecks(&fixture, &roster, &mut native, &mut sequence).await; + foreign_rechecks(&fixture, &roster, &mut foreign).await; + assert!(check(&roster, &native, &foreign).is_ok()); + let interval = foreign[1].interval(); + let rows = foreign[1].entries().to_vec(); + assert!( + foreign[1] + .recheck(&fixture.native.directory, &roster, 1, Instant::now(), clock) + .await + .is_err() + ); + assert_eq!(foreign[1].interval(), interval); + assert_eq!(foreign[1].entries(), rows); + assert!(check(&roster, &native, &foreign).is_err()); + drop((native, foreign)); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn full_role_coverage_matches_enrolled_empty_lanes_to_original_cross_node_producer() { + let fixture = ManagedFixture::with_members(4, 2).await; + evacuation::maintenance(&fixture).await; + evacuation::spare(&fixture).await; + evacuation::rotated(&fixture).await; + let roster = aggregate::roster(&fixture).await; + let (mut native, mut foreign, mut sequence) = captures(&fixture, &roster).await; + for index in [2, 3] { + assert!( + !native[index] + .follower_lanes() + .iter() + .any(|lane| lane.epoch == 2) + ); + assert!(matches!( + native[index].validate_enrollments(&roster), + Err(Error::Control( + "Established follower has no persisted native lane" + )) + )); + assert_eq!(foreign[index].entries().len(), 1); + assert_eq!(foreign[index].entries()[0].log.epoch(), 2); + } + native_rechecks(&fixture, &roster, &mut native, &mut sequence).await; + foreign_rechecks(&fixture, &roster, &mut foreign).await; + let coverage = check(&roster, &native, &foreign).unwrap(); + roster + .confirm( + fixture.native.journal.as_ref(), + Instant::now() + Duration::from_secs(5), + ) + .await + .unwrap(); + assert_eq!(coverage.native_boots(), 4); + assert_eq!(coverage.pending_enrollments(), 0); + assert!( + native[1] + .follower_lanes() + .iter() + .all(|lane| lane.state == cellule_runtime::follower::FollowerLaneState::Retired) + ); + // No missing local lane became absence: both actual Established epoch-two + // responsibilities and the configured source binding remain visible. + assert_eq!(native[0].node_log(), Some((session(0), node_id(0), 2))); + assert_eq!(native[0].follower_enrollments()[0].members.len(), 2); + assert!( + FleetRoleCoverage::check( + &roster, + &[&native[1], &native[2], &native[3]], + &foreign.iter().collect::>(), + clock().unwrap() + ) + .is_err() + ); + drop((native, foreign, coverage)); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn earlier_foreign_checks_cannot_cover_a_later_global_native_round() { + let fixture = ManagedFixture::new().await; + let roster = aggregate::roster(&fixture).await; + let (mut native, mut foreign, mut sequence) = captures(&fixture, &roster).await; + foreign_rechecks(&fixture, &roster, &mut foreign).await; + tokio::time::sleep(Duration::from_millis(2)).await; + native_rechecks(&fixture, &roster, &mut native, &mut sequence).await; + assert!(check(&roster, &native, &foreign).is_err()); + foreign_rechecks(&fixture, &roster, &mut foreign).await; + assert!(check(&roster, &native, &foreign).is_ok()); + drop((native, foreign)); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn changed_full_roster_cannot_reuse_original_role_captures_or_confirmation() { + let fixture = ManagedFixture::new().await; + let original = aggregate::roster(&fixture).await; + let (mut native, mut foreign, mut sequence) = captures(&fixture, &original).await; + native_rechecks(&fixture, &original, &mut native, &mut sequence).await; + foreign_rechecks(&fixture, &original, &mut foreign).await; + let coverage = check(&original, &native, &foreign).unwrap(); + evacuation::maintenance(&fixture).await; + let changed = aggregate::roster(&fixture).await; + assert_ne!(changed.digest().unwrap(), original.digest().unwrap()); + assert!(check(&changed, &native, &foreign).is_err()); + assert!( + original + .confirm( + fixture.native.journal.as_ref(), + Instant::now() + Duration::from_secs(5) + ) + .await + .is_err() + ); + assert_eq!(coverage.snapshot(), original.snapshot()); + assert!( + FleetRoleCoverage::check( + &original, + &native.iter().collect::>(), + &foreign.iter().collect::>(), + clock().unwrap() + 30_001 + ) + .is_err() + ); + drop((native, foreign, coverage)); + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/follower_tests/evacuation.rs b/crates/cellule-host/minion/scenario/follower_tests/evacuation.rs new file mode 100644 index 00000000..0c6bb65c --- /dev/null +++ b/crates/cellule-host/minion/scenario/follower_tests/evacuation.rs @@ -0,0 +1,373 @@ +//! Live-owner replacement through four managed boots and the original supervisor. +use super::*; +use cellule_host::{FollowerEvacuation, NodeLogRotationPhase}; +use cellule_runtime::fleet::operations::{ + JournalTransition, MaintenanceEvent, MaintenanceOperation, OperationId, +}; + +pub(super) async fn maintenance( + fixture: &ManagedFixture, +) -> (EnrollmentRecord, MaintenanceOperation) { + let original = fixture + .native + .rows() + .await + .into_iter() + .find(|row| { + row.spec().target.node == node_id(1) + && matches!( + row.spec().role, + cellule_runtime::fleet::operations::EnrollmentRole::Follower { log_epoch: 1 } + ) + }) + .unwrap(); + assert_eq!(original.status(), EnrollmentStatus::Established); + let now = clock().unwrap(); + let mut snapshot = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + for transition in [ + JournalTransition::BeginMaintenance( + MaintenanceOperation::new( + OperationId::from_bytes([91; 16]).unwrap(), + Digest::from_bytes([92; 32]), + node_id(1), + session(1), + 2, + now, + now + 60_000, + ) + .unwrap(), + ), + JournalTransition::Maintenance(MaintenanceEvent::Cordoned), + JournalTransition::Maintenance(MaintenanceEvent::BeginEvacuation), + ] { + snapshot = fixture + .native + .journal + .compare_exchange( + &snapshot, + snapshot.head().controller().unwrap().epoch, + clock().unwrap(), + &transition, + ) + .await + .unwrap(); + } + fixture.boots[1] + .refresh_capacity( + 1, + fixture.native.journal.as_ref(), + Instant::now() + Duration::from_secs(3), + ) + .await + .unwrap(); + let donor = fixture + .native + .directory + .load(session(1), clock().unwrap()) + .await + .unwrap() + .unwrap(); + let store = fixture.nodes[1] + .try_owned_component::(cellule_host::FOLLOWER_STORE_COMPONENT) + .unwrap() + .unwrap(); + assert!(store.retained_bytes() > 0); + assert_eq!( + donor.advertisement().capacity().follower_retained_bytes, + store.retained_bytes() + ); + (original, snapshot.head().maintenance().unwrap().clone()) +} + +pub(super) async fn spare(fixture: &ManagedFixture) { + fixture.boots[3] + .refresh_capacity( + 3, + fixture.native.journal.as_ref(), + Instant::now() + Duration::from_secs(3), + ) + .await + .unwrap(); + assert_eq!( + fixture + .native + .directory + .select_log_members(session(0), 1, clock().unwrap(), 4) + .await + .unwrap(), + [node_id(2), node_id(3)] + ); +} + +pub(super) async fn rotated(fixture: &ManagedFixture) -> cellule_host::NodeLogRotationRequest { + // The actual actor publication barrier covers the acknowledged old tail. + fixture.handle.drain().await.unwrap(); + let request = fixture.native.node.request_node_log_rotation(1).unwrap(); + until(|| request.observe().unwrap().phase() == NodeLogRotationPhase::Completed).await; + request +} + +async fn evidence( + fixture: &ManagedFixture, + original: &EnrollmentRecord, + operation: &MaintenanceOperation, +) -> cellule_runtime::Result { + fixture + .native + .node + .follower_evacuation( + original, + operation, + 2, + Instant::now() + Duration::from_secs(3), + ) + .await +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn managed_follower_evacuation_proves_full_replacement_and_exact_replay() { + let fixture = ManagedFixture::with_members(4, 2).await; + let (original, operation) = maintenance(&fixture).await; + spare(&fixture).await; + assert_eq!( + fixture + .handle + .query(64, 64, |tx| { + let value: i64 = tx.query_row("SELECT value FROM counter", [], |row| row.get(0))?; + Ok(value.to_be_bytes().to_vec()) + }) + .await + .unwrap(), + 29i64.to_be_bytes() + ); + let request = rotated(&fixture).await; + let completion = request.observe().unwrap().completion().unwrap().clone(); + assert!(completion.retirement().barrier().covered_through() > 0); + let first = evidence(&fixture, &original, &operation).await.unwrap(); + assert_eq!(first.replacement().prepared().followers().len(), 2); + assert!(Arc::ptr_eq(first.rotation(), &completion)); + assert_eq!(first.original(), &original); + assert_eq!(first.retired().status(), EnrollmentStatus::Retired); + assert_eq!(first.minimum_members(), 2); + assert_eq!(first.authority().log().unwrap().epoch(), 2); + assert_eq!( + first.authority().log().unwrap().members(), + [node_id(2), node_id(3)] + ); + assert_eq!( + first + .replacements() + .iter() + .map(|row| row.spec().target.node) + .collect::>(), + [node_id(2), node_id(3)] + ); + let second = evidence(&fixture, &original, &operation).await.unwrap(); + assert_eq!(first.retired(), second.retired()); + assert_eq!(first.replacements(), second.replacements()); + assert_eq!(first.snapshot(), second.snapshot()); + assert!(second.interval().0 >= first.interval().1); + assert!(Arc::ptr_eq(first.rotation(), second.rotation())); + let references = fixture + .native + .directory + .follower_logs_page(node_id(1), None, 1, clock().unwrap()) + .await + .unwrap(); + assert!(references.entries().is_empty()); + // Retired lane fences remain present; role evacuation never deletes files. + let store = fixture.nodes[1] + .try_owned_component::(cellule_host::FOLLOWER_STORE_COMPONENT) + .unwrap() + .unwrap(); + let lane = store + .fleet_lanes_page(None, 1, clock().unwrap()) + .await + .unwrap(); + assert_eq!( + lane.entries()[0].state, + cellule_runtime::follower::FollowerLaneState::Retired + ); + drop((first, second, lane, store)); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn managed_follower_evacuation_cannot_settle_without_spare_or_completion() { + let fixture = ManagedFixture::with_members(3, 2).await; + let (original, operation) = maintenance(&fixture).await; + assert!(evidence(&fixture, &original, &operation).await.is_err()); + assert!( + fixture + .native + .directory + .select_log_members(session(0), 1, clock().unwrap(), 3) + .await + .unwrap() + .is_empty() + ); + fixture.handle.drain().await.unwrap(); + let request = fixture.native.node.request_node_log_rotation(1).unwrap(); + until(|| request.observe().unwrap().phase() == NodeLogRotationPhase::Recruiting).await; + assert!(request.observe().unwrap().retirement().is_some()); + assert!(request.observe().unwrap().completion().is_none()); + assert!(evidence(&fixture, &original, &operation).await.is_err()); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn managed_follower_evacuation_preserves_lost_member_reply_across_retry() { + let fixture = ManagedFixture::with_members(4, 2).await; + let (original, operation) = maintenance(&fixture).await; + spare(&fixture).await; + fixture.handle.drain().await.unwrap(); + fixture + .native + .transport + .lose_retire + .store(true, Ordering::Release); + let (entered, resume) = fixture.native.transport.pause_retirement_retry(); + let request = fixture.native.node.request_node_log_rotation(1).unwrap(); + captured(entered).await; + let first = request.observe().unwrap().first_failure().unwrap().clone(); + assert!(evidence(&fixture, &original, &operation).await.is_err()); + drop(request); + fixture + .native + .transport + .lose_retire + .store(false, Ordering::Release); + resume.send(()).unwrap(); + let retained = fixture + .native + .node + .node_log_rotation_request(1) + .unwrap() + .unwrap(); + until(|| retained.observe().unwrap().phase() == NodeLogRotationPhase::Completed).await; + assert!(Arc::ptr_eq( + retained.observe().unwrap().first_failure().unwrap(), + &first + )); + let settled = evidence(&fixture, &original, &operation).await.unwrap(); + assert_eq!( + settled.retired().accepted_at_ms(), + original.accepted_at_ms() + ); + assert_eq!( + settled.retired().established_evidence(), + original.established_evidence() + ); + drop(settled); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn managed_follower_evacuation_rejects_expired_waiter_and_invalid_original_or_policy() { + let fixture = ManagedFixture::with_members(4, 2).await; + let (original, operation) = maintenance(&fixture).await; + spare(&fixture).await; + rotated(&fixture).await; + for minimum in [0, 3] { + assert!( + fixture + .native + .node + .follower_evacuation( + &original, + &operation, + minimum, + Instant::now() + Duration::from_secs(3) + ) + .await + .is_err() + ); + } + assert!(matches!( + fixture + .native + .node + .follower_evacuation(&original, &operation, 2, Instant::now()) + .await, + Err(Error::Deadline) + )); + let foreign = fixture + .native + .rows() + .await + .into_iter() + .find(|row| { + row.spec().target.node == node_id(2) + && matches!( + row.spec().role, + cellule_runtime::fleet::operations::EnrollmentRole::Follower { log_epoch: 2 } + ) + }) + .unwrap(); + assert!(evidence(&fixture, &foreign, &operation).await.is_err()); + let settled = evidence(&fixture, &original, &operation).await.unwrap(); + drop(settled); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn managed_follower_evacuation_rejects_withdrawn_replacement_after_local_completion() { + let fixture = ManagedFixture::with_members(4, 2).await; + let (original, operation) = maintenance(&fixture).await; + spare(&fixture).await; + rotated(&fixture).await; + let settled = evidence(&fixture, &original, &operation).await.unwrap(); + fixture.nodes[3].shutdown().await.unwrap(); + assert!(evidence(&fixture, &original, &operation).await.is_err()); + assert_eq!(settled.replacements().len(), 2); + drop(settled); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn managed_follower_evacuation_rejects_fenced_live_owner_after_local_completion() { + let fixture = ManagedFixture::with_members(4, 2).await; + let (original, operation) = maintenance(&fixture).await; + spare(&fixture).await; + rotated(&fixture).await; + let settled = evidence(&fixture, &original, &operation).await.unwrap(); + fixture.boots[0].guard.as_ref().unwrap().fence(); + assert!(evidence(&fixture, &original, &operation).await.is_err()); + assert_eq!(settled.replacements().len(), 2); + drop(settled); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn managed_follower_evacuation_requires_current_operation_after_deadline_extension() { + let fixture = ManagedFixture::with_members(4, 2).await; + let (original, operation) = maintenance(&fixture).await; + spare(&fixture).await; + rotated(&fixture).await; + let settled = evidence(&fixture, &original, &operation).await.unwrap(); + let snapshot = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + let after = fixture + .native + .journal + .compare_exchange( + &snapshot, + snapshot.head().controller().unwrap().epoch, + clock().unwrap(), + &JournalTransition::Maintenance(MaintenanceEvent::ExtendDeadline( + operation.deadline_ms() + 10_000, + )), + ) + .await + .unwrap(); + assert!(evidence(&fixture, &original, &operation).await.is_err()); + let updated = evidence(&fixture, &original, after.head().maintenance().unwrap()) + .await + .unwrap(); + assert_eq!(updated.original(), settled.original()); + assert_eq!(updated.retired(), settled.retired()); + assert_eq!(updated.replacements(), settled.replacements()); + assert!(Arc::ptr_eq(updated.rotation(), settled.rotation())); + assert_ne!(updated.snapshot(), settled.snapshot()); + drop((updated, settled)); + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/follower_tests/fixture/managed.rs b/crates/cellule-host/minion/scenario/follower_tests/fixture/managed.rs new file mode 100644 index 00000000..fef0dd74 --- /dev/null +++ b/crates/cellule-host/minion/scenario/follower_tests/fixture/managed.rs @@ -0,0 +1,360 @@ +//! Actual managed boots, node-owned follower stores and an acknowledged Cell. +use super::super::super::{adapters, startup}; +use super::*; +use cellule_runtime::fleet::operations::EnrollmentRole; + +pub(crate) struct ManagedFixture { + pub native: Fixture, + pub nodes: Vec>, + pub boots: Vec, + pub handle: CellHandle, +} + +impl ManagedFixture { + pub async fn new() -> Self { + Self::with_members(3, 1).await + } + + pub async fn with_members(count: usize, max_epochs: usize) -> Self { + assert!((3..=4).contains(&count)); + assert!((1..=3).contains(&max_epochs)); + let root = tempfile::tempdir().unwrap(); + let app = application::compile().unwrap(); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + ObjectPath::from("managed-followers"), + [3; 16], + ); + let directory = NodeDirectory::new( + layout.clone(), + scope().fleet, + Digest::from_bytes([31; 32]), + app.registry().release_digest(), + ); + let journal = Arc::new( + SqliteJournal::open( + root.path().join("journal.sqlite"), + scope(), + FleetProfile::default(), + clock().unwrap(), + ) + .await + .unwrap(), + ); + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + scope().application, + application::NAMESPACE, + &[1], + ) + .unwrap(); + let proof = CellCatalog::new(layout.clone(), target.tenant()) + .provision( + CatalogEntry::new( + &target, + CatalogRole::Sql, + app.registry().module_digests()[0], + 1, + ) + .unwrap(), + ) + .await + .unwrap(); + let incarnation = IncarnationId::from_bytes([1; 16]); + let cell_authority = CellAuthority::new(layout.clone()); + let initial = cell_authority + .create_initial(&proof, incarnation, owner(0)) + .await + .unwrap(); + let replica = CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + limits(), + ) + .unwrap(); + let records = Arc::new(HashMap::from([( + target.cell_id(), + Record { + target, + incarnation, + catalog: proof.clone(), + replica: replica.clone(), + authority: cell_authority.clone(), + }, + )])); + let mut nodes = Vec::new(); + let mut boots = Vec::new(); + for index in 0..count { + let intent = journal + .register_initial_intent( + &NodeIntent::initial(scope(), node_id(index), session(index)).unwrap(), + ) + .await + .unwrap(); + let mut builder = CellNodeBuilder::new(app.clone()) + .with_runtime( + SqlWorkerPool::new(2, 8) + .unwrap() + .with_native_memory_limit(128 << 20) + .unwrap(), + 64 << 20, + ) + .with_replica_host(Host::default().with_local_disk_budget(DiskBudget::new(8 << 30))) + .with_session(session(index)) + .with_fleet_startup_intent(intent.clone()); + if index != 0 { + builder = builder.with_follower_store( + root.path().join(format!("follower-{index}")), + limits(), + DiskBudget::new(1 << 30), + ); + } + let node = Arc::new(builder.build().unwrap()); + node.install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + node.install_fleet_actions( + scope(), + node_id(index), + journal.clone(), + Arc::new(adapters::Cells { + records: records.clone(), + local: index, + root: root.path().into(), + }), + ) + .unwrap(); + let ad = startup::advertisement(index, &node, &intent).await.unwrap(); + let spec = startup::spec(&intent).unwrap(); + let original = + startup::enroll(&journal, &directory, &spec, ad.clone(), clock().unwrap()) + .await + .unwrap(); + let guard = NodeLeaseGuard::new(clock().unwrap(), ad.expires_at_ms()).unwrap(); + node.install_node_lease_for_startup(guard.clone()).unwrap(); + node.confirm_fleet_startup(journal.as_ref(), spec.key().unwrap()) + .await + .unwrap(); + let observed = directory + .load(session(index), clock().unwrap()) + .await + .unwrap() + .unwrap(); + node.install_fleet_boot_withdrawal( + directory.clone(), + observed, + original, + journal.clone(), + ) + .unwrap(); + boots.push(startup::BootOwner { + node: node.clone(), + directory: directory.clone(), + spec, + advertisement: ad, + guard: Some(guard), + }); + nodes.push(node); + } + let locals = (1..count) + .map(|index| { + let store = nodes[index] + .try_owned_component::(cellule_host::FOLLOWER_STORE_COMPONENT) + .unwrap() + .unwrap(); + ( + node_id(index), + LocalFollowerTransport::new(node_id(index), (*store).clone()), + ) + }) + .collect(); + let transport = Arc::new(Transport { + directory: directory.clone(), + locals, + requests: Mutex::new(Vec::new()), + lose_retire: AtomicBool::new(false), + appends: AtomicUsize::new(0), + retirement_retry: Mutex::new(None), + }); + let authority = Arc::new(Authority { + directory: directory.clone(), + serial: tokio::sync::Mutex::new(()), + closed: Mutex::new(HashMap::new()), + members: Mutex::new(HashMap::new()), + coverage_gate: Mutex::new(None), + attempts: AtomicUsize::new(0), + lose_reply: AtomicBool::new(false), + }); + let provider = Arc::new(Provider { + directory: directory.clone(), + transport: transport.clone(), + authority: authority.clone(), + lease: boots[0].guard.as_ref().unwrap().clone(), + prepared: AtomicUsize::new(0), + max_epochs, + events: Mutex::new(Vec::new()), + preparation: Mutex::new(None), + }); + nodes[0] + .install_fleet_node_durability_provider( + scope(), + node_id(0), + journal.clone(), + provider.clone(), + NodeDurabilitySupervisorConfig::new( + scope().application, + limits(), + 1, + count, + Duration::from_millis(10), + Duration::from_secs(60), + u64::MAX, + ) + .unwrap(), + ) + .unwrap(); + // No enrollment producer runs until all original boots are retained. + let snapshot = journal.load_snapshot(scope()).await.unwrap(); + journal + .bootstrap_registry(snapshot.registry()) + .await + .unwrap(); + for index in 1..count { + nodes[index].start().unwrap(); + if index >= 3 { + continue; + } + let ad = boots[index] + .refresh_capacity( + index, + journal.as_ref(), + Instant::now() + Duration::from_secs(3), + ) + .await + .unwrap(); + assert!(ad.capacity().follower_free_bytes > 0); + } + nodes[0].start().unwrap(); + until(|| nodes[0].runtime().node_durability().is_some()).await; + // Recruitment can use the source's closed startup advertisement for + // outbound work. A live replacement proof requires its real Ready-mode + // heartbeat, never an invented admission sample. + boots[0] + .refresh_capacity(0, journal.as_ref(), Instant::now() + Duration::from_secs(3)) + .await + .unwrap(); + let native = Fixture { + root, + layout, + node: nodes[0].clone(), + directory, + journal, + provider, + transport, + authority, + }; + let handle = nodes[0] + .runtime() + .bootstrap( + proof, + replica, + cell_authority, + initial, + native.root.path().join("source.sqlite"), + |tx| { + tx.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES (17)", + )?; + Ok(()) + }, + ) + .await + .unwrap(); + let now = clock().unwrap(); + let (entered, resume, completed) = native.authority.pause_coverage(); + let outcome = handle + .execute( + MutationIdentity { + request_id: RequestId::from_bytes([1; 16]), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }, + Digest::from_bytes([1; 32]), + now, + 64, + 64, + |tx| { + tx.execute_batch("UPDATE counter SET value = 29")?; + Ok(HandlerOutcome::Success(29i64.to_be_bytes().to_vec())) + }, + ) + .await + .unwrap(); + assert!(matches!(outcome, StoredOutcome::Success { .. })); + super::super::captured(entered).await; + assert!( + native + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .unwrap() + .advertisement() + .log() + .unwrap() + .active() + ); + resume.send(()).unwrap(); + super::super::captured(completed).await; + until(|| native.transport.appends.load(Ordering::Acquire) >= 2).await; + let snapshot = native.journal.load_snapshot(scope()).await.unwrap(); + native + .journal + .claim_controller( + scope(), + snapshot.head().revision(), + session(9), + clock().unwrap(), + ) + .await + .unwrap(); + Self { + native, + nodes, + boots, + handle, + } + } + + pub async fn finish(self) { + // Keep the enrolled receiving boots live through the canonical owner drain. + for node in &self.nodes { + node.shutdown().await.unwrap(); + } + for boot in &self.boots { + boot.withdraw(self.native.journal.as_ref()).await.unwrap(); + } + for node in &self.nodes { + assert_eq!(node.state(), NodeState::Stopped); + assert_eq!(node.stats().retained_bytes(), 0); + assert_eq!(node.stats().local_disk_reserved_bytes(), 0); + } + let rows = self.native.rows().await; + let followers = rows + .iter() + .filter(|row| matches!(row.spec().role, EnrollmentRole::Follower { .. })) + .count(); + let expected_followers = if self.nodes.len() == 4 && self.native.provider.max_epochs >= 2 { + 2 * self.native.provider.max_epochs + } else { + 2 + }; + assert_eq!(rows.len(), self.nodes.len() + expected_followers); + assert!( + rows.iter() + .all(|row| row.status() == EnrollmentStatus::Retired) + ); + assert_eq!(followers, expected_followers); + self.native.journal.close().await.unwrap(); + } +} diff --git a/crates/cellule-host/minion/scenario/follower_tests/fixture/mod.rs b/crates/cellule-host/minion/scenario/follower_tests/fixture/mod.rs new file mode 100644 index 00000000..98726d57 --- /dev/null +++ b/crates/cellule-host/minion/scenario/follower_tests/fixture/mod.rs @@ -0,0 +1,571 @@ +use super::*; + +pub(super) struct Authority { + directory: NodeDirectory, + serial: tokio::sync::Mutex<()>, + closed: Mutex>, + members: Mutex>>, + coverage_gate: Mutex< + Option<( + tokio::sync::oneshot::Sender<()>, + tokio::sync::oneshot::Receiver<()>, + tokio::sync::oneshot::Sender<()>, + )>, + >, + pub attempts: AtomicUsize, + pub lose_reply: AtomicBool, +} +impl Authority { + pub fn pause_coverage( + &self, + ) -> ( + tokio::sync::oneshot::Receiver<()>, + tokio::sync::oneshot::Sender<()>, + tokio::sync::oneshot::Receiver<()>, + ) { + let (entered, captured) = tokio::sync::oneshot::channel(); + let (resume, release) = tokio::sync::oneshot::channel(); + let (done, completed) = tokio::sync::oneshot::channel(); + *self.coverage_gate.lock().unwrap() = Some((entered, release, done)); + (captured, resume, completed) + } + async fn current( + &self, + epoch: u64, + ) -> cellule_runtime::Result { + let observed = self + .directory + .load_if_live(session(0), clock()?) + .await? + .ok_or(Error::Fenced)?; + if observed.advertisement().node() != node_id(0) + || observed.advertisement().log().is_none_or(|log| { + log.epoch() != epoch + || self + .members + .lock() + .unwrap() + .get(&epoch) + .is_none_or(|members| log.members() != members) + }) + { + return Err(Error::Fenced); + } + Ok(observed) + } +} +impl NodeLogAuthority for Authority { + fn activate<'a>(&'a self, epoch: u64) -> BoxFuture<'a, cellule_runtime::Result<()>> { + Box::pin(async move { + let _serial = self.serial.lock().await; + let observed = self.current(epoch).await?; + self.directory.activate_log(&observed, clock()?).await?; + Ok(()) + }) + } + fn advance_coverage<'a>( + &'a self, + epoch: u64, + through: u64, + ) -> BoxFuture<'a, cellule_runtime::Result<()>> { + Box::pin(async move { + let gate = self.coverage_gate.lock().unwrap().take(); + let done = if let Some((entered, release, done)) = gate { + entered.send(()).unwrap(); + release.await.unwrap(); + Some(done) + } else { + None + }; + let _serial = self.serial.lock().await; + let observed = self.current(epoch).await?; + self.directory + .advance_log_coverage(&observed, through, clock()?) + .await?; + if let Some(done) = done { + done.send(()).unwrap(); + } + Ok(()) + }) + } + fn close<'a>( + &'a self, + retirement: &'a NodeLogRetirementObservation, + ) -> BoxFuture<'a, cellule_runtime::Result<()>> { + Box::pin(async move { + retirement.confirmed()?; + let _serial = self.serial.lock().await; + self.attempts.fetch_add(1, Ordering::AcqRel); + if let Some(original) = self + .closed + .lock() + .unwrap() + .get(&retirement.barrier().log_epoch()) + { + assert_eq!(original, retirement.barrier()); + return Ok(()); + } + let observed = self.current(retirement.barrier().log_epoch()).await?; + let closed = self + .directory + .close_log(&observed, retirement.barrier(), clock()?) + .await?; + assert!(closed.advertisement().log().is_none()); + // Preserve the exact checked close receipt before losing its reply. + self.closed.lock().unwrap().insert( + retirement.barrier().log_epoch(), + retirement.barrier().clone(), + ); + if self.lose_reply.swap(false, Ordering::AcqRel) { + return Err(Error::Node("original canonical close reply lost")); + } + Ok(()) + }) + } +} +pub(super) struct Transport { + directory: NodeDirectory, + locals: Vec<(NodeId, LocalFollowerTransport)>, + pub requests: Mutex>, + pub lose_retire: AtomicBool, + pub appends: AtomicUsize, + retirement_retry: Mutex< + Option<( + tokio::sync::oneshot::Sender<()>, + tokio::sync::oneshot::Receiver<()>, + )>, + >, +} +impl Transport { + pub fn pause_retirement_retry( + &self, + ) -> ( + tokio::sync::oneshot::Receiver<()>, + tokio::sync::oneshot::Sender<()>, + ) { + let (entered, captured) = tokio::sync::oneshot::channel(); + let (resume, release) = tokio::sync::oneshot::channel(); + *self.retirement_retry.lock().unwrap() = Some((entered, release)); + (captured, resume) + } + fn local(&self, node: NodeId) -> &LocalFollowerTransport { + &self + .locals + .iter() + .find(|(member, _)| *member == node) + .unwrap() + .1 + } +} +impl NodeLogTransport for Transport { + fn append<'a>( + &'a self, + member: NodeId, + request: AppendRequest, + ) -> BoxFuture<'a, cellule_runtime::Result> { + Box::pin(async move { + self.directory + .authorize_log_append( + request.leader_session, + member, + request.log_epoch, + request.covered_through, + clock()?, + ) + .await?; + let receipt = self.local(member).append(member, request).await?; + self.appends.fetch_add(1, Ordering::AcqRel); + Ok(receipt) + }) + } + fn seal<'a>( + &'a self, + member: NodeId, + request: SealRequest, + ) -> BoxFuture<'a, cellule_runtime::Result> { + self.local(member).seal(member, request) + } + fn tail<'a>( + &'a self, + member: NodeId, + request: TailRequest, + ) -> BoxFuture<'a, cellule_runtime::Result>> { + self.local(member).tail(member, request) + } + fn retire<'a>( + &'a self, + member: NodeId, + request: RetireRequest, + ) -> BoxFuture<'a, cellule_runtime::Result> { + Box::pin(async move { + let retry = { + let mut requests = self.requests.lock().unwrap(); + let retry = requests.iter().any(|(peer, original)| { + *peer == member + && original.leader_session == request.leader_session + && original.log_epoch == request.log_epoch + }); + requests.push((member, request)); + retry + }; + let gate = if retry { + self.retirement_retry.lock().unwrap().take() + } else { + None + }; + if let Some((entered, release)) = gate { + entered.send(()).unwrap(); + release.await.unwrap(); + } + self.directory + .authorize_log_retire( + request.leader_session, + member, + request.log_epoch, + request.covered_through, + clock()?, + ) + .await?; + let receipt = self.local(member).retire(member, request).await?; + if member == node_id(1) && self.lose_retire.load(Ordering::Acquire) { + return Err(Error::Node("original member retirement reply lost")); + } + Ok(receipt) + }) + } +} +pub(super) struct Provider { + directory: NodeDirectory, + transport: Arc, + authority: Arc, + lease: NodeLeaseGuard, + pub prepared: AtomicUsize, + max_epochs: usize, + pub events: Mutex>, + preparation: Mutex< + Option<( + tokio::sync::oneshot::Sender<()>, + tokio::sync::oneshot::Receiver<()>, + )>, + >, +} +impl Provider { + pub fn pause_preparation( + &self, + ) -> ( + tokio::sync::oneshot::Receiver<()>, + tokio::sync::oneshot::Sender<()>, + ) { + let (entered, captured) = tokio::sync::oneshot::channel(); + let (resume, release) = tokio::sync::oneshot::channel(); + *self.preparation.lock().unwrap() = Some((entered, release)); + (captured, resume) + } +} +impl FleetNodeDurabilityProvider for Provider { + fn rotation_required( + self: Arc, + _live_node_limit: usize, + ) -> std::pin::Pin> + Send>> { + Box::pin(async { Ok(false) }) + } + + fn prepare( + self: Arc, + _limits: Limits, + bytes: u64, + live: usize, + ) -> std::pin::Pin< + Box< + dyn std::future::Future>> + + Send, + >, + > { + Box::pin(async move { + // Each read-only preparation uses a never-reused epoch. The original + // single-epoch profiles remain unable to invent a replacement. + let index = self.prepared.fetch_add(1, Ordering::AcqRel); + if index >= self.max_epochs { + return Ok(None); + } + let gate = self.preparation.lock().unwrap().take(); + if let Some((entered, release)) = gate { + entered.send(()).unwrap(); + release.await.unwrap(); + } + let now = clock()?; + let source = self + .directory + .load_if_live(session(0), now) + .await? + .ok_or(Error::Fenced)?; + let prepared = self + .directory + .prepare_log_enrollment(&source, index as u64 + 1, bytes, live, now) + .await? + .ok_or(Error::Fenced)?; + let attempt = self + .directory + .prepare_log_enrollment_attempt(&prepared, now) + .await?; + self.authority + .members + .lock() + .unwrap() + .insert(prepared.log().epoch(), prepared.log().members().to_vec()); + Ok(Some(FleetNodeLogRecruitment::new( + self.directory.clone(), + attempt, + self.transport.clone(), + self.authority.clone(), + self.lease.clone(), + Default::default(), + )?)) + }) + } + fn rotation_event(&self, event: NodeDurabilityRotation) { + self.events.lock().unwrap().push(event); + } +} +pub(super) struct Fixture { + pub root: tempfile::TempDir, + pub layout: CellStorageLayout, + pub node: Arc, + pub directory: NodeDirectory, + pub journal: Arc, + pub provider: Arc, + pub transport: Arc, + pub authority: Arc, +} +impl Fixture { + pub async fn new() -> Self { + let root = tempfile::tempdir().unwrap(); + let app = application::compile().unwrap(); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + ObjectPath::from("enrolled-followers"), + [3; 16], + ); + let directory = NodeDirectory::new( + layout.clone(), + scope().fleet, + Digest::from_bytes([31; 32]), + app.registry().release_digest(), + ); + let now = clock().unwrap(); + for index in 0..3 { + let ad = NodeAdvertisement::sign( + node_id(index), + session(index), + owner(index).endpoint, + scope().fleet, + Digest::from_bytes([30; 32]), + Digest::from_bytes([31; 32]), + app.registry().release_digest(), + &SigningKey::from_bytes(&[index as u8 + 1; 32]), + 1, + now, + now + 30_000, + app.registry().module_digests(), + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + follower_free_bytes: 1 << 30, + free_memory_bytes: 1 << 30, + free_disk_bytes: 1 << 30, + job_credits: 4, + log_protocol: 1, + ..Default::default() + }, + ) + .unwrap(); + directory.create(ad, now).await.unwrap(); + } + let journal = Arc::new( + SqliteJournal::open( + root.path().join("journal.sqlite"), + scope(), + FleetProfile::default(), + now, + ) + .await + .unwrap(), + ); + for index in 0..3 { + journal + .register_initial_intent( + &NodeIntent::initial(scope(), node_id(index), session(index)).unwrap(), + ) + .await + .unwrap(); + } + let lease = NodeLeaseGuard::new(now, now + 60_000).unwrap(); + let node = Arc::new( + CellNodeBuilder::new(app) + .with_runtime( + SqlWorkerPool::new(2, 8) + .unwrap() + .with_native_memory_limit(128 << 20) + .unwrap(), + 16 << 20, + ) + .with_session(session(0)) + .with_replica_host(Host::default().with_local_disk_budget(DiskBudget::new(8 << 30))) + .build() + .unwrap(), + ); + node.install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + node.install_node_lease_for_startup(lease.clone()).unwrap(); + let locals = (1..3) + .map(|index| { + let store = FollowerStore::open( + root.path().join(format!("follower-{index}")), + Limits::default(), + DiskBudget::new(1 << 30), + ) + .unwrap(); + ( + node_id(index), + LocalFollowerTransport::new(node_id(index), store), + ) + }) + .collect(); + let transport = Arc::new(Transport { + directory: directory.clone(), + locals, + requests: Mutex::new(Vec::new()), + lose_retire: AtomicBool::new(false), + appends: AtomicUsize::new(0), + retirement_retry: Mutex::new(None), + }); + let authority = Arc::new(Authority { + directory: directory.clone(), + serial: tokio::sync::Mutex::new(()), + closed: Mutex::new(HashMap::new()), + members: Mutex::new(HashMap::new()), + coverage_gate: Mutex::new(None), + attempts: AtomicUsize::new(0), + lose_reply: AtomicBool::new(false), + }); + let provider = Arc::new(Provider { + directory: directory.clone(), + transport: transport.clone(), + authority: authority.clone(), + lease, + prepared: AtomicUsize::new(0), + max_epochs: 1, + events: Mutex::new(Vec::new()), + preparation: Mutex::new(None), + }); + Self { + root, + layout, + node, + directory, + journal, + provider, + transport, + authority, + } + } + pub fn install(&self) { + self.install_limits(limits()); + } + pub fn install_limits(&self, limits: Limits) { + self.node + .install_fleet_node_durability_provider( + scope(), + node_id(0), + self.journal.clone(), + self.provider.clone(), + NodeDurabilitySupervisorConfig::new( + scope().application, + limits, + 1, + 3, + Duration::from_millis(10), + Duration::from_secs(60), + u64::MAX, + ) + .unwrap(), + ) + .unwrap(); + self.node.start().unwrap(); + } + pub async fn rows(&self) -> Vec { + tokio::time::timeout(Duration::from_secs(3), async { + loop { + let version = self + .journal + .load_snapshot(scope()) + .await + .unwrap() + .registry(); + match self.journal.enrollments_page(version, None, 128).await { + Ok(page) => return page.entries().to_vec(), + Err(error) + if matches!( + error + .downcast_ref::( + ), + Some(cellule_runtime::fleet::operations::OperationError::Conflict) + ) => + { + tokio::task::yield_now().await + } + Err(error) => panic!("follower registry scan: {error}"), + } + } + }) + .await + .unwrap() + } + pub async fn installed(&self) { + until(|| self.node.runtime().node_durability().is_some()).await; + let rows = self.rows().await; + assert_eq!(rows.len(), 2); + assert!( + rows.iter() + .all(|row| row.status() == EnrollmentStatus::Established) + ); + } + pub async fn finish(self) { + self.node.shutdown().await.unwrap(); + assert_eq!(self.node.state(), NodeState::Stopped); + assert!(self.rows().await.iter().all(|row| matches!( + row.status(), + EnrollmentStatus::Retired | EnrollmentStatus::Refused + ))); + assert_eq!(self.node.stats().retained_bytes(), 0); + assert_eq!(self.node.stats().local_disk_reserved_bytes(), 0); + assert!( + self.node + .follower_enrollment_completion(1) + .unwrap() + .is_none() + ); + self.journal.close().await.unwrap(); + assert!(self.root.path().exists()); + } +} +pub(super) async fn until(mut predicate: impl FnMut() -> bool) { + tokio::time::timeout(Duration::from_secs(3), async { + while !predicate() { + tokio::time::sleep(Duration::from_millis(5)).await; + } + }) + .await + .unwrap(); +} + +pub(super) fn limits() -> Limits { + Limits { + max_database_bytes: 64 << 20, + max_capture_bytes: 16 << 20, + ..Limits::default() + } +} + +mod managed; +pub(super) use managed::ManagedFixture; diff --git a/crates/cellule-host/minion/scenario/follower_tests/inventory.rs b/crates/cellule-host/minion/scenario/follower_tests/inventory.rs new file mode 100644 index 00000000..fc96f55b --- /dev/null +++ b/crates/cellule-host/minion/scenario/follower_tests/inventory.rs @@ -0,0 +1,219 @@ +//! Public capture remains advisory through the original supervisor's paused I/O. +use super::*; +use cellule_host::{FollowerEnrollmentInventoryCursor, FollowerEnrollmentInventoryPage}; + +pub(super) fn capture(fixture: &Fixture) -> FollowerEnrollmentInventoryPage { + fixture + .node + .fleet_follower_enrollments_page(None, 32, clock().unwrap()) + .unwrap() + .unwrap() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn preparation_without_a_request_is_visible_and_inventory_does_not_wait() { + let fixture = Fixture::new().await; + assert!( + fixture + .node + .fleet_follower_enrollments_page(None, 1, clock().unwrap()) + .unwrap() + .is_none() + ); + let (entered, resume) = fixture.provider.pause_preparation(); + fixture.install(); + captured(entered).await; + let before = fixture.node.stats().retained_bytes(); + let now = clock().unwrap(); + let page = fixture + .node + .fleet_follower_enrollments_page(None, 1, now) + .unwrap() + .unwrap(); + assert_eq!(page.scope(), scope()); + assert_eq!(page.node(), node_id(0)); + assert_eq!(page.session(), session(0)); + assert_eq!(page.observed_at_ms(), now); + assert!(page.protocol_busy()); + assert!(!page.draining()); + assert_eq!(page.pending_epoch(), None); + assert_eq!(page.total_epochs(), 0); + assert!(page.entries().is_empty()); + assert!(page.next().is_none()); + assert!(fixture.rows().await.is_empty()); + assert!(fixture.node.runtime().node_durability().is_none()); + assert!(fixture.transport.requests.lock().unwrap().is_empty()); + assert_eq!(fixture.node.stats().retained_bytes(), before + (1 << 20)); + drop(page); + assert_eq!(fixture.node.stats().retained_bytes(), before); + resume.send(()).unwrap(); + fixture.installed().await; + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn pending_member_capture_preserves_inputs_budget_and_changed_continuations() { + let fixture = Fixture::new().await; + let (first, resume_first) = fixture.journal.pause_next_enrollment_reply(false, false); + fixture.install(); + captured(first).await; + let (second, resume_second) = fixture.journal.pause_before_enrollment_acceptance(); + resume_first.send(()).unwrap(); + captured(second).await; + let before = fixture.node.stats().retained_bytes(); + let original = fixture + .node + .follower_enrollment_completion(1) + .unwrap() + .unwrap(); + let page = capture(&fixture); + assert!(page.protocol_busy()); + assert_eq!(page.pending_epoch(), Some(1)); + assert_eq!(page.total_epochs(), 1); + assert!(page.next().is_none()); + let entry = &page.entries()[0]; + assert_eq!(entry.epoch, 1); + assert_eq!(entry.attempt, original.attempt.evidence_digest().unwrap()); + assert_eq!(entry.members.len(), 2); + for (observed, original) in entry.members.iter().zip(&original.members) { + assert_eq!(observed.spec, original.spec); + assert_eq!(observed.accepted, original.accepted); + assert_eq!(observed.event, original.event); + assert_eq!(observed.published, original.published); + } + assert!(entry.members[0].accepted.is_some()); + assert!(entry.members[1].accepted.is_none()); + assert!(!entry.native_started); + assert!(!entry.delivered); + assert!(!entry.no_effect); + assert!(!entry.native_closed); + assert!(entry.enrollment.is_none()); + assert!(entry.refusal.is_none()); + assert!(entry.retirement.is_none()); + assert_eq!(fixture.node.stats().retained_bytes(), before + (1 << 20)); + let mut bytes = [0; 40]; + bytes[..32].copy_from_slice(page.topology().as_bytes()); + bytes[32..].copy_from_slice(&1u64.to_le_bytes()); + let cursor = FollowerEnrollmentInventoryCursor::from_bytes(&bytes).unwrap(); + drop(page); + assert_eq!(fixture.node.stats().retained_bytes(), before); + let continuation = fixture + .node + .fleet_follower_enrollments_page(Some(cursor), 1, clock().unwrap()) + .unwrap() + .unwrap(); + assert_eq!(continuation.total_epochs(), 1); + assert!(continuation.entries().is_empty()); + drop(continuation); + bytes[32..].copy_from_slice(&2u64.to_le_bytes()); + let absent = FollowerEnrollmentInventoryCursor::from_bytes(&bytes).unwrap(); + assert!( + fixture + .node + .fleet_follower_enrollments_page(Some(absent), 1, clock().unwrap()) + .is_err() + ); + for (limit, now) in [(0, 0), (33, 0), (1, -1)] { + assert!( + fixture + .node + .fleet_follower_enrollments_page(None, limit, now) + .is_err() + ); + } + let runtime = fixture.node.runtime(); + let available = fixture.node.stats().retained_capacity_bytes() - before; + let exhausted = runtime.try_reserve_node_bytes(available).unwrap(); + assert!(matches!( + fixture + .node + .fleet_follower_enrollments_page(None, 1, clock().unwrap()), + Err(Error::Capacity(_)) + )); + drop(exhausted); + assert_eq!(fixture.node.stats().retained_bytes(), before); + assert_eq!(fixture.rows().await.len(), 1); + assert!(fixture.node.runtime().node_durability().is_none()); + resume_second.send(()).unwrap(); + fixture.installed().await; + assert!( + fixture + .node + .fleet_follower_enrollments_page(Some(cursor), 1, clock().unwrap()) + .is_err() + ); + let page = capture(&fixture); + let entry = &page.entries()[0]; + assert!(entry.native_started); + assert!(entry.delivered); + assert!(entry.enrollment.is_some()); + assert!(entry.members.iter().all(|member| member.published)); + drop(page); + let node = fixture.node.clone(); + fixture.finish().await; + assert!( + node.fleet_follower_enrollments_page(None, 1, clock().unwrap()) + .unwrap() + .is_none() + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn requested_rotation_inventory_preserves_original_failed_member_and_error() { + let fixture = Fixture::new().await; + fixture.install(); + fixture.installed().await; + fixture.transport.lose_retire.store(true, Ordering::Release); + let (retry, resume) = fixture.transport.pause_retirement_retry(); + let request = fixture.node.request_node_log_rotation(1).unwrap(); + captured(retry).await; + let completion = tokio::time::timeout(Duration::from_secs(3), async { + loop { + let completion = fixture + .node + .follower_enrollment_completion(1) + .unwrap() + .unwrap(); + if completion.execution_error.is_some() && completion.retirement.is_some() { + return completion; + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + }) + .await + .unwrap(); + let before = fixture.node.stats().retained_bytes(); + let page = capture(&fixture); + let entry = &page.entries()[0]; + assert_eq!(entry.epoch, 1); + assert!(entry.delivered); + assert!(!entry.native_closed); + assert!(Arc::ptr_eq( + entry.execution_error.as_ref().unwrap(), + completion.execution_error.as_ref().unwrap() + )); + assert!(Arc::ptr_eq( + entry.retirement.as_ref().unwrap(), + completion.retirement.as_ref().unwrap() + )); + assert!(entry.retirement.as_ref().unwrap().confirmed().is_err()); + assert_eq!(entry.retirement.as_ref().unwrap().members().len(), 2); + assert_eq!(fixture.authority.attempts.load(Ordering::Acquire), 0); + assert!( + fixture + .rows() + .await + .iter() + .all(|row| row.status() == EnrollmentStatus::Established) + ); + assert_eq!(fixture.node.stats().retained_bytes(), before + (1 << 20)); + drop(page); + assert_eq!(fixture.node.stats().retained_bytes(), before); + drop(request); + fixture + .transport + .lose_retire + .store(false, Ordering::Release); + resume.send(()).unwrap(); + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/follower_tests/mod.rs b/crates/cellule-host/minion/scenario/follower_tests/mod.rs new file mode 100644 index 00000000..d9190cfa --- /dev/null +++ b/crates/cellule-host/minion/scenario/follower_tests/mod.rs @@ -0,0 +1,44 @@ +//! Public host producer against native followers and the shared durable journal. +use super::*; +use bytes::Bytes; +use cellule_host::{ + FacilityResult, FleetNodeDurabilityProvider, FleetNodeLogRecruitment, NodeDurabilityRotation, + NodeDurabilitySupervisorConfig, +}; +use cellule_runtime::{ + Error, + fleet::operations::{EnrollmentEvent, EnrollmentRecord, EnrollmentStatus}, + follower::{FollowerReceipt, FollowerStore}, + node::{ + NodeAdvertisement, NodeCapacity, NodeDirectory, NodeFailureDomain, + durability::NodeLogAuthority, + log::{NodeLogRetirementObservation, NodeLogRotationBarrier}, + log_transport::{ + AppendRequest, LocalFollowerTransport, NodeLogTransport, RetireRequest, SealRequest, + TailRequest, + }, + }, +}; +use ed25519_dalek::SigningKey; +use futures_util::future::BoxFuture; +use std::sync::{ + Mutex, + atomic::{AtomicBool, AtomicUsize, Ordering}, +}; + +mod fixture; +use fixture::*; +mod aggregate; +mod coverage; +mod evacuation; +mod inventory; +mod observation; +mod persisted; +mod tests; + +async fn captured(reply: tokio::sync::oneshot::Receiver<()>) { + tokio::time::timeout(Duration::from_secs(3), reply) + .await + .unwrap() + .unwrap(); +} diff --git a/crates/cellule-host/minion/scenario/follower_tests/observation.rs b/crates/cellule-host/minion/scenario/follower_tests/observation.rs new file mode 100644 index 00000000..815024ea --- /dev/null +++ b/crates/cellule-host/minion/scenario/follower_tests/observation.rs @@ -0,0 +1,260 @@ +//! Retained role evidence through the exported observation and reconciler paths. +use super::*; +use cellule_host::fleet::{ + FleetActionCompletion, FleetAdapterFuture, FleetObservation, FleetObserver, FleetOwnedCell, + FleetRoleCoverage, FleetRoster, FleetTransport, +}; +use cellule_runtime::fleet::operations::{ + FleetAction, FleetInspectionObservation, FleetInspectionRequest, +}; + +async fn nodes(fixture: &ManagedFixture) -> Vec { + let mut nodes = Vec::new(); + for index in 0..fixture.nodes.len() { + nodes.push( + fixture + .native + .directory + .load(session(index), clock().unwrap()) + .await + .unwrap() + .unwrap() + .advertisement() + .clone(), + ); + } + nodes +} + +async fn enable(fixture: &ManagedFixture) { + let snapshot = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + fixture + .native + .journal + .set_scheduling(snapshot.registry(), true) + .await + .unwrap(); +} + +fn observation( + roster: &FleetRoster, + coverage: &FleetRoleCoverage, + nodes: Vec, + cells: Vec, +) -> FleetObservation { + FleetObservation::new( + scope(), + roster.snapshot().registry(), + roster.snapshot().registry().revision(), + coverage.interval().0, + clock().unwrap(), + false, + nodes, + cells, + ) + .unwrap() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn exported_observation_retains_original_role_graph_and_rejects_replacement_or_restamping() { + let fixture = ManagedFixture::new().await; + let roster = aggregate::roster(&fixture).await; + let (mut native, mut foreign, mut sequence) = coverage::captures(&fixture, &roster).await; + coverage::native_rechecks(&fixture, &roster, &mut native, &mut sequence).await; + coverage::foreign_rechecks(&fixture, &roster, &mut foreign).await; + let graph = coverage::check(&roster, &native, &foreign).unwrap(); + let digest = graph.digest(); + let interval = graph.interval(); + let advertised = nodes(&fixture).await; + let cells = native + .iter() + .flat_map(|i| i.cells().iter().cloned()) + .collect::>(); + let observation = observation(&roster, &graph, advertised.clone(), cells.clone()) + .with_role_coverage(graph) + .unwrap(); + assert_eq!(observation.role_coverage().unwrap().digest(), digest); + assert_eq!(observation.role_coverage().unwrap().interval(), interval); + assert_eq!( + observation.role_coverage().unwrap().snapshot(), + roster.snapshot() + ); + assert!( + observation + .with_role_coverage(coverage::check(&roster, &native, &foreign).unwrap()) + .is_err() + ); + for (started, finished, registry) in [ + ( + interval.0 + 1, + clock().unwrap(), + roster.snapshot().registry(), + ), + (interval.0, interval.1 - 1, roster.snapshot().registry()), + ( + interval.0, + clock().unwrap(), + roster + .snapshot() + .registry() + .advance(roster.snapshot().registry().revision()) + .unwrap(), + ), + ] { + let observation = FleetObservation::new( + scope(), + registry, + registry.revision(), + started, + finished, + false, + advertised.clone(), + cells.clone(), + ) + .unwrap(); + assert!( + observation + .with_role_coverage(coverage::check(&roster, &native, &foreign).unwrap()) + .is_err() + ); + } + drop((native, foreign)); + fixture.finish().await; +} + +struct FreshObserver(Arc); +impl FleetObserver for FreshObserver { + fn observe<'a>( + &'a self, + roster: &'a FleetRoster, + _: i64, + _: Instant, + ) -> FleetAdapterFuture<'a, FleetObservation> { + Box::pin(async move { + let (mut native, mut foreign, mut sequence) = coverage::captures(&self.0, roster).await; + coverage::native_rechecks(&self.0, roster, &mut native, &mut sequence).await; + coverage::foreign_rechecks(&self.0, roster, &mut foreign).await; + let graph = coverage::check(roster, &native, &foreign)?; + let cells = native + .iter() + .flat_map(|i| i.cells().iter().cloned()) + .collect(); + Ok(observation(roster, &graph, nodes(&self.0).await, cells) + .with_role_coverage(graph)?) + }) + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn current_role_graph_cannot_upgrade_partial_adapter_observation_to_count_coverage() { + let fixture = Arc::new(ManagedFixture::new().await); + enable(&fixture).await; + let transport = Arc::new(RetainedObserver { + observation: Mutex::new(None), + calls: AtomicUsize::new(0), + }); + let observer = Arc::new(FreshObserver(fixture.clone())); + let driver = FleetReconciler::new( + scope(), + session(9), + FleetProfile::default(), + fixture.native.journal.clone(), + observer.clone(), + transport.clone(), + ) + .unwrap(); + let report = driver + .reconcile_once(clock, Instant::now() + Duration::from_secs(5)) + .await + .unwrap(); + assert!( + report + .blockers + .contains(&cellule_runtime::fleet::operations::DrainBlocker::IncompleteObservation) + ); + assert_eq!(report.allocated, 0); + assert_eq!(transport.calls.load(Ordering::SeqCst), 0); + drop((driver, observer, transport)); + Arc::try_unwrap(fixture).ok().unwrap().finish().await; +} + +struct RetainedObserver { + observation: Mutex>, + calls: AtomicUsize, +} +impl FleetObserver for RetainedObserver { + fn observe<'a>( + &'a self, + _: &'a FleetRoster, + _: i64, + _: Instant, + ) -> FleetAdapterFuture<'a, FleetObservation> { + Box::pin(async move { Ok(self.observation.lock().unwrap().take().unwrap()) }) + } +} +impl FleetTransport for RetainedObserver { + fn dispatch<'a>( + &'a self, + _: &'a FleetAction, + _: Instant, + ) -> FleetAdapterFuture<'a, Arc> { + self.calls.fetch_add(1, Ordering::SeqCst); + Box::pin(async { Err(invalid("unexpected effect before role barrier validation")) }) + } + fn inspect<'a>( + &'a self, + _: &'a FleetInspectionRequest, + _: Instant, + ) -> FleetAdapterFuture<'a, Arc> { + self.calls.fetch_add(1, Ordering::SeqCst); + Box::pin(async { + Err(invalid( + "unexpected inspection before role barrier validation", + )) + }) + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reconciler_refuses_role_graph_from_an_earlier_head_with_the_same_registry() { + let fixture = ManagedFixture::new().await; + enable(&fixture).await; + let roster = aggregate::roster(&fixture).await; + let (mut native, mut foreign, mut sequence) = coverage::captures(&fixture, &roster).await; + coverage::native_rechecks(&fixture, &roster, &mut native, &mut sequence).await; + coverage::foreign_rechecks(&fixture, &roster, &mut foreign).await; + let graph = coverage::check(&roster, &native, &foreign).unwrap(); + let cells = native + .iter() + .flat_map(|i| i.cells().iter().cloned()) + .collect(); + let observation = observation(&roster, &graph, nodes(&fixture).await, cells) + .with_role_coverage(graph) + .unwrap(); + let observer = Arc::new(RetainedObserver { + observation: Mutex::new(Some(observation)), + calls: AtomicUsize::new(0), + }); + let driver = FleetReconciler::new( + scope(), + session(9), + FleetProfile::default(), + fixture.native.journal.clone(), + observer.clone(), + observer.clone(), + ) + .unwrap(); + assert!(matches!( + driver + .reconcile_once(clock, Instant::now() + Duration::from_secs(5)) + .await, + Err(Error::Node("fleet role coverage roster differs")) + )); + let changed = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + assert_eq!(changed.registry(), roster.snapshot().registry()); + assert_ne!(changed.head(), roster.snapshot().head()); + assert!(changed.head().attempts().is_empty()); + assert_eq!(observer.calls.load(Ordering::SeqCst), 0); + drop((driver, observer, native, foreign)); + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/follower_tests/persisted/mod.rs b/crates/cellule-host/minion/scenario/follower_tests/persisted/mod.rs new file mode 100644 index 00000000..9d9a6f5f --- /dev/null +++ b/crates/cellule-host/minion/scenario/follower_tests/persisted/mod.rs @@ -0,0 +1,161 @@ +//! Native live-owner rotation with durable history and reconstructed confirmation. +use super::evacuation::{maintenance, rotated, spare}; +use super::*; +use cellule_host::{ + FollowerEvacuation, + fleet::{ + FleetFollowerEvacuationJournal, FleetFollowerEvacuationPublication, + FleetFollowerEvacuationVerifier, + }, +}; +use cellule_runtime::fleet::operations::{FollowerEvacuationRecord, FollowerReplacementPolicy}; +mod races; +mod tests; + +fn deadline() -> Instant { + Instant::now() + Duration::from_secs(5) +} +fn verifier(fixture: &ManagedFixture) -> FleetFollowerEvacuationVerifier { + FleetFollowerEvacuationVerifier::new( + fixture.native.directory.clone(), + Arc::new(Snapshots::new(fixture)), + ) +} +struct Snapshots { + nodes: Vec>, + pause: Mutex< + Option<( + tokio::sync::oneshot::Sender<()>, + tokio::sync::oneshot::Receiver<()>, + )>, + >, +} +impl Snapshots { + fn new(fixture: &ManagedFixture) -> Self { + Self { + nodes: fixture.nodes.clone(), + pause: Mutex::new(None), + } + } + fn pause_next( + &self, + ) -> ( + tokio::sync::oneshot::Receiver<()>, + tokio::sync::oneshot::Sender<()>, + ) { + let (entered, captured) = tokio::sync::oneshot::channel(); + let (resume, release) = tokio::sync::oneshot::channel(); + *self.pause.lock().unwrap() = Some((entered, release)); + (captured, resume) + } +} +impl cellule_host::fleet::FleetSnapshotTransport for Snapshots { + fn capture<'a>( + &'a self, + request: &'a cellule_host::fleet::FleetSnapshotRequest, + _deadline: Instant, + ) -> cellule_host::fleet::FleetAdapterFuture<'a, Arc> + { + Box::pin(async move { + let gate = self.pause.lock().unwrap().take(); + if let Some((entered, resume)) = gate { + let _ = entered.send(()); + let _ = resume.await; + } + let node = self + .nodes + .iter() + .enumerate() + .find(|(index, _)| { + node_id(*index) == request.node() && session(*index) == request.session() + }) + .map(|(_, node)| node) + .ok_or(Error::Fenced)?; + node.fleet_snapshot(request.clone()) + .await + .map_err(|error| Box::new(error) as Box) + }) + } +} +async fn setup() -> ( + ManagedFixture, + FollowerEvacuation, + FollowerReplacementPolicy, +) { + setup_epochs(2).await +} +async fn setup_epochs( + epochs: usize, +) -> ( + ManagedFixture, + FollowerEvacuation, + FollowerReplacementPolicy, +) { + let fixture = ManagedFixture::with_members(4, epochs).await; + let (original, operation) = maintenance(&fixture).await; + spare(&fixture).await; + rotated(&fixture).await; + let snapshot = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + let policy = FollowerReplacementPolicy::new(scope(), 1, 2).unwrap(); + fixture + .native + .journal + .set_follower_replacement_policy(&snapshot, policy, clock().unwrap()) + .await + .unwrap(); + let capture = fixture + .native + .node + .follower_evacuation(&original, &operation, 2, deadline()) + .await + .unwrap(); + (fixture, capture, policy) +} +async fn publish( + fixture: &ManagedFixture, + capture: &FollowerEvacuation, + policy: FollowerReplacementPolicy, +) -> FleetFollowerEvacuationPublication { + FleetFollowerEvacuationPublication::publish( + capture, + policy, + fixture.native.journal.as_ref(), + &verifier(fixture), + deadline(), + clock, + ) + .await + .unwrap() +} +async fn client(fixture: &ManagedFixture) -> SqliteJournal { + SqliteJournal::open( + fixture.native.root.path().join("journal.sqlite"), + scope(), + FleetProfile::default(), + clock().unwrap(), + ) + .await + .unwrap() +} +async fn stored(fixture: &ManagedFixture, digest: Digest) -> FollowerEvacuationRecord { + fixture + .native + .journal + .load_follower_evacuation(scope(), digest) + .await + .unwrap() + .unwrap() +} +async fn latest( + fixture: &ManagedFixture, + record: &FollowerEvacuationRecord, +) -> FollowerEvacuationRecord { + let snapshot = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + fixture + .native + .journal + .latest_follower_evacuation(&snapshot, record.operation().id(), record.original_key()) + .await + .unwrap() + .unwrap() +} diff --git a/crates/cellule-host/minion/scenario/follower_tests/persisted/races.rs b/crates/cellule-host/minion/scenario/follower_tests/persisted/races.rs new file mode 100644 index 00000000..353a6181 --- /dev/null +++ b/crates/cellule-host/minion/scenario/follower_tests/persisted/races.rs @@ -0,0 +1,200 @@ +use super::*; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn follower_policy_publication_fenced_native_owner_cannot_hide_behind_live_directory() { + let (fixture, capture, policy) = setup().await; + let result = publish(&fixture, &capture, policy).await; + let record = result.record().unwrap(); + result.confirmed().unwrap(); + let transport = Arc::new(Snapshots::new(&fixture)); + let verifier = + FleetFollowerEvacuationVerifier::new(fixture.native.directory.clone(), transport.clone()); + let (entered, resume) = transport.pause_next(); + let mut work = + Box::pin(verifier.recheck(fixture.native.journal.as_ref(), record, deadline(), clock)); + tokio::select! {result=&mut work=>panic!("unexpected completion {}",result.is_ok()),result=entered=>result.unwrap()} + fixture.boots[0].guard.as_ref().unwrap().fence(); + assert!( + fixture + .native + .directory + .load_if_live(session(0), clock().unwrap()) + .await + .unwrap() + .is_some() + ); + resume.send(()).unwrap(); + assert!(work.await.is_err()); + assert_eq!(stored(&fixture, record.digest().unwrap()).await, *record); + drop(capture); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn follower_policy_publication_adopts_lost_reply_and_preserves_original_error() { + let (fixture, capture, policy) = setup().await; + let (record, verifier) = (capture.durable_record(policy).unwrap(), verifier(&fixture)); + let (entered, resume) = fixture + .native + .journal + .pause_next_follower_evacuation_reply(true); + let mut work = Box::pin(FleetFollowerEvacuationPublication::publish( + &capture, + policy, + fixture.native.journal.as_ref(), + &verifier, + deadline(), + clock, + )); + tokio::select! {result=&mut work=>panic!("unexpected completion {}",result.is_ok()),result=entered=>result.unwrap()} + resume.send(()).unwrap(); + let result = work.await.unwrap(); + let error = result.record().err().unwrap(); + assert!(Arc::ptr_eq(&error, &result.record().err().unwrap())); + let Error::Facility { source, .. } = &*error else { + panic!("source error flattened") + }; + assert!( + source + .downcast_ref::() + .unwrap() + .to_string() + .contains("after durable commit") + ); + assert_eq!(stored(&fixture, record.digest().unwrap()).await, record); + let snapshot = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + let adopted = publish(&fixture, &capture, policy).await; + adopted.confirmed().unwrap(); + assert_eq!(adopted.record().unwrap(), &record); + assert_eq!( + fixture.native.journal.load_snapshot(scope()).await.unwrap(), + snapshot + ); + drop(capture); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn follower_policy_publication_cancelled_waiter_keeps_complete_durable_history() { + let (fixture, capture, policy) = setup().await; + let record = capture.durable_record(policy).unwrap(); + let verifier = verifier(&fixture); + let (entered, resume) = fixture + .native + .journal + .pause_next_follower_evacuation_reply(false); + let mut work = Box::pin(FleetFollowerEvacuationPublication::publish( + &capture, + policy, + fixture.native.journal.as_ref(), + &verifier, + deadline(), + clock, + )); + tokio::select! {result=&mut work=>panic!("unexpected completion {}",result.is_ok()),result=entered=>result.unwrap()} + drop(work); + let _ = resume.send(()); + let client = client(&fixture).await; + assert_eq!( + client + .load_follower_evacuation(scope(), record.digest().unwrap()) + .await + .unwrap() + .unwrap(), + record + ); + verifier + .recheck(&client, &record, deadline(), clock) + .await + .unwrap(); + client.close().await.unwrap(); + drop(capture); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn follower_policy_publication_policy_change_after_commit_prevents_final_confirmation() { + let (fixture, capture, policy) = setup().await; + let verifier = verifier(&fixture); + let (entered, resume) = fixture + .native + .journal + .pause_next_follower_evacuation_reply(false); + let mut work = Box::pin(FleetFollowerEvacuationPublication::publish( + &capture, + policy, + fixture.native.journal.as_ref(), + &verifier, + deadline(), + clock, + )); + tokio::select! {result=&mut work=>panic!("unexpected completion {}",result.is_ok()),result=entered=>result.unwrap()} + let snapshot = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + fixture + .native + .journal + .set_follower_replacement_policy( + &snapshot, + FollowerReplacementPolicy::new(scope(), 2, 1).unwrap(), + clock().unwrap(), + ) + .await + .unwrap(); + resume.send(()).unwrap(); + let result = work.await.unwrap(); + let record = result.record().unwrap(); + assert!(result.confirmed().is_err()); + assert_eq!(stored(&fixture, record.digest().unwrap()).await, *record); + let fresh = verifier + .refresh(fixture.native.journal.as_ref(), record, deadline(), clock) + .await + .unwrap(); + let updated = FleetFollowerEvacuationPublication::publish_refreshed( + &fresh, + fixture.native.journal.as_ref(), + &verifier, + deadline(), + clock, + ) + .await + .unwrap(); + updated.confirmed().unwrap(); + assert_eq!(updated.record().unwrap().retired(), record.retired()); + drop(capture); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn follower_policy_publication_policy_cas_race_has_one_registry_mutation() { + let (fixture, capture, policy) = setup().await; + let client = client(&fixture).await; + let snapshot = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + let next = FollowerReplacementPolicy::new(scope(), policy.revision() + 1, 1).unwrap(); + let (one, two) = tokio::join!( + fixture + .native + .journal + .set_follower_replacement_policy(&snapshot, next, clock().unwrap()), + client.set_follower_replacement_policy(&snapshot, next, clock().unwrap()) + ); + assert_ne!(one.is_ok(), two.is_ok()); + let after = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + assert_eq!( + after.registry().revision(), + snapshot.registry().revision() + 1 + ); + assert_eq!( + client.follower_replacement_policy(&after).await.unwrap(), + Some(next) + ); + assert!( + client + .set_follower_replacement_policy(&after, next, clock().unwrap()) + .await + .is_err() + ); + assert_eq!(client.load_snapshot(scope()).await.unwrap(), after); + client.close().await.unwrap(); + drop(capture); + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/follower_tests/persisted/tests.rs b/crates/cellule-host/minion/scenario/follower_tests/persisted/tests.rs new file mode 100644 index 00000000..a8f79f13 --- /dev/null +++ b/crates/cellule-host/minion/scenario/follower_tests/persisted/tests.rs @@ -0,0 +1,404 @@ +use super::*; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn follower_policy_publication_retains_full_ensembles_after_independent_reconstruction() { + let (fixture, capture, policy) = setup().await; + let result = publish(&fixture, &capture, policy).await; + let record = result.record().unwrap(); + assert_eq!(result.confirmed().unwrap().native().len(), 3); + assert_eq!(record.retired(), capture.retired_members()); + assert_eq!( + record.covered_through(), + capture.rotation().retirement().barrier().covered_through() + ); + assert_eq!( + record.original_key(), + capture.original().spec().key().unwrap() + ); + assert_eq!(record.interval(), capture.interval()); + assert_eq!(record.replacements().len(), 2); + let snapshot = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + assert_eq!( + snapshot.registry().revision(), + capture.snapshot().registry().revision() + 1 + ); + let replay = publish(&fixture, &capture, policy).await; + assert_eq!(replay.record().unwrap(), record); + replay.confirmed().unwrap(); + assert_eq!( + fixture.native.journal.load_snapshot(scope()).await.unwrap(), + snapshot + ); + let client = client(&fixture).await; + verifier(&fixture) + .recheck(&client, record, deadline(), clock) + .await + .unwrap(); + assert_eq!( + client + .load_follower_evacuation(scope(), record.digest().unwrap()) + .await + .unwrap() + .unwrap(), + *record + ); + client.close().await.unwrap(); + drop(capture); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn follower_policy_publication_refreshes_new_epoch_after_original_local_receipt_eviction() { + let (fixture, capture, policy) = setup_epochs(3).await; + let result = publish(&fixture, &capture, policy).await; + let record = result.record().unwrap(); + result.confirmed().unwrap(); + assert!( + !result + .confirmed() + .unwrap() + .authority() + .log() + .unwrap() + .active() + ); + let request = fixture.native.node.request_node_log_rotation(2).unwrap(); + until(|| request.observe().unwrap().phase() == cellule_host::NodeLogRotationPhase::Completed) + .await; + assert!( + fixture + .native + .node + .node_log_rotation_request(1) + .unwrap() + .is_none() + ); + assert!( + verifier(&fixture) + .recheck(fixture.native.journal.as_ref(), record, deadline(), clock) + .await + .is_err() + ); + let refreshed = verifier(&fixture) + .refresh(fixture.native.journal.as_ref(), record, deadline(), clock) + .await + .unwrap(); + assert_eq!(refreshed.record().replacement_epoch(), 3); + let updated = FleetFollowerEvacuationPublication::publish_refreshed( + &refreshed, + fixture.native.journal.as_ref(), + &verifier(&fixture), + deadline(), + clock, + ) + .await + .unwrap(); + updated.confirmed().unwrap(); + assert_eq!(updated.record().unwrap().retired(), record.retired()); + assert_eq!( + updated.record().unwrap().covered_through(), + record.covered_through() + ); + assert_eq!( + updated.record().unwrap().original_digest(), + record.original_digest() + ); + drop(capture); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn follower_policy_publication_refreshes_current_policy_without_repeating_retirement() { + let (fixture, capture, policy) = setup().await; + let result = publish(&fixture, &capture, policy).await; + let old = result.record().unwrap(); + result.confirmed().unwrap(); + let attempts = fixture.native.authority.attempts.load(Ordering::Acquire); + let original = capture.retired().clone(); + let snapshot = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + fixture + .native + .journal + .set_follower_replacement_policy( + &snapshot, + FollowerReplacementPolicy::new(scope(), 2, 1).unwrap(), + clock().unwrap(), + ) + .await + .unwrap(); + assert!( + verifier(&fixture) + .recheck(fixture.native.journal.as_ref(), old, deadline(), clock) + .await + .is_err() + ); + let refreshed = verifier(&fixture) + .refresh(fixture.native.journal.as_ref(), old, deadline(), clock) + .await + .unwrap(); + let updated = FleetFollowerEvacuationPublication::publish_refreshed( + &refreshed, + fixture.native.journal.as_ref(), + &verifier(&fixture), + deadline(), + clock, + ) + .await + .unwrap(); + updated.confirmed().unwrap(); + let new = updated.record().unwrap(); + assert_eq!(new.retired(), old.retired()); + assert_eq!(new.original_digest(), old.original_digest()); + assert_eq!(new.policy().revision(), 2); + assert_ne!(new.digest().unwrap(), old.digest().unwrap()); + assert_eq!(latest(&fixture, old).await, *new); + let snapshot = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + assert_eq!( + fixture + .native + .journal + .persist_follower_evacuation(capture.snapshot(), old, clock().unwrap()) + .await + .unwrap(), + *old + ); + assert_eq!( + fixture.native.journal.load_snapshot(scope()).await.unwrap(), + snapshot + ); + assert_eq!(latest(&fixture, old).await, *new); + assert!( + publish(&fixture, &capture, policy) + .await + .confirmed() + .is_err() + ); + assert_eq!(stored(&fixture, old.digest().unwrap()).await, *old); + assert_eq!(capture.retired(), &original); + assert_eq!( + fixture.native.authority.attempts.load(Ordering::Acquire), + attempts + ); + drop(capture); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn follower_policy_publication_withdrawn_receiver_cannot_revalidate() { + let (fixture, capture, policy) = setup().await; + let result = publish(&fixture, &capture, policy).await; + let record = result.record().unwrap(); + result.confirmed().unwrap(); + fixture.nodes[3].shutdown().await.unwrap(); + assert!( + verifier(&fixture) + .recheck(fixture.native.journal.as_ref(), record, deadline(), clock) + .await + .is_err() + ); + assert!( + verifier(&fixture) + .refresh(fixture.native.journal.as_ref(), record, deadline(), clock) + .await + .is_err() + ); + assert_eq!(stored(&fixture, record.digest().unwrap()).await, *record); + drop(capture); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn follower_policy_publication_closing_and_deadline_refresh_preserve_original_rows() { + use cellule_runtime::fleet::operations::{DrainEvidence, JournalTransition, MaintenanceEvent}; + let (fixture, capture, policy) = setup().await; + let result = publish(&fixture, &capture, policy).await; + let record = result.record().unwrap(); + result.confirmed().unwrap(); + let snapshot = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + let operation = snapshot.head().maintenance().unwrap(); + let after = fixture + .native + .journal + .compare_exchange( + &snapshot, + snapshot.head().controller().unwrap().epoch, + clock().unwrap(), + &JournalTransition::Maintenance(MaintenanceEvent::ReadyToClose(DrainEvidence { + node: operation.node(), + session: operation.session(), + remaining_cells: 0, + unresolved_attempts: 0, + relocated: true, + readers_settled: true, + followers_settled: true, + facilities_closed: false, + stopped: false, + withdrawn: false, + })), + ) + .await + .unwrap(); + verifier(&fixture) + .recheck(fixture.native.journal.as_ref(), record, deadline(), clock) + .await + .unwrap(); + let after = fixture + .native + .journal + .compare_exchange( + &after, + after.head().controller().unwrap().epoch, + clock().unwrap(), + &JournalTransition::Maintenance(MaintenanceEvent::ExtendDeadline( + record.operation().deadline_ms() + 10_000, + )), + ) + .await + .unwrap(); + assert!( + verifier(&fixture) + .recheck(fixture.native.journal.as_ref(), record, deadline(), clock) + .await + .is_err() + ); + let refreshed = verifier(&fixture) + .refresh(fixture.native.journal.as_ref(), record, deadline(), clock) + .await + .unwrap(); + assert_eq!( + refreshed.record().operation(), + after.head().maintenance().unwrap() + ); + let updated = FleetFollowerEvacuationPublication::publish_refreshed( + &refreshed, + fixture.native.journal.as_ref(), + &verifier(&fixture), + deadline(), + clock, + ) + .await + .unwrap(); + updated.confirmed().unwrap(); + assert_eq!(updated.record().unwrap().retired(), record.retired()); + drop(capture); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn follower_policy_publication_requires_current_barrier_stored_history_and_fresh_clock() { + let (fixture, capture, policy) = setup().await; + let record = capture.durable_record(policy).unwrap(); + assert!( + verifier(&fixture) + .refresh(fixture.native.journal.as_ref(), &record, deadline(), clock) + .await + .is_err() + ); + let snapshot = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + fixture + .native + .journal + .set_follower_replacement_policy( + &snapshot, + FollowerReplacementPolicy::new(scope(), 2, 2).unwrap(), + clock().unwrap(), + ) + .await + .unwrap(); + assert!( + FleetFollowerEvacuationPublication::publish( + &capture, + policy, + fixture.native.journal.as_ref(), + &verifier(&fixture), + deadline(), + clock + ) + .await + .is_err() + ); + assert!( + fixture + .native + .journal + .load_follower_evacuation(scope(), record.digest().unwrap()) + .await + .unwrap() + .is_none() + ); + let (candidate_policy, snapshot) = { + let snapshot = fixture.native.journal.load_snapshot(scope()).await.unwrap(); + ( + fixture + .native + .journal + .follower_replacement_policy(&snapshot) + .await + .unwrap() + .unwrap(), + snapshot, + ) + }; + let fresh = fixture + .native + .node + .follower_evacuation( + capture.original(), + snapshot.head().maintenance().unwrap(), + 2, + deadline(), + ) + .await + .unwrap(); + let result = publish(&fixture, &fresh, candidate_policy).await; + let record = result.record().unwrap(); + result.confirmed().unwrap(); + let now = clock().unwrap(); + let mut n = 0; + assert!( + verifier(&fixture) + .recheck(fixture.native.journal.as_ref(), record, deadline(), || { + n += 1; + Ok(now - n) + }) + .await + .is_err() + ); + assert!( + verifier(&fixture) + .recheck( + fixture.native.journal.as_ref(), + record, + Instant::now(), + clock + ) + .await + .is_err() + ); + drop((fresh, capture)); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn follower_policy_publication_closed_original_leader_requires_recovery_instead_of_refresh() { + let (fixture, capture, policy) = setup().await; + let result = publish(&fixture, &capture, policy).await; + let record = result.record().unwrap(); + result.confirmed().unwrap(); + fixture.native.node.shutdown().await.unwrap(); + assert!( + verifier(&fixture) + .recheck(fixture.native.journal.as_ref(), record, deadline(), clock) + .await + .is_err() + ); + assert!( + verifier(&fixture) + .refresh(fixture.native.journal.as_ref(), record, deadline(), clock) + .await + .is_err() + ); + assert_eq!(stored(&fixture, record.digest().unwrap()).await, *record); + drop(capture); + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/follower_tests/tests.rs b/crates/cellule-host/minion/scenario/follower_tests/tests.rs new file mode 100644 index 00000000..d08c751f --- /dev/null +++ b/crates/cellule-host/minion/scenario/follower_tests/tests.rs @@ -0,0 +1,543 @@ +use super::*; + +async fn refused(fixture: &Fixture) -> Vec { + tokio::time::timeout(Duration::from_secs(3), async { + loop { + let rows = fixture.rows().await; + if rows.len() == 2 + && rows + .iter() + .all(|row| row.status() == EnrollmentStatus::Refused) + { + return rows; + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + }) + .await + .unwrap() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn all_selected_members_are_pending_before_the_one_canonical_enrollment() { + let fixture = Fixture::new().await; + let (first, resume_first) = fixture.journal.pause_next_enrollment_reply(false, false); + fixture.install(); + captured(first).await; + let rows = fixture.rows().await; + assert_eq!(rows.len(), 1); + assert_eq!(rows[0].status(), EnrollmentStatus::Pending); + let (second, resume_second) = fixture.journal.pause_before_enrollment_acceptance(); + resume_first.send(()).unwrap(); + captured(second).await; + let completion = fixture + .node + .follower_enrollment_completion(1) + .unwrap() + .unwrap(); + assert_eq!(completion.members.len(), 2); + assert!(!completion.native_started); + assert!(completion.members[0].accepted.is_some()); + assert!(completion.members[1].accepted.is_none()); + assert!( + fixture + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .unwrap() + .advertisement() + .log() + .is_none() + ); + assert!(fixture.node.runtime().node_durability().is_none()); + resume_second.send(()).unwrap(); + fixture.installed().await; + let completion = fixture + .node + .follower_enrollment_completion(1) + .unwrap() + .unwrap(); + assert!(completion.native_started); + assert!(completion.enrollment.is_some()); + assert!(completion.members.iter().all(|member| member.published + && member.accepted.as_ref().unwrap().status() == EnrollmentStatus::Pending + && matches!(member.event, Some(EnrollmentEvent::Established(_))))); + assert_eq!( + completion.attempt.prepared().log().members(), + &[node_id(1), node_id(2)] + ); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn lost_acceptance_fences_every_original_member_without_native_dispatch() { + let fixture = Fixture::new().await; + let (accepted, resume) = fixture.journal.pause_next_enrollment_reply(false, true); + fixture.install(); + captured(accepted).await; + let original = fixture.rows().await.remove(0); + resume.send(()).unwrap(); + let rows = refused(&fixture).await; + let first = rows + .iter() + .find(|row| row.spec() == original.spec()) + .unwrap(); + assert_eq!(first.accepted_at_ms(), original.accepted_at_ms()); + for row in &rows { + assert!( + matches!(fixture.journal.accept_enrollment(row.spec(), clock().unwrap()).await.unwrap(), + cellule_host::fleet::FleetEnrollmentAcceptance::Existing(ref replay) if replay == row) + ); + } + assert!(fixture.node.runtime().node_durability().is_none()); + assert!( + fixture + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .unwrap() + .advertisement() + .log() + .is_none() + ); + assert!(fixture.transport.requests.lock().unwrap().is_empty()); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn partial_acceptance_and_receiver_cordon_refuse_the_complete_fixed_ensemble() { + use cellule_runtime::fleet::operations::{ + JournalTransition, MaintenanceOperation, OperationId, + }; + let fixture = Fixture::new().await; + let (accepted, resume) = fixture.journal.pause_next_enrollment_reply(false, false); + fixture.install(); + captured(accepted).await; + let completion = fixture + .node + .follower_enrollment_completion(1) + .unwrap() + .unwrap(); + let receiver = completion.members[1].spec.target; + let now = clock().unwrap(); + let snapshot = fixture.journal.load_snapshot(scope()).await.unwrap(); + let claimed = fixture + .journal + .claim_controller(scope(), snapshot.head().revision(), session(9), now) + .await + .unwrap(); + fixture + .journal + .compare_exchange( + &claimed, + claimed.head().controller().unwrap().epoch, + now, + &JournalTransition::BeginMaintenance( + MaintenanceOperation::new( + OperationId::from_bytes([203; 16]).unwrap(), + Digest::from_bytes([204; 32]), + receiver.node, + receiver.session, + receiver.intent_revision + 1, + now, + now + 60_000, + ) + .unwrap(), + ), + ) + .await + .unwrap(); + resume.send(()).unwrap(); + let rows = refused(&fixture).await; + assert!(rows.iter().all(|row| { + completion + .members + .iter() + .any(|member| &member.spec == row.spec()) + })); + assert!(fixture.node.runtime().node_durability().is_none()); + assert!( + fixture + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .unwrap() + .advertisement() + .log() + .is_none() + ); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn lost_establishment_replays_original_events_without_a_new_native_attempt() { + let fixture = Fixture::new().await; + let (published, resume) = fixture.journal.pause_next_enrollment_reply(true, true); + fixture.install(); + captured(published).await; + let original = fixture + .node + .follower_enrollment_completion(1) + .unwrap() + .unwrap(); + let generation = fixture + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .unwrap() + .advertisement() + .generation(); + assert!(original.native_started); + assert!(original.enrollment.is_some()); + assert!(fixture.node.runtime().node_durability().is_none()); + resume.send(()).unwrap(); + fixture.installed().await; + let replay = fixture + .node + .follower_enrollment_completion(1) + .unwrap() + .unwrap(); + assert!(replay.execution_error.is_none()); + assert!(replay.journal_error.is_some()); + for (a, b) in original.members.iter().zip(&replay.members) { + assert_eq!(a.spec, b.spec); + assert_eq!(a.accepted, b.accepted); + assert_eq!(a.event, b.event); + } + assert_eq!(fixture.provider.prepared.load(Ordering::Acquire), 1); + assert_eq!( + fixture + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .unwrap() + .advertisement() + .generation(), + generation + ); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn lost_close_and_retirement_publication_keep_original_fences_and_owned_drain() { + let fixture = Fixture::new().await; + fixture.install(); + fixture.installed().await; + fixture.authority.lose_reply.store(true, Ordering::Release); + let (published, resume) = fixture.journal.pause_next_enrollment_reply(true, true); + let node = fixture.node.clone(); + let waiter = tokio::spawn(async move { node.shutdown().await }); + captured(published).await; + let completion = fixture + .node + .follower_enrollment_completion(1) + .unwrap() + .unwrap(); + assert!(completion.native_closed); + assert!(completion.execution_error.is_some()); + let original = completion.retirement.unwrap(); + original.confirmed().unwrap(); + assert_eq!(original.members().len(), 2); + assert_eq!(fixture.authority.attempts.load(Ordering::Acquire), 2); + assert_eq!(fixture.transport.requests.lock().unwrap().len(), 2); + assert!( + fixture + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .unwrap() + .advertisement() + .log() + .is_none() + ); + assert_eq!(fixture.node.state(), NodeState::Draining); + assert!(fixture.node.stats().retained_bytes() > 0); + waiter.abort(); + assert!(waiter.await.unwrap_err().is_cancelled()); + resume.send(()).unwrap(); + fixture.node.shutdown().await.unwrap(); + assert_eq!(fixture.authority.attempts.load(Ordering::Acquire), 2); + assert_eq!(fixture.transport.requests.lock().unwrap().len(), 2); + let rows = fixture.rows().await; + assert!( + rows.iter() + .all(|row| row.status() == EnrollmentStatus::Retired) + ); + assert!(original.confirmed().is_ok()); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn failed_member_keeps_established_rows_until_a_later_native_fence() { + let fixture = Fixture::new().await; + fixture.install(); + fixture.installed().await; + fixture.transport.lose_retire.store(true, Ordering::Release); + assert!( + fixture + .node + .shutdown_until(std::time::Instant::now() + Duration::from_millis(80)) + .await + .is_err() + ); + assert_eq!(fixture.node.state(), NodeState::Draining); + // The deadline cancels the drain waiter, not the retained native owner. + // A busy runner may still be awaiting the original member reply at 80 ms. + // Await that actual failure before asserting its retained evidence. + let completion = tokio::time::timeout(Duration::from_secs(3), async { + loop { + let completion = fixture + .node + .follower_enrollment_completion(1) + .unwrap() + .unwrap(); + if completion.execution_error.is_some() && completion.retirement.is_some() { + return completion; + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + }) + .await + .unwrap(); + assert!(!completion.native_closed); + assert!(completion.execution_error.is_some()); + assert!(matches!( + fixture + .node + .fleet_follower_enrollments_page(None, 1, clock().unwrap()), + Err(Error::RuntimeClosed) + )); + assert!(completion.retirement.unwrap().confirmed().is_err()); + assert_eq!(fixture.authority.attempts.load(Ordering::Acquire), 0); + assert!( + fixture + .rows() + .await + .iter() + .all(|row| row.status() == EnrollmentStatus::Established) + ); + assert!(fixture.node.stats().retained_bytes() > 0); + fixture + .transport + .lose_retire + .store(false, Ordering::Release); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn deadline_during_acceptance_preserves_supervisor_join_then_refuses_without_cas() { + let fixture = Fixture::new().await; + let (accepted, resume) = fixture.journal.pause_next_enrollment_reply(false, false); + fixture.install(); + captured(accepted).await; + assert!( + fixture + .node + .shutdown_until(std::time::Instant::now() + Duration::from_millis(30)) + .await + .is_err() + ); + assert_eq!(fixture.node.state(), NodeState::Draining); + assert!( + !fixture + .node + .follower_enrollment_completion(1) + .unwrap() + .unwrap() + .native_started + ); + assert!(fixture.node.stats().retained_bytes() > 0); + resume.send(()).unwrap(); + fixture.node.shutdown().await.unwrap(); + refused(&fixture).await; + assert!(fixture.transport.requests.lock().unwrap().is_empty()); + assert!( + fixture + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .unwrap() + .advertisement() + .log() + .is_none() + ); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn invalid_shipper_limits_fail_before_pending_or_authority_mutation() { + let fixture = Fixture::new().await; + fixture.install_limits(Limits { + max_capture_bytes: u64::MAX, + ..Limits::default() + }); + until(|| { + fixture + .provider + .events + .lock() + .unwrap() + .contains(&NodeDurabilityRotation::Failed) + }) + .await; + assert!(fixture.rows().await.is_empty()); + assert!( + fixture + .node + .follower_enrollment_completion(1) + .unwrap() + .is_none() + ); + assert!(fixture.node.runtime().node_durability().is_none()); + assert!( + fixture + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .unwrap() + .advertisement() + .log() + .is_none() + ); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn managed_enrollment_preserves_acknowledged_command_and_exact_root_through_drain() { + let fixture = Fixture::new().await; + fixture.install(); + fixture.installed().await; + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + scope().application, + application::NAMESPACE, + b"managed-follower-command", + ) + .unwrap(); + let incarnation = IncarnationId::from_bytes([241; 16]); + let catalog = CellCatalog::new(fixture.layout.clone(), target.tenant()); + let proof = catalog + .provision( + CatalogEntry::new( + &target, + CatalogRole::Sql, + fixture.node.application().registry().module_digests()[0], + 1, + ) + .unwrap(), + ) + .await + .unwrap(); + let authority = CellAuthority::new(fixture.layout.clone()); + let initial = authority + .create_initial(&proof, incarnation, owner(0)) + .await + .unwrap(); + let replica = CellReplica::new( + fixture.layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + limits(), + ) + .unwrap(); + let handle = fixture + .node + .runtime() + .bootstrap( + proof, + replica.clone(), + authority.clone(), + initial, + fixture.root.path().join("command.sqlite"), + |tx| { + tx.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES (16)", + )?; + Ok(()) + }, + ) + .await + .unwrap(); + let now = clock().unwrap(); + let request = RequestId::from_bytes([245; 16]); + let response = handle + .execute( + MutationIdentity { + request_id: request, + issued_at_ms: now, + expires_at_ms: now + 60_000, + }, + Digest::from_bytes([245; 32]), + now, + 64, + 64, + |tx| { + tx.execute("UPDATE counter SET value = value + 1", [])?; + Ok(HandlerOutcome::Success(vec![17])) + }, + ) + .await + .unwrap(); + assert!(matches!(response, StoredOutcome::Success { ref result, .. } if result==&[17])); + let (published, resume) = fixture.journal.pause_next_enrollment_reply(true, false); + let node = fixture.node.clone(); + let drain = tokio::spawn(async move { node.shutdown().await }); + captured(published).await; + let observation = fixture + .node + .follower_enrollment_completion(1) + .unwrap() + .unwrap() + .retirement + .unwrap(); + assert!(observation.barrier().covered_through() > 0); + observation.confirmed().unwrap(); + for member in observation.members() { + assert_eq!( + member.result().unwrap().durable_through, + observation.barrier().covered_through() + ); + } + resume.send(()).unwrap(); + drain.await.unwrap().unwrap(); + let control = authority.load(target.cell_id()).await.unwrap().unwrap(); + let root = control.value().ltx_root().unwrap(); + let restored = fixture.root.path().join("restored.sqlite"); + assert_eq!( + replica + .open_root(&root) + .await + .unwrap() + .restore(&restored) + .await + .unwrap(), + root.position + ); + let database = rusqlite::Connection::open(restored).unwrap(); + assert_eq!( + database + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) + .unwrap(), + 17 + ); + assert_eq!( + database + .query_row( + "SELECT result FROM sys_requests WHERE request_id=?1", + [request.as_bytes().as_slice()], + |row| row.get::<_, Vec>(0) + ) + .unwrap(), + vec![17] + ); + drop(database); + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/mod.rs b/crates/cellule-host/minion/scenario/mod.rs new file mode 100644 index 00000000..9632400a --- /dev/null +++ b/crates/cellule-host/minion/scenario/mod.rs @@ -0,0 +1,771 @@ +//! Finite real-node movement fixture shared by the executable and its tests. +//! Pressure is measured from an owned disk admission reservation. This is an +//! admission-load scenario, not physical disk throughput or provider qualification. + +mod adapters; +mod application; +mod balance; +#[cfg(test)] +mod follower_tests; +mod observation; +#[cfg(test)] +mod reader_tests; +#[cfg(test)] +mod recovered_followers; +mod startup; +#[cfg(test)] +mod successor_tests; +#[cfg(test)] +mod tests; + +use super::journal::{JournalError, JournalResult, SqliteJournal}; +use cellule_host::fleet::{FleetEnrollmentJournal, FleetJournal, FleetReconciler}; +use cellule_host::{CellNode, CellNodeBuilder, NodeState}; +use cellule_runtime::{ + cell::{ + actor::{CellHandle, CellInventoryEntry}, + catalog::{CatalogEntry, CatalogProof, CatalogRole, CellCatalog}, + executor::{HandlerOutcome, MutationIdentity, Resolution, StoredOutcome}, + worker::SqlWorkerPool, + }, + control::{Owner, authority::CellAuthority}, + fleet::operations::{FleetProfile, FleetScope, NodeIntent}, + identity::{ + ApplicationId, CellId, CellTarget, Digest, IncarnationId, NodeId, RequestId, SessionId, + TenantId, + }, + ltx::{CellReplica, CellStorageLayout, DiskBudget, Host, Limits}, + node::{NodePressure, lease::NodeLeaseGuard}, +}; +use cellule_store::Store; +use object_store::{memory::InMemory, path::Path as ObjectPath}; +use std::{ + collections::HashMap, + path::PathBuf, + sync::Arc, + time::{SystemTime, UNIX_EPOCH}, +}; +use tokio::time::{Duration, Instant}; +use tokio_util::sync::CancellationToken; + +const CELL_COUNT: usize = 12; +fn scope() -> FleetScope { + FleetScope { + fleet: Digest::from_bytes([200; 32]), + application: ApplicationId::from_bytes([3; 16]), + } +} +fn node_id(n: usize) -> NodeId { + NodeId::from_bytes([n as u8 + 1; 16]) +} +fn session(n: usize) -> SessionId { + SessionId::from_bytes([n as u8 + 11; 16]) +} +fn owner(n: usize) -> Owner { + Owner { + session: session(n), + endpoint: format!("https://node-{n}.example:8789"), + } +} +fn clock() -> cellule_runtime::Result { + let duration = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_err(|source| cellule_runtime::Error::Facility { + name: "example-clock", + source: Box::new(source), + })?; + i64::try_from(duration.as_millis()).map_err(|source| cellule_runtime::Error::Facility { + name: "example-clock", + source: Box::new(source), + }) +} +fn invalid(message: &'static str) -> JournalError { + std::io::Error::other(message).into() +} + +struct Record { + target: CellTarget, + incarnation: IncarnationId, + catalog: CatalogProof, + replica: CellReplica, + authority: CellAuthority, +} +struct Acknowledged { + identity: MutationIdentity, + digest: Digest, + outcome: StoredOutcome, + value: i64, + source: CellHandle, +} + +struct SettlementContext<'a> { + records: &'a HashMap, + acknowledged: &'a HashMap, + profile: FleetProfile, + previous: Option<&'a FleetReconciler>, +} + +#[derive(Debug)] +pub(super) struct ScenarioSummary { + pub released: usize, + pub activated: usize, + pub retired: usize, + pub receipt_checks: usize, + pub max_inflight: usize, + pub max_restore_bytes: u64, + pub joined_nodes: usize, + pub boot_retirements: usize, + pub receiver_nodes: usize, + pub lost_release_replies: usize, + pub controller_epoch: u64, + pub expired_receiver_cleanups: usize, + pub blockers: Vec, + pub final_counts: [usize; 3], +} + +/// Owns the private directory until all runtime and journal jobs are joined. +pub(super) async fn overload() -> JournalResult { + execute(false, false).await +} + +pub(super) async fn controller_restart() -> JournalResult { + execute(true, false).await +} + +pub(super) async fn count_balance() -> JournalResult { + execute(false, true).await +} + +async fn execute(restart: bool, count_balance: bool) -> JournalResult { + let root = tempfile::tempdir()?; + let path = root.path().join("fleet-journal.sqlite"); + let profile = if restart { + FleetProfile { + controller_lease_ms: 3_000, + reconcile_interval_ms: 500, + ..FleetProfile::default() + } + } else { + FleetProfile::default() + }; + let journal = Arc::new(SqliteJournal::open(path.clone(), scope(), profile, clock()?).await?); + let mut nodes = Vec::new(); + let mut boots = Vec::new(); + let result = if count_balance { + balance::run(&root, journal.clone(), &mut nodes, &mut boots, profile).await + } else { + run( + &root, + path, + journal.clone(), + &mut nodes, + &mut boots, + profile, + restart, + ) + .await + }; + let mut cleanup_error = None; + for node in &nodes { + if let Err(error) = node.shutdown().await + && cleanup_error.is_none() + { + cleanup_error = Some(Box::new(error) as JournalError); + } + } + for boot in &boots { + if let Err(error) = boot.withdraw(&journal).await + && cleanup_error.is_none() + { + cleanup_error = Some(error); + } + } + if result.is_ok() && cleanup_error.is_none() { + let checked = async { + let version = journal.load_snapshot(scope()).await?.registry(); + let page = journal.enrollments_page(version, None, 128).await?; + if boots.len() != nodes.len() + || page.next().is_some() + || page.entries().len() != boots.len() + || page.entries().iter().any(|entry| { + entry.status() != cellule_runtime::fleet::operations::EnrollmentStatus::Retired + }) + { + return Err(invalid("example boot registry still has obligations")); + } + Ok::<_, JournalError>(()) + } + .await; + if let Err(error) = checked { + cleanup_error = Some(error); + } + } + if let Err(error) = journal.close().await + && cleanup_error.is_none() + { + cleanup_error = Some(error); + } + // Check all nodes even if an earlier shutdown failed; cleanup must never + // return before joining an independent sibling's accepted work. + for node in &nodes { + let stats = node.stats(); + if (node.state() != NodeState::Stopped + || stats.active_cells() != 0 + || stats.resident_bytes() != 0 + || stats.retained_bytes() != 0 + || stats.worker_jobs() != 0 + || stats.primitive_jobs() != 0 + || stats.hydration_jobs() != 0 + || stats.file_descriptors() != 0 + || stats.local_disk_reserved_bytes() != 0 + || stats.io_slots() != 0 + || stats.blocking_jobs() != 0 + || stats.recovery_jobs() != 0) + && cleanup_error.is_none() + { + cleanup_error = Some(invalid("example shutdown left runtime resources")); + } + } + let mut summary = match result { + Ok(summary) => summary, + Err(error) => { + if let Some(cleanup) = cleanup_error { + eprintln!("additional example cleanup failure: {cleanup:?}"); + } + return Err(error); + } + }; + if let Some(error) = cleanup_error { + return Err(error); + } + summary.joined_nodes = nodes.len(); + summary.boot_retirements = boots.len(); + Ok(summary) +} + +/// The private reference profile provisions only catalog-backed SQL writers. +/// Register every boot before readiness; retain partial owners for exit cleanup. +async fn initialize( + root: &tempfile::TempDir, + journal: &Arc, + nodes: &mut Vec>, + boots: &mut Vec, + receipt_lifetime_ms: i64, +) -> JournalResult<(Arc>, HashMap)> { + let application = application::compile()?; + let code = *application + .registry() + .module_digests() + .first() + .ok_or_else(|| invalid("example module absent"))?; + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + ObjectPath::from("fleet-example-cells"), + [3; 16], + ); + let authority = CellAuthority::new(layout.clone()); + let limits = Limits { + max_database_bytes: 64 << 20, + max_capture_bytes: 16 << 20, + ..Limits::default() + }; + let mut records = HashMap::new(); + for n in 0..CELL_COUNT { + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + scope().application, + application::NAMESPACE, + &[n as u8], + )?; + let incarnation = IncarnationId::from_bytes([n as u8 + 101; 16]); + let catalog = CellCatalog::new(layout.clone(), target.tenant()) + .provision(CatalogEntry::new(&target, CatalogRole::Sql, code, 1)?) + .await?; + authority + .create_initial(&catalog, incarnation, owner(0)) + .await?; + let replica = CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + limits, + )?; + records.insert( + target.cell_id(), + Record { + target, + incarnation, + catalog, + replica, + authority: authority.clone(), + }, + ); + } + let records = Arc::new(records); + let directory = cellule_runtime::node::NodeDirectory::new( + layout.clone(), + scope().fleet, + Digest::from_bytes([31; 32]), + application.registry().release_digest(), + ); + for index in 0..3 { + let intent = journal + .register_initial_intent(&NodeIntent::initial( + scope(), + node_id(index), + session(index), + )?) + .await?; + let node = Arc::new( + CellNodeBuilder::new(application.clone()) + .with_runtime( + SqlWorkerPool::new(2, 32)?.with_native_memory_limit(128 << 20)?, + 64 << 20, + ) + .with_replica_host(Host::default().with_local_disk_budget(DiskBudget::new(8 << 30))) + .with_session(session(index)) + .with_fleet_startup_intent(intent.clone()) + .build()?, + ); + // Register the runtime owner immediately so any later setup failure + // still joins it before dropping private paths or the journal. + nodes.push(node.clone()); + node.install_task_group(CancellationToken::new(), CancellationToken::new())?; + node.install_fleet_actions( + scope(), + node_id(index), + journal.clone(), + Arc::new(adapters::Cells { + records: records.clone(), + local: index, + root: root.path().into(), + }), + )?; + let ad = startup::advertisement(index, &node, &intent).await?; + let expires = ad.expires_at_ms(); + let spec = startup::spec(&intent)?; + let boot_index = boots.len(); + boots.push(startup::BootOwner { + node: node.clone(), + directory: directory.clone(), + spec: spec.clone(), + advertisement: ad.clone(), + guard: None, + }); + let boot = startup::enroll(journal, &directory, &spec, ad, clock()?).await?; + let guard = NodeLeaseGuard::new(clock()?, expires)?; + node.install_node_lease_for_startup(guard.clone())?; + boots[boot_index].guard = Some(guard); + node.confirm_fleet_startup(journal.as_ref(), boot.spec().key()?) + .await?; + let observed = directory + .load(session(index), clock()?) + .await? + .ok_or_else(|| invalid("example original boot is absent"))?; + node.install_fleet_boot_withdrawal(directory.clone(), observed, boot, journal.clone())?; + node.start()?; + } + let mut acknowledged = HashMap::new(); + // A reproducible, adversarial order makes the lowest Cell identities the + // oldest local eviction candidates. The fleet must use actual actor demand + // rather than succeed by chance on HashMap iteration order. + let mut initial_cells = records.iter().collect::>(); + initial_cells.sort_by_key(|(cell, _)| *cell.as_bytes()); + for (cell, record) in initial_cells { + let initial = record + .authority + .load(*cell) + .await? + .ok_or_else(|| invalid("example initial authority absent"))?; + let handle = nodes[0] + .runtime() + .bootstrap( + record.catalog.clone(), + record.replica.clone(), + record.authority.clone(), + initial, + root.path().join(format!("source-{cell:?}.sqlite")), + |tx| { + tx.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES (0)", + )?; + Ok(()) + }, + ) + .await?; + let now = clock()?; + let identity = MutationIdentity { + request_id: RequestId::from_bytes(*record.incarnation.as_bytes()), + issued_at_ms: now, + expires_at_ms: now + .checked_add(receipt_lifetime_ms) + .ok_or_else(|| invalid("example receipt lifetime overflow"))?, + }; + let digest = Digest::from_bytes([record.incarnation.as_bytes()[0]; 32]); + let value = i64::from(record.incarnation.as_bytes()[0]); + let outcome = handle + .execute(identity, digest, now, 64, 64, move |tx| { + tx.execute("UPDATE counter SET value = ?1", [value])?; + Ok(HandlerOutcome::Success(value.to_be_bytes().to_vec())) + }) + .await?; + acknowledged.insert( + *cell, + Acknowledged { + identity, + digest, + outcome, + value, + source: handle, + }, + ); + } + tokio::time::timeout(Duration::from_secs(10), async { + loop { + let page = nodes[0].runtime().fleet_cells_page(None,128).await?; + if page.entries().len() == CELL_COUNT && page.entries().iter().all(|entry| matches!(entry, CellInventoryEntry::Owned(row) if row.cost.is_some() && row.stable_observations == 2 && row.blockers.is_empty())) { return Ok::<_, JournalError>(()); } + drop(page); tokio::time::sleep(Duration::from_millis(10)).await; + } + }).await??; + let version = journal.load_snapshot(scope()).await?.registry(); + let version = journal.bootstrap_registry(version).await?; + journal.set_scheduling(version, true).await?; + Ok((records, acknowledged)) +} + +async fn run( + root: &tempfile::TempDir, + path: PathBuf, + journal: Arc, + nodes: &mut Vec>, + boots: &mut Vec, + profile: FleetProfile, + restart: bool, +) -> JournalResult { + let (records, acknowledged) = initialize(root, &journal, nodes, boots, 60_000).await?; + let fleet = Arc::new(adapters::LocalFleet { + nodes: nodes.clone(), + journal: journal.clone(), + boots: boots.clone(), + records: records.clone(), + capture_sequence: std::sync::atomic::AtomicU64::new(0), + lose_release_replies: restart, + lost_release_replies: std::sync::atomic::AtomicUsize::new(0), + expired_receiver_cleanups: std::sync::atomic::AtomicUsize::new(0), + }); + let driver = FleetReconciler::new( + scope(), + SessionId::from_bytes([206; 16]), + profile, + journal.clone(), + fleet.clone(), + fleet.clone(), + )?; + // Hold a real admission token; the actor's own 250-ms ledger samples and + // 1000-ms dwell classify this load. No sample time or tier is fabricated. + let pressure = nodes[0] + .runtime() + .local_disk_budget() + .try_reserve(7 << 30)?; + tokio::time::timeout(Duration::from_secs(10), async { + loop { + if nodes[0] + .runtime() + .operational_sample()? + .is_some_and(|sample| sample.pressure == NodePressure::Shedding) + { + return Ok::<_, JournalError>(()); + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + }) + .await??; + let first = driver + .reconcile_once(clock, Instant::now() + Duration::from_secs(5)) + .await?; + drop(pressure); + if first.allocated != 2 { + return Err(std::io::Error::other(format!( + "measured overload did not allocate the bounded two-move batch: report={first:?} current_source_sample={:?}", + nodes[0].runtime().operational_sample()? + )).into()); + } + let specs = first + .snapshot + .head() + .attempts() + .iter() + .map(|attempt| attempt.spec().clone()) + .collect::>(); + journal + .set_scheduling(first.snapshot.registry(), false) + .await?; + let prepared = driver + .reconcile_once(clock, Instant::now() + Duration::from_secs(5)) + .await?; + if prepared.dispatched != 2 || !prepared.failures.is_empty() { + return Err(invalid("real receivers did not prepare")); + } + if restart { + let lost = driver + .reconcile_once(clock, Instant::now() + Duration::from_secs(5)) + .await?; + if lost.dispatched != 2 + || lost.failures.len() != 2 + || lost.released != 0 + || fleet + .lost_release_replies + .load(std::sync::atomic::Ordering::SeqCst) + != 2 + || lost.snapshot.head().attempts().len() != 2 + { + return Err(std::io::Error::other(format!( + "reply loss did not retain both unconfirmed releases: {lost:?}; lost replies={}", + fleet + .lost_release_replies + .load(std::sync::atomic::Ordering::SeqCst) + )) + .into()); + } + let expires = lost + .snapshot + .head() + .controller() + .ok_or_else(|| invalid("controller lease absent"))? + .expires_at_ms; + // Wait on real time. Neither the node lease nor reservation evidence is + // restamped; the successor must settle expired prepared credit. + tokio::time::timeout(Duration::from_secs(5), async { + while clock()? < expires { + tokio::time::sleep(Duration::from_millis(10)).await; + } + Ok::<_, JournalError>(()) + }) + .await??; + } + // Reopen an independent controller client over the same durable file while + // node effects retain their original journal client and accepted envelopes. + let reopened = Arc::new(SqliteJournal::open(path, scope(), profile, clock()?).await?); + let mut blockers = first.blockers; + for blocker in prepared.blockers { + if !blockers.contains(&blocker) { + blockers.push(blocker); + } + } + let result = settle( + reopened.clone(), + fleet, + specs, + blockers, + SettlementContext { + records: &records, + acknowledged: &acknowledged, + profile, + previous: restart.then_some(&driver), + }, + ) + .await; + let closed = reopened.close().await; + let summary = result?; + closed?; + Ok(summary) +} + +async fn settle( + journal: Arc, + fleet: Arc, + specs: Vec, + blockers: Vec, + context: SettlementContext<'_>, +) -> JournalResult { + let SettlementContext { + records, + acknowledged, + profile, + previous, + } = context; + let driver = FleetReconciler::new( + scope(), + SessionId::from_bytes([if previous.is_some() { 207 } else { 206 }; 16]), + profile, + journal.clone(), + fleet.clone(), + fleet.clone(), + )?; + let mut summary = ScenarioSummary { + released: 0, + activated: 0, + retired: 0, + receipt_checks: 0, + max_inflight: specs.len(), + max_restore_bytes: specs.iter().map(|spec| spec.cost.disk_bytes).sum(), + joined_nodes: 0, + boot_retirements: 0, + receiver_nodes: specs + .iter() + .map(|spec| spec.destination) + .collect::>() + .len(), + lost_release_replies: 0, + controller_epoch: 0, + expired_receiver_cleanups: 0, + blockers, + final_counts: [0; 3], + }; + let mut passes = Vec::new(); + for pass in 0..12 { + let report = driver + .reconcile_once(clock, Instant::now() + Duration::from_secs(5)) + .await?; + if !report.failures.is_empty() { + return Err(invalid("real movement endpoint failed")); + } + summary.controller_epoch = report + .snapshot + .head() + .controller() + .ok_or_else(|| invalid("controller lease absent"))? + .epoch; + if pass == 0 + && let Some(previous) = previous + { + let before = journal.load_snapshot(scope()).await?; + let error = previous + .reconcile_once(clock, Instant::now() + Duration::from_secs(1)) + .await + .err() + .ok_or_else(|| invalid("old controller renewed successor lease"))?; + if !matches!(&error, cellule_runtime::Error::Facility { name: "fleet-journal", source } if matches!(source.downcast_ref::(), Some(cellule_runtime::fleet::operations::OperationError::Fenced))) + { + return Err(error.into()); + } + if summary.controller_epoch != 2 || journal.load_snapshot(scope()).await? != before { + return Err(invalid("old controller was not fenced after replacement")); + } + } + passes.push(format!("pass={pass} dispatched={} inspected={} released={} activated={} retired={} cancelled={} attempts={:?}", report.dispatched, report.inspected, report.released, report.activated, report.retired, report.cancelled, report.snapshot.head().attempts())); + for blocker in report.blockers { + if !summary.blockers.contains(&blocker) { + summary.blockers.push(blocker); + } + } + summary.released += report.released; + summary.activated += report.activated; + summary.retired += report.retired; + summary.max_inflight = summary + .max_inflight + .max(report.snapshot.head().attempts().len()); + summary.max_restore_bytes = summary + .max_restore_bytes + .max(report.snapshot.head().reserved_restore_bytes()); + if report.snapshot.head().attempts().is_empty() { + break; + } + } + summary.lost_release_replies = fleet + .lost_release_replies + .load(std::sync::atomic::Ordering::SeqCst); + summary.expired_receiver_cleanups = fleet + .expired_receiver_cleanups + .load(std::sync::atomic::Ordering::SeqCst); + if previous.is_some() + && (summary.lost_release_replies != 2 || summary.expired_receiver_cleanups != 2) + { + return Err(invalid( + "controller restart did not settle both original reservations", + )); + } + if summary.receiver_nodes != 2 + || summary.released != 2 + || summary.activated != 2 + || summary.retired != 2 + || !journal + .load_snapshot(scope()) + .await? + .head() + .attempts() + .is_empty() + { + let retained = journal.load_snapshot(scope()).await?; + return Err(std::io::Error::other(format!("real movement did not settle both attempts: summary={summary:?} retained={:?} passes={passes:?}", retained.head().attempts())).into()); + } + for spec in &specs { + verify_movement(&fleet, records, acknowledged, spec).await?; + summary.receipt_checks += 1; + } + Ok(summary) +} + +async fn verify_movement( + fleet: &adapters::LocalFleet, + records: &HashMap, + acknowledged: &HashMap, + spec: &cellule_runtime::fleet::operations::MoveAttemptSpec, +) -> JournalResult<()> { + let record = records + .get(&spec.target.cell_id()) + .ok_or_else(|| invalid("missing readback record"))?; + let receipt = acknowledged + .get(&spec.target.cell_id()) + .ok_or_else(|| invalid("missing original acknowledgment"))?; + let node = fleet + .nodes + .iter() + .enumerate() + .find(|(index, _)| session(*index) == spec.destination) + .map(|(_, node)| node) + .ok_or_else(|| invalid("missing receiver"))?; + let current = record + .authority + .load(spec.target.cell_id()) + .await? + .ok_or_else(|| invalid("receiver authority absent"))?; + // Restored serving actors may still be hydrating. This lookup reads the + // existing actor against authority; it cannot create another writer. + let handle = node + .runtime() + .local_handle(record.catalog.clone(), ¤t) + .await? + .ok_or_else(|| invalid("receiver has no serving actor"))?; + if handle + .resolve(receipt.identity, receipt.digest, clock()?, 64) + .await? + != Resolution::Committed(receipt.outcome.clone()) + { + return Err(invalid("original command receipt did not survive movement")); + } + let value = handle + .query(64, 64, |db| { + Ok(db + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))? + .to_be_bytes() + .to_vec()) + }) + .await?; + if value != receipt.value.to_be_bytes() { + return Err(invalid("restored Cell readback differs")); + } + let current = record + .authority + .load(spec.target.cell_id()) + .await? + .ok_or_else(|| invalid("receiver authority absent"))?; + if current + .value() + .owner + .as_ref() + .is_none_or(|owner| owner.session != spec.destination) + || current.value().epoch <= spec.source_epoch + { + return Err(invalid("receiver lacks successor authority")); + } + if !matches!( + receipt.source.query(64, 64, |_| Ok(Vec::new())).await, + Err(cellule_runtime::Error::Fenced + | cellule_runtime::Error::CellDraining + | cellule_runtime::Error::CellNotActive) + ) { + return Err(invalid("old source handle still served after movement")); + } + Ok(()) +} diff --git a/crates/cellule-host/minion/scenario/observation/mod.rs b/crates/cellule-host/minion/scenario/observation/mod.rs new file mode 100644 index 00000000..bc1e76df --- /dev/null +++ b/crates/cellule-host/minion/scenario/observation/mod.rs @@ -0,0 +1,400 @@ +//! Full observation of this example's closed, writer-only construction profile. +//! Role-enabled applications need their producer/native/policy collectors. + +use super::*; +use cellule_host::fleet::{ + FleetFollowerReferences, FleetNodeInventory, FleetNodeInventoryScan, FleetNodeSnapshot, + FleetObservation, FleetOwnedCell, FleetRoleCoverage, FleetRoster, FleetSnapshotNativePage, + FleetSnapshotRequest, FleetSnapshotSubject, +}; +use cellule_runtime::control::{Control, ControlState}; +use cellule_runtime::fleet::operations::{EnrollmentRole, EnrollmentStatus, PublishedPosition}; +use cellule_runtime::node::NodeAdvertisement; +use std::collections::HashSet; +use std::sync::atomic::Ordering; + +struct Capture { + started: i64, + finished: i64, + complete: bool, + nodes: Vec, + cells: Vec, + role_coverage: Option, +} + +pub(super) async fn complete_counts( + fleet: &adapters::LocalFleet, + roster: &FleetRoster, + deadline: Instant, +) -> JournalResult> { + let capture = collect(fleet, roster, deadline).await?; + if !capture.complete { + return Ok(None); + } + let mut counts = [0; 3]; + for owned in capture.cells { + let index = (0..3) + .find(|n| node_id(*n) == owned.node && session(*n) == owned.session) + .ok_or_else(|| invalid("example count endpoint differs"))?; + counts[index] += 1; + } + Ok(Some(counts)) +} + +pub(super) async fn observe( + fleet: &adapters::LocalFleet, + roster: &FleetRoster, + deadline: Instant, +) -> JournalResult { + let capture = collect(fleet, roster, deadline).await?; + let observation = FleetObservation::new( + scope(), + roster.snapshot().registry(), + roster.snapshot().registry().revision(), + capture.started, + capture.finished, + capture.complete, + capture.nodes, + capture.cells, + )?; + Ok(match capture.role_coverage { + Some(coverage) => observation.with_role_coverage(coverage)?, + None => observation, + }) +} + +async fn page( + fleet: &adapters::LocalFleet, + roster: &FleetRoster, + index: usize, + subject: FleetSnapshotSubject, + deadline: Instant, +) -> JournalResult> { + let issued = clock()?; + let remaining = deadline.saturating_duration_since(Instant::now()); + let interval = i64::try_from(remaining.as_millis())?.min(30_000); + let lease = roster + .snapshot() + .head() + .controller() + .ok_or_else(|| invalid("example snapshot has no controller"))?; + let expires = issued + .checked_add(interval) + .ok_or_else(|| invalid("example snapshot time overflow"))? + .min(lease.expires_at_ms); + let sequence = fleet + .capture_sequence + .try_update(Ordering::SeqCst, Ordering::SeqCst, |old| old.checked_add(1)) + .map_err(|_| invalid("example capture nonce exhausted"))?; + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.example-native-capture.v1\0"); + hash.update(&sequence.to_be_bytes()); + hash.update(session(index).as_bytes()); + let request = FleetSnapshotRequest::new( + roster.snapshot().clone(), + Digest::from_bytes(*hash.finalize().as_bytes()), + node_id(index), + session(index), + subject, + 32, + issued, + expires, + )?; + let node = fleet + .nodes + .get(index) + .ok_or_else(|| invalid("example snapshot node missing"))?; + let response = node + .fleet_snapshot(request.clone()) + .await + .map_err(|source| Box::new(source) as JournalError)?; + response.validate(&request, clock()?)?; + Ok(response) +} + +async fn collect( + fleet: &adapters::LocalFleet, + roster: &FleetRoster, + deadline: Instant, +) -> JournalResult { + let started = clock()?; + if fleet.nodes.len() != 3 || fleet.boots.len() != 3 || fleet.records.len() != CELL_COUNT { + return Err(invalid("example construction profile differs")); + } + let directory = &fleet.boots[0].directory; + let expected_sessions = (0..3).map(session).collect::>(); + let mut advertised = directory.advertised_sessions(started, 128).await?; + advertised.sort_by_key(|boot| *boot.as_bytes()); + let mut complete = advertised == expected_sessions + && roster.enrollments().iter().all(|record| { + matches!(record.spec().role, EnrollmentRole::Node { .. }) + && record.status() != EnrollmentStatus::Pending + }); + let mut nodes = Vec::new(); + let mut cells = Vec::new(); + let mut seen = HashSet::new(); + let mut inventories: Vec> = Vec::new(); + let mut references = Vec::new(); + for index in 0..3 { + let mut scan = FleetNodeInventoryScan::new(roster, node_id(index), session(index))?; + let mut stable = true; + while let Some(subject) = scan.next_subject()? { + let response = page(fleet, roster, index, subject, deadline).await?; + match scan.accept(response.request(), &response, clock()?) { + Ok(()) => {} + Err(cellule_runtime::Error::Node("native inventory category changed")) => { + stable = false; + break; + } + Err(source) => return Err(source.into()), + } + } + // Movement can change topology during the scan. Keep accepted rows for + // independent authority/actor revalidation, with completeness disabled. + complete &= stable; + for cell in scan + .cells() + .iter() + .map(|row| row.observation.target.cell_id()) + .chain(scan.transitioning_cells().iter().copied()) + { + if !fleet.records.contains_key(&cell) { + return Err(invalid("example unregistered native Cell")); + } + // A release/activation can appear on both sides of this interval. + // The fresh exact authority scan below retains only its actual owner. + complete &= seen.insert(cell); + } + cells.extend(scan.cells().iter().cloned()); + let inventory = if stable { + let inventory = scan.finish()?; + let bindings = inventory.bindings(); + // Absence requires this bootstrapped closed writer composition, + // native traversal and unexpected directory discovery together. + complete &= bindings.managed_startup + && !bindings.readers + && !bindings.follower_store + && !bindings.follower_producer + && !bindings.durability_supervisor + && inventory.node_log().is_none() + && inventory.transitioning_cells().is_empty() + && inventory.validate_enrollments(roster).is_ok(); + Some(inventory) + } else { + None + }; + inventories.push(inventory); + // Include expired and fenced leader obligations; live discovery alone + // could hide a follower role left by a failed boot. + let logs = FleetFollowerReferences::collect( + directory, + roster, + node_id(index), + 128, + deadline, + clock, + ) + .await?; + complete &= logs.entries().is_empty() && logs.validate_enrollments(roster).is_ok(); + references.push(logs); + nodes.push( + fleet.boots[index] + .refresh_capacity(index, fleet.journal.as_ref(), deadline) + .await?, + ); + } + let mut authority = HashMap::new(); + for (cell, record) in fleet.records.iter() { + let current = record + .authority + .load(*cell) + .await? + .ok_or_else(|| invalid("example current Cell authority missing"))?; + authority.insert(*cell, current.value().clone()); + } + cells.retain(|owned| { + let matches = authority + .get(&owned.observation.target.cell_id()) + .is_some_and(|current| matches_authority(owned, current)); + complete &= matches; + matches + }); + for (cell, current) in &authority { + complete &= match current.state { + ControlState::Serving => cells + .iter() + .any(|row| row.observation.target.cell_id() == *cell), + ControlState::Idle | ControlState::Tombstoned => !seen.contains(cell), + ControlState::Recovering => false, + }; + } + // Recheck exact authority after the full scan. A concurrent publication or + // takeover invalidates that Cell's planning row, not just count completeness. + complete &= recheck_authority(fleet, &authority, &mut cells).await?; + // Recheck every role category after *all* authority and membership reads. + // Stable local-only traversals cannot supply this fleet-wide interval. + for (index, inventory) in inventories.iter_mut().enumerate() { + let Some(inventory) = inventory else { + let response = page( + fleet, + roster, + index, + FleetSnapshotSubject::Cells(None), + deadline, + ) + .await?; + let FleetSnapshotNativePage::Cells(actors) = response.page() else { + return Err(invalid("example repeated actor page category differs")); + }; + retain_unchanged_writers(&mut cells, index, actors.entries()); + continue; + }; + let mut recheck = inventory.recheck(); + while let Some(subject) = recheck.next_subject()? { + let response = page(fleet, roster, index, subject.clone(), deadline).await?; + if recheck + .accept(response.request(), &response, clock()?) + .is_err() + { + complete = false; + match response.page() { + FleetSnapshotNativePage::Cells(actors) => { + // Count planning stops on changed topology. Keep only + // independently unchanged writer rows for pressure relief. + retain_unchanged_writers(&mut cells, index, actors.entries()); + } + FleetSnapshotNativePage::Host => { + cells.retain(|owned| owned.node != node_id(index)); + } + _ => {} + } + break; + } + } + complete &= recheck.finish().is_ok(); + } + for logs in &mut references { + match logs.recheck(directory, roster, 128, deadline, clock).await { + Ok(()) => {} + Err(cellule_runtime::Error::Node("authoritative follower inventory changed")) => { + complete = false; + } + Err(source) => return Err(source.into()), + } + } + let mut after = directory.advertised_sessions(clock()?, 128).await?; + after.sort_by_key(|boot| *boot.as_bytes()); + complete &= after == advertised && roster.covers_advertisements(&nodes, clock()?)?; + let role_coverage = if complete { + let native = inventories + .iter() + .filter_map(Option::as_ref) + .collect::>(); + let foreign = references.iter().collect::>(); + match FleetRoleCoverage::check(roster, &native, &foreign, clock()?) { + Ok(coverage) => Some(coverage), + Err(_) => { + complete = false; + None + } + } + } else { + None + }; + roster.confirm(fleet.journal.as_ref(), deadline).await?; + Ok(Capture { + started, + finished: clock()?, + complete, + nodes, + cells, + role_coverage, + }) +} + +async fn recheck_authority( + fleet: &adapters::LocalFleet, + authority: &HashMap, + cells: &mut Vec, +) -> JournalResult { + let mut unchanged = true; + for (cell, original) in authority { + let current = fleet + .records + .get(cell) + .ok_or_else(|| invalid("example authority input disappeared"))? + .authority + .load(*cell) + .await? + .ok_or_else(|| invalid("example authority disappeared during capture"))?; + if current.value() != original { + let value = current.value(); + let protected_same = value.cell == original.cell + && value.incarnation == original.incarnation + && value.epoch == original.epoch + && value.state == original.state + && value.owner == original.owner + && value.root == original.root + && value.recovery == original.recovery + && value.code == original.code + && value.schema == original.schema + && value.next_due_ms == original.next_due_ms; + eprintln!( + "FLEET_CAPTURE authority_changed cell={cell:?} revision={}..{} progress={}..{} protected_fields_unchanged={protected_same}", + original.revision, value.revision, original.progress, value.progress, + ); + unchanged = false; + cells.retain(|row| row.observation.target.cell_id() != *cell); + } + } + Ok(unchanged) +} + +fn retain_unchanged_writers( + cells: &mut Vec, + index: usize, + entries: &[CellInventoryEntry], +) { + cells.retain(|owned| { + owned.node != node_id(index) + || entries.iter().any(|entry| { + let CellInventoryEntry::Owned(current) = entry else { + return false; + }; + let original = &owned.observation; + current.target == original.target + && current.generation == original.generation + && current.incarnation == original.incarnation + && current.code == original.code + && current.schema == original.schema + && current.position == original.position + && current.cost == original.cost + && current.blockers == original.blockers + }) + }); +} + +fn matches_authority(owned: &FleetOwnedCell, current: &Control) -> bool { + let row = &owned.observation; + current.cell == row.target.cell_id() + && current.incarnation == row.incarnation + && current.code == row.code + && current.schema == row.schema + && current.state == ControlState::Serving + && current.owner.as_ref().is_some_and(|owner| { + owner.session == owned.session + && (0..3).any(|n| node_id(n) == owned.node && owner == &super::owner(n)) + }) + && current.recovery.is_none() + && current.root.as_ref().is_some_and(|root| { + row.position + == Some(PublishedPosition { + incarnation: current.incarnation, + epoch: current.epoch, + root: root.clone(), + }) + }) +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-host/minion/scenario/observation/tests/inventory.rs b/crates/cellule-host/minion/scenario/observation/tests/inventory.rs new file mode 100644 index 00000000..42ecbe44 --- /dev/null +++ b/crates/cellule-host/minion/scenario/observation/tests/inventory.rs @@ -0,0 +1,292 @@ +//! Public collector over real managed boots, actors and request-bound pages. +use super::*; + +async fn capture_node(fixture: &Fixture, roster: &FleetRoster, index: usize) -> FleetNodeInventory { + let mut scan = FleetNodeInventoryScan::new(roster, node_id(index), session(index)).unwrap(); + while let Some(subject) = scan.next_subject().unwrap() { + let response = page( + &fixture.fleet, + roster, + index, + subject, + Instant::now() + Duration::from_secs(5), + ) + .await + .unwrap(); + // Force original native continuations; the routine profile uses 32 rows. + let old = response.request(); + let request = FleetSnapshotRequest::new( + old.expected().clone(), + old.nonce(), + old.node(), + old.session(), + old.subject().clone(), + 1, + old.issued_at_ms(), + old.deadline_ms(), + ) + .unwrap(); + drop(response); + let response = fixture.fleet.nodes[index] + .fleet_snapshot(request.clone()) + .await + .unwrap(); + scan.accept(&request, &response, clock().unwrap()).unwrap(); + } + scan.finish().unwrap() +} + +#[tokio::test] +async fn full_native_traversal_uses_continuations_and_global_recheck() { + let fixture = Fixture::new().await; + let roster = fixture.roster().await; + let mut inventory = capture_node(&fixture, &roster, 0).await; + assert_eq!(inventory.cells().len(), CELL_COUNT); + assert!(inventory.transitioning_cells().is_empty()); + assert!(!inventory.bindings().readers); + assert_eq!(inventory.readers_closed(), None); + assert!(inventory.reader_jobs().is_none()); + assert_eq!(inventory.follower_store_state(), None); + assert_eq!(inventory.follower_producer_state(), None); + inventory.validate_enrollments(&roster).unwrap(); + let original = inventory.interval(); + let mut check = inventory.recheck(); + let mut pages = 0; + while let Some(subject) = check.next_subject().unwrap() { + let response = page( + &fixture.fleet, + &roster, + 0, + subject, + Instant::now() + Duration::from_secs(5), + ) + .await + .unwrap(); + check + .accept(response.request(), &response, clock().unwrap()) + .unwrap(); + pages += 1; + } + assert_eq!(pages, 7); + let interval = check.finish().unwrap(); + assert_eq!(interval.0, original.0); + assert!(interval.1 >= original.1); + drop(inventory); + fixture.close().await; +} + +#[tokio::test] +async fn duplicated_nonce_poisoning_and_partial_traversals_cannot_confirm() { + let fixture = Fixture::new().await; + let roster = fixture.roster().await; + let mut scan = FleetNodeInventoryScan::new(&roster, node_id(0), session(0)).unwrap(); + let host = page( + &fixture.fleet, + &roster, + 0, + FleetSnapshotSubject::Host, + Instant::now() + Duration::from_secs(5), + ) + .await + .unwrap(); + scan.accept(host.request(), &host, clock().unwrap()) + .unwrap(); + let original = host.request(); + let repeated = FleetSnapshotRequest::new( + original.expected().clone(), + original.nonce(), + original.node(), + original.session(), + FleetSnapshotSubject::Cells(None), + 1, + original.issued_at_ms(), + original.deadline_ms(), + ) + .unwrap(); + let repeated_page = fixture.fleet.nodes[0] + .fleet_snapshot(repeated.clone()) + .await + .unwrap(); + assert!( + scan.accept(&repeated, &repeated_page, clock().unwrap()) + .is_err() + ); + drop(repeated_page); + assert!(scan.next_subject().is_err()); + assert!(scan.finish().is_err()); + let mut inventory = capture_node(&fixture, &roster, 0).await; + assert!(inventory.recheck().finish().is_err()); + let mut check = inventory.recheck(); + assert!( + check + .accept(host.request(), &host, clock().unwrap()) + .is_err() + ); + assert!(check.finish().is_err()); + drop(host); + drop(inventory); + fixture.close().await; +} + +#[tokio::test] +async fn changed_complete_category_after_remote_scan_refuses_recheck() { + let fixture = Fixture::new().await; + let roster = fixture.roster().await; + let mut inventory = capture_node(&fixture, &roster, 0).await; + // A real actor admission changes topology without changing the roster. + tokio::time::timeout(Duration::from_secs(5), async { + loop { + if fixture.fleet.nodes[0] + .runtime() + .evict_idle(1) + .await + .unwrap() + == 1 + { + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + let mut check = inventory.recheck(); + while let Some(subject) = check.next_subject().unwrap() { + let response = page( + &fixture.fleet, + &roster, + 0, + subject.clone(), + Instant::now() + Duration::from_secs(5), + ) + .await + .unwrap(); + let accepted = check.accept(response.request(), &response, clock().unwrap()); + if matches!(subject, FleetSnapshotSubject::Cells(_)) { + assert!(accepted.is_err()); + break; + } + accepted.unwrap(); + } + assert!(check.finish().is_err()); + drop(inventory); + fixture.close().await; +} + +#[tokio::test] +async fn topology_change_retains_partial_rows_for_independent_pressure_checks() { + let fixture = Fixture::new().await; + let roster = fixture.roster().await; + let mut scan = FleetNodeInventoryScan::new(&roster, node_id(0), session(0)).unwrap(); + while let Some(subject) = scan.next_subject().unwrap() { + if matches!(subject, FleetSnapshotSubject::Cells(None)) && !scan.cells().is_empty() { + tokio::time::timeout(Duration::from_secs(5), async { + loop { + if fixture.fleet.nodes[0] + .runtime() + .evict_idle(1) + .await + .unwrap() + == 1 + { + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + let response = page( + &fixture.fleet, + &roster, + 0, + subject, + Instant::now() + Duration::from_secs(5), + ) + .await + .unwrap(); + assert!(matches!( + scan.accept(response.request(), &response, clock().unwrap()), + Err(cellule_runtime::Error::Node( + "native inventory category changed" + )) + )); + break; + } + let response = page( + &fixture.fleet, + &roster, + 0, + subject, + Instant::now() + Duration::from_secs(5), + ) + .await + .unwrap(); + scan.accept(response.request(), &response, clock().unwrap()) + .unwrap(); + } + assert_eq!(scan.cells().len(), CELL_COUNT); + assert!(scan.next_subject().is_err()); + assert!(scan.finish().is_err()); + tokio::time::timeout(Duration::from_secs(5), async { + while fixture.fleet.nodes[0] + .runtime() + .unreleased_cell_count() + .await + .unwrap() + != CELL_COUNT - 1 + { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + let fresh = fixture.capture().await; + assert!(fresh.complete); + assert_eq!(fresh.cells.len(), CELL_COUNT - 1); + fixture.close().await; +} + +#[tokio::test] +async fn canonical_owner_renewal_invalidates_the_exact_authority_interval() { + let fixture = Fixture::new().await; + let roster = fixture.roster().await; + let inventory = capture_node(&fixture, &roster, 0).await; + let mut cells = inventory.cells().to_vec(); + let cell = cells[0].observation.target.cell_id(); + let record = fixture.fleet.records.get(&cell).unwrap(); + let original = record.authority.load(cell).await.unwrap().unwrap(); + let mut renewal = original.value().clone(); + renewal.revision += 1; + renewal.progress += 1; + let renewed = record + .authority + .transition( + &original, + renewal, + cellule_runtime::control::Transition::Renew, + ) + .await + .unwrap(); + assert_eq!( + renewed.value().owner_fence(), + original.value().owner_fence() + ); + assert_eq!(renewed.value().root, original.value().root); + let unchanged = recheck_authority( + &fixture.fleet, + &HashMap::from([(cell, original.value().clone())]), + &mut cells, + ) + .await + .unwrap(); + assert!(!unchanged); + assert_eq!(cells.len(), CELL_COUNT - 1); + assert!( + cells + .iter() + .all(|owned| owned.observation.target.cell_id() != cell) + ); + drop(inventory); + fixture.close().await; +} diff --git a/crates/cellule-host/minion/scenario/observation/tests/mod.rs b/crates/cellule-host/minion/scenario/observation/tests/mod.rs new file mode 100644 index 00000000..9b6a5877 --- /dev/null +++ b/crates/cellule-host/minion/scenario/observation/tests/mod.rs @@ -0,0 +1,536 @@ +use super::*; +use cellule_host::fleet::FleetEnrollmentJournal; +use cellule_runtime::fleet::operations::{EnrollmentEndpoint, EnrollmentSpec}; +use cellule_runtime::node::{NodeCapacity, NodeFailureDomain}; +use ed25519_dalek::SigningKey; + +struct Fixture { + _root: tempfile::TempDir, + fleet: adapters::LocalFleet, +} +impl Fixture { + async fn new() -> Self { + let root = tempfile::tempdir().unwrap(); + let journal = Arc::new( + SqliteJournal::open( + root.path().join("observation.sqlite"), + scope(), + FleetProfile::default(), + clock().unwrap(), + ) + .await + .unwrap(), + ); + let mut nodes = Vec::new(); + let mut boots = Vec::new(); + let (records, _) = initialize(&root, &journal, &mut nodes, &mut boots, 60_000) + .await + .unwrap(); + let snapshot = journal.load_snapshot(scope()).await.unwrap(); + journal + .claim_controller( + scope(), + snapshot.head().revision(), + SessionId::from_bytes([206; 16]), + clock().unwrap(), + ) + .await + .unwrap(); + Self { + _root: root, + fleet: adapters::LocalFleet { + nodes, + boots, + records, + journal, + capture_sequence: std::sync::atomic::AtomicU64::new(0), + lose_release_replies: false, + lost_release_replies: std::sync::atomic::AtomicUsize::new(0), + expired_receiver_cleanups: std::sync::atomic::AtomicUsize::new(0), + }, + } + } + async fn roster(&self) -> FleetRoster { + let snapshot = self.fleet.journal.load_snapshot(scope()).await.unwrap(); + FleetRoster::collect( + self.fleet.journal.as_ref(), + &snapshot, + Instant::now() + Duration::from_secs(5), + ) + .await + .unwrap() + } + async fn capture(&self) -> Capture { + collect( + &self.fleet, + &self.roster().await, + Instant::now() + Duration::from_secs(5), + ) + .await + .unwrap() + } + async fn close(self) { + self.close_checked(false).await; + } + async fn close_checked(self, source_was_fenced: bool) { + for (index, node) in self.fleet.nodes.iter().enumerate() { + let joined = node.shutdown().await; + if let Err(original) = joined { + assert!( + source_was_fenced && index == 0, + "unexpected drain failure: {original:?}" + ); + let first = fenced_cause(&original).expect("original fenced drain source"); + assert_ne!(node.state(), NodeState::Stopped); + let again = node.shutdown().await.unwrap_err(); + assert!(std::ptr::eq( + first, + fenced_cause(&again).expect("retained fenced source") + )); + assert!( + self.fleet.boots[index] + .withdraw(&self.fleet.journal) + .await + .is_err() + ); + let record = self + .fleet + .journal + .load_enrollment(scope(), self.fleet.boots[index].spec.key().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(record.status(), EnrollmentStatus::Established); + } else { + assert_eq!(node.state(), NodeState::Stopped); + self.fleet.boots[index] + .withdraw(&self.fleet.journal) + .await + .unwrap(); + } + let stats = node.stats(); + assert_eq!(stats.retained_bytes(), 0); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.active_cells(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.file_descriptors(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); + } + self.fleet.journal.close().await.unwrap(); + } +} + +#[tokio::test] +async fn complete_writer_profile_uses_original_pages_authority_and_canonical_heartbeats() { + let fixture = Fixture::new().await; + let before = fixture.fleet.journal.load_snapshot(scope()).await.unwrap(); + let capture = fixture.capture().await; + assert!(capture.complete); + assert_eq!(capture.cells.len(), CELL_COUNT); + assert_eq!(capture.nodes.len(), 3); + assert!(capture.started <= capture.finished); + assert_eq!( + fixture.fleet.journal.load_snapshot(scope()).await.unwrap(), + before + ); + assert_eq!(fixture.fleet.capture_sequence.load(Ordering::SeqCst), 63); + for (index, ad) in capture.nodes.iter().enumerate() { + let canonical = fixture.fleet.boots[index] + .directory + .load(session(index), clock().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(canonical.advertisement(), ad); + assert_eq!(ad.generation(), 2); + assert_eq!( + ad.release(), + fixture.fleet.nodes[index] + .application() + .registry() + .release_digest() + ); + assert_eq!( + ad.verifying_key().unwrap(), + fixture.fleet.boots[index] + .advertisement + .verifying_key() + .unwrap() + ); + fixture.fleet.boots[index] + .guard + .as_ref() + .unwrap() + .check() + .unwrap(); + } + // A second pass advances actual native classifier sequences and canonical + // heartbeat generations; original establishment evidence remains unchanged. + let second = fixture.capture().await; + assert!(second.complete); + for (old, new) in capture.nodes.iter().zip(&second.nodes) { + assert_eq!(new.generation(), 3); + assert!( + new.operational_sample().unwrap().sequence > old.operational_sample().unwrap().sequence + ); + } + assert_eq!( + fixture.fleet.journal.load_snapshot(scope()).await.unwrap(), + before + ); + fixture.close().await; +} + +#[tokio::test] +async fn unexpected_advertised_boot_prevents_complete_counts_even_without_live_roles() { + let fixture = Fixture::new().await; + let original = &fixture.fleet.boots[0].advertisement; + let now = clock().unwrap(); + let extra = NodeAdvertisement::sign( + node_id(3), + session(3), + owner(3).endpoint, + scope().fleet, + original.certificate(), + original.image(), + original.release(), + &SigningKey::from_bytes(&[4; 32]), + 1, + now, + now + 30_000, + original.module_digests().to_vec(), + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + log_protocol: 1, + ..Default::default() + }, + ) + .unwrap(); + let observed = fixture.fleet.boots[0] + .directory + .create(extra, now) + .await + .unwrap(); + let capture = fixture.capture().await; + assert!(!capture.complete); + assert_eq!(capture.cells.len(), CELL_COUNT); + fixture.fleet.boots[0] + .directory + .withdraw(&observed, clock().unwrap()) + .await + .unwrap(); + fixture.close().await; +} + +#[tokio::test] +async fn pending_role_and_changed_journal_barrier_cannot_become_empty_coverage() { + let fixture = Fixture::new().await; + let original = fixture.roster().await; + let record = fixture.fleet.records.values().next().unwrap(); + let authority = record + .authority + .load(record.target.cell_id()) + .await + .unwrap() + .unwrap(); + let spec = EnrollmentSpec { + scope: scope(), + request: Digest::from_bytes([87; 32]), + role: EnrollmentRole::Reader { + target: record.target.clone(), + position: PublishedPosition { + incarnation: record.incarnation, + epoch: authority.value().epoch, + root: authority.value().root.clone().unwrap(), + }, + }, + source: Some(EnrollmentEndpoint { + node: node_id(0), + session: session(0), + intent_revision: 1, + }), + target: EnrollmentEndpoint { + node: node_id(1), + session: session(1), + intent_revision: 1, + }, + }; + fixture + .fleet + .journal + .accept_enrollment(&spec, clock().unwrap()) + .await + .unwrap(); + assert!( + collect( + &fixture.fleet, + &original, + Instant::now() + Duration::from_secs(5) + ) + .await + .is_err() + ); + let capture = fixture.capture().await; + assert!(!capture.complete); + assert_eq!(capture.cells.len(), CELL_COUNT); + fixture.close().await; +} + +#[tokio::test] +async fn foreign_current_owner_removes_stale_actor_from_pressure_and_count_inputs() { + let fixture = Fixture::new().await; + let record = fixture.fleet.records.values().next().unwrap(); + let original = record + .authority + .load(record.target.cell_id()) + .await + .unwrap() + .unwrap(); + let changed = original.value().takeover(owner(1)).unwrap(); + record + .authority + .transition( + &original, + changed, + cellule_runtime::control::Transition::Takeover, + ) + .await + .unwrap(); + let capture = fixture.capture().await; + assert!(!capture.complete); + assert_eq!(capture.cells.len(), CELL_COUNT - 1); + assert!( + capture + .cells + .iter() + .all(|row| row.observation.target.cell_id() != record.target.cell_id()) + ); + fixture.close_checked(true).await; +} + +#[tokio::test] +async fn fenced_guard_cannot_publish_capacity_or_recreate_a_retired_boot() { + let fixture = Fixture::new().await; + let boot = &fixture.fleet.boots[1]; + let original = boot + .directory + .load(session(1), clock().unwrap()) + .await + .unwrap() + .unwrap(); + let before = fixture.fleet.journal.load_snapshot(scope()).await.unwrap(); + boot.guard.as_ref().unwrap().fence(); + let error = boot + .refresh_capacity( + 1, + fixture.fleet.journal.as_ref(), + Instant::now() + Duration::from_secs(3), + ) + .await + .unwrap_err(); + assert!(matches!( + error.downcast_ref::(), + Some(cellule_runtime::Error::Fenced) + )); + let unchanged = boot + .directory + .load(session(1), clock().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(unchanged.advertisement(), original.advertisement()); + assert_eq!( + fixture.fleet.journal.load_snapshot(scope()).await.unwrap(), + before + ); + boot.node.shutdown().await.unwrap(); + boot.withdraw(&fixture.fleet.journal).await.unwrap(); + assert!(boot.directory.is_retired(session(1)).await.unwrap()); + assert!( + boot.refresh_capacity( + 1, + fixture.fleet.journal.as_ref(), + Instant::now() + Duration::from_secs(3) + ) + .await + .is_err() + ); + assert!(boot.directory.is_retired(session(1)).await.unwrap()); + fixture.close().await; +} + +#[tokio::test] +async fn canonical_heartbeat_consumes_live_maintenance_intent_and_preserves_existing_writers() { + use cellule_runtime::fleet::operations::{ + JournalTransition, MaintenanceOperation, OperationId, + }; + use cellule_runtime::node::NodeMode; + + let fixture = Fixture::new().await; + let boot = &fixture.fleet.boots[0]; + let key = boot.spec.key().unwrap(); + let original = fixture + .fleet + .journal + .load_enrollment(scope(), key) + .await + .unwrap() + .unwrap(); + let snapshot = fixture.fleet.journal.load_snapshot(scope()).await.unwrap(); + let now = clock().unwrap(); + let operation = MaintenanceOperation::new( + OperationId::from_bytes([80; 16]).unwrap(), + Digest::from_bytes([81; 32]), + node_id(0), + session(0), + 2, + now, + now + 60_000, + ) + .unwrap(); + fixture + .fleet + .journal + .compare_exchange( + &snapshot, + snapshot.head().controller().unwrap().epoch, + now, + &JournalTransition::BeginMaintenance(operation), + ) + .await + .unwrap(); + // No Cordon action is dispatched. The ordinary caller-driven heartbeat + // consumes the retained intent before its canonical CAS and guard renewal. + let refreshed = boot + .refresh_capacity( + 0, + fixture.fleet.journal.as_ref(), + Instant::now() + Duration::from_secs(3), + ) + .await + .unwrap(); + assert_eq!( + refreshed.operational_sample().unwrap().mode, + NodeMode::Draining + ); + assert!(!refreshed.accepts_new_roles(clock().unwrap())); + assert!( + boot.node + .runtime() + .node_admission() + .check_new_role() + .is_err() + ); + assert_eq!(boot.node.state(), NodeState::Ready); + assert!(boot.node.is_ready() && boot.node.is_management_ready()); + assert_eq!( + fixture + .fleet + .journal + .load_enrollment(scope(), key) + .await + .unwrap(), + Some(original) + ); + let canonical = boot + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(canonical.advertisement(), &refreshed); + boot.guard.as_ref().unwrap().check().unwrap(); + for record in fixture.fleet.records.values() { + let current = record + .authority + .load(record.target.cell_id()) + .await + .unwrap() + .unwrap(); + let handle = boot + .node + .runtime() + .local_handle(record.catalog.clone(), ¤t) + .await + .unwrap() + .unwrap(); + handle.query(64, 64, |_| Ok(Vec::new())).await.unwrap(); + } + assert_eq!(boot.node.stats().active_cells(), CELL_COUNT); + fixture.close().await; +} + +fn fenced_cause<'a>( + mut error: &'a (dyn std::error::Error + 'static), +) -> Option<&'a cellule_runtime::Error> { + loop { + if let Some(source) = error.downcast_ref::() + && matches!(source, cellule_runtime::Error::Fenced) + { + return Some(source); + } + error = error.source()?; + } +} + +#[tokio::test] +async fn one_canonical_release_keeps_other_native_writers_available_for_pressure_relief() { + let fixture = Fixture::new().await; + let roster = fixture.roster().await; + let original = fixture.capture().await; + assert!(original.complete); + let selected = original.cells.first().unwrap().observation.clone(); + let before = page( + &fixture.fleet, + &roster, + 0, + FleetSnapshotSubject::Cells(None), + Instant::now() + Duration::from_secs(5), + ) + .await + .unwrap(); + let FleetSnapshotNativePage::Cells(before_page) = before.page() else { + panic!("missing original actor page") + }; + let topology = before_page.topology(); + let released = fixture.fleet.nodes[0] + .runtime() + .release_idle_cell_at( + selected.target.cell_id(), + session(0), + selected.generation, + selected.incarnation, + selected.position.as_ref().unwrap().epoch, + ) + .await + .unwrap(); + assert_eq!(released, selected.position.unwrap()); + drop(before); + let after = page( + &fixture.fleet, + &roster, + 0, + FleetSnapshotSubject::Cells(None), + Instant::now() + Duration::from_secs(5), + ) + .await + .unwrap(); + let FleetSnapshotNativePage::Cells(after_page) = after.page() else { + panic!("missing repeated actor page") + }; + assert_ne!(after_page.topology(), topology); + assert_eq!(after_page.owned_cells(), CELL_COUNT - 1); + let mut candidates = original.cells; + retain_unchanged_writers(&mut candidates, 0, after_page.entries()); + assert_eq!(candidates.len(), CELL_COUNT - 1); + assert!( + candidates + .iter() + .all(|row| row.observation.target.cell_id() != selected.target.cell_id()) + ); + drop(after); + fixture.close().await; +} + +mod inventory; diff --git a/crates/cellule-host/minion/scenario/reader_tests/evacuation/fixture.rs b/crates/cellule-host/minion/scenario/reader_tests/evacuation/fixture.rs new file mode 100644 index 00000000..0dcbf759 --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/evacuation/fixture.rs @@ -0,0 +1,294 @@ +use super::*; + +impl Fixture { + pub(super) async fn new() -> Self { + let root = tempfile::tempdir().unwrap(); + let journal = Arc::new( + SqliteJournal::open( + root.path().join("evacuation.sqlite"), + scope(), + FleetProfile::default(), + clock().unwrap(), + ) + .await + .unwrap(), + ); + let app = application::compile().unwrap(); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + ObjectPath::from("reader-evacuation"), + [3; 16], + ); + let directory = NodeDirectory::new( + layout.clone(), + scope().fleet, + Digest::from_bytes([31; 32]), + app.registry().release_digest(), + ); + let limits = Limits { + max_database_bytes: 64 << 20, + max_capture_bytes: 16 << 20, + ..Limits::default() + }; + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + scope().application, + application::NAMESPACE, + &[1], + ) + .unwrap(); + let code = app.registry().module_digests()[0]; + let proof = CellCatalog::new(layout.clone(), target.tenant()) + .provision(CatalogEntry::new(&target, CatalogRole::Sql, code, 1).unwrap()) + .await + .unwrap(); + let incarnation = IncarnationId::from_bytes([1; 16]); + let authority = CellAuthority::new(layout.clone()); + let initial = authority + .create_initial(&proof, incarnation, owner(0)) + .await + .unwrap(); + let replica = CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + limits, + ) + .unwrap(); + let records = Arc::new(HashMap::from([( + target.cell_id(), + Record { + target: target.clone(), + incarnation, + catalog: proof.clone(), + replica: replica.clone(), + authority: authority.clone(), + }, + )])); + let mut nodes = Vec::new(); + let mut managers = Vec::new(); + let mut boots = Vec::new(); + for index in 0..3 { + let intent = journal + .register_initial_intent( + &NodeIntent::initial(scope(), node_id(index), session(index)).unwrap(), + ) + .await + .unwrap(); + let node = Arc::new( + CellNodeBuilder::new(app.clone()) + .with_runtime( + SqlWorkerPool::new(2, 8) + .unwrap() + .with_native_memory_limit(128 << 20) + .unwrap(), + 16 << 20, + ) + .with_replica_host( + Host::default().with_local_disk_budget(DiskBudget::new(8 << 30)), + ) + .with_session(session(index)) + .with_fleet_startup_intent(intent.clone()) + .build() + .unwrap(), + ); + node.install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + node.install_fleet_actions( + scope(), + node_id(index), + journal.clone(), + Arc::new(super::super::super::adapters::Cells { + records: records.clone(), + local: index, + root: root.path().into(), + }), + ) + .unwrap(); + let manager = node + .install_read_replicas( + layout.clone(), + directory.clone(), + root.path().join(format!("readers-{index}")), + limits, + ) + .unwrap(); + node.install_fleet_reader_enrollment(scope(), node_id(index), journal.clone()) + .unwrap(); + let ad = startup::advertisement(index, &node, &intent).await.unwrap(); + let spec = startup::spec(&intent).unwrap(); + let original = + startup::enroll(&journal, &directory, &spec, ad.clone(), clock().unwrap()) + .await + .unwrap(); + let guard = NodeLeaseGuard::new(clock().unwrap(), ad.expires_at_ms()).unwrap(); + node.install_node_lease_for_startup(guard.clone()).unwrap(); + node.confirm_fleet_startup(journal.as_ref(), spec.key().unwrap()) + .await + .unwrap(); + let observed = directory + .load(session(index), clock().unwrap()) + .await + .unwrap() + .unwrap(); + node.install_fleet_boot_withdrawal( + directory.clone(), + observed, + original, + journal.clone(), + ) + .unwrap(); + node.start().unwrap(); + boots.push(startup::BootOwner { + node: node.clone(), + directory: directory.clone(), + spec, + advertisement: ad, + guard: Some(guard), + }); + nodes.push(node); + managers.push(manager); + } + let handle = nodes[0] + .runtime() + .bootstrap( + proof, + replica, + authority, + initial, + root.path().join("source.sqlite"), + |tx| { + tx.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES (17)", + )?; + Ok(()) + }, + ) + .await + .unwrap(); + managers[1] + .set_target(&target, 0, 1) + .await + .unwrap() + .unwrap(); + boots[1] + .refresh_capacity(1, journal.as_ref(), Instant::now() + Duration::from_secs(3)) + .await + .unwrap(); + let transport = Arc::new(transport::NativePeers::new( + &nodes, + &managers, + &layout, + directory.clone(), + )); + let peer = ReplicaPeerClient::new( + app.registry(), + Arc::new(PeerSigner::new( + session(0), + app.registry().release_digest(), + SigningKey::from_bytes(&[1; 32]), + )), + PeerPrincipal { + issuer: "managed-owner".into(), + subject: "live-owner".into(), + actions: vec!["replica-maintenance".into()], + }, + transport.clone(), + ); + let ad = directory + .load(session(1), clock().unwrap()) + .await + .unwrap() + .unwrap(); + let description = CellDescription { + cell: target.cell_id(), + incarnation, + code, + schema: 1, + }; + peer.activate(&target, &directory, ad.advertisement().clone(), description) + .await + .unwrap(); + let reader = managers[1].resolve(target.clone()).await.unwrap(); + let completion = managers[1] + .enrollment_completion(target.cell_id()) + .await + .unwrap() + .unwrap(); + let original = journal + .load_enrollment(scope(), completion.spec.key().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(original.status(), EnrollmentStatus::Established); + let snapshot = journal.load_snapshot(scope()).await.unwrap(); + journal + .bootstrap_registry(snapshot.registry()) + .await + .unwrap(); + let now = clock().unwrap(); + let snapshot = journal.load_snapshot(scope()).await.unwrap(); + let mut snapshot = journal + .claim_controller( + scope(), + snapshot.head().revision(), + SessionId::from_bytes([206; 16]), + now, + ) + .await + .unwrap(); + for transition in [ + JournalTransition::BeginMaintenance( + MaintenanceOperation::new( + OperationId::from_bytes([80; 16]).unwrap(), + Digest::from_bytes([81; 32]), + node_id(1), + session(1), + 2, + now, + now + 60_000, + ) + .unwrap(), + ), + JournalTransition::Maintenance(MaintenanceEvent::Cordoned), + JournalTransition::Maintenance(MaintenanceEvent::BeginEvacuation), + ] { + snapshot = journal + .compare_exchange( + &snapshot, + snapshot.head().controller().unwrap().epoch, + clock().unwrap(), + &transition, + ) + .await + .unwrap(); + } + let operation = snapshot.head().maintenance().unwrap().clone(); + boots[1] + .refresh_capacity(1, journal.as_ref(), Instant::now() + Duration::from_secs(3)) + .await + .unwrap(); + assert_eq!( + nodes[1].runtime().node_admission().mode().unwrap(), + cellule_runtime::node::NodeMode::Draining + ); + assert!(nodes[1].is_management_ready()); + Self { + layout, + root, + journal, + description, + directory, + nodes, + managers, + boots, + handle, + target, + original, + reader, + operation, + peer, + transport, + } + } +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/evacuation/inventory_tests.rs b/crates/cellule-host/minion/scenario/reader_tests/evacuation/inventory_tests.rs new file mode 100644 index 00000000..d531276c --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/evacuation/inventory_tests.rs @@ -0,0 +1,136 @@ +//! Aggregate capture follows real native views and original producer retirement. +use super::*; +use cellule_host::fleet::{ + FleetNodeInventory, FleetNodeInventoryScan, FleetRoster, FleetSnapshotRequest, +}; + +async fn collect(fixture: &Fixture, index: usize) -> (FleetRoster, FleetNodeInventory) { + let snapshot = fixture.journal.load_snapshot(scope()).await.unwrap(); + let roster = FleetRoster::collect( + fixture.journal.as_ref(), + &snapshot, + Instant::now() + Duration::from_secs(5), + ) + .await + .unwrap(); + let mut scan = FleetNodeInventoryScan::new(&roster, node_id(index), session(index)).unwrap(); + let mut sequence = 0u64; + while let Some(subject) = scan.next_subject().unwrap() { + sequence += 1; + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.reader-inventory-test.v1\0"); + hash.update(&sequence.to_be_bytes()); + hash.update(&clock().unwrap().to_be_bytes()); + let now = clock().unwrap(); + let request = FleetSnapshotRequest::new( + snapshot.clone(), + Digest::from_bytes(*hash.finalize().as_bytes()), + node_id(index), + session(index), + subject, + 1, + now, + (now + 5_000).min(snapshot.head().controller().unwrap().expires_at_ms), + ) + .unwrap(); + let response = fixture.nodes[index] + .fleet_snapshot(request.clone()) + .await + .unwrap(); + scan.accept(&request, &response, clock().unwrap()).unwrap(); + } + let inventory = scan.finish().unwrap(); + (roster, inventory) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn aggregate_reader_capture_retains_original_inputs_and_joined_retirement() { + let fixture = Fixture::new().await; + let (roster, inventory) = collect(&fixture, 1).await; + assert_eq!(inventory.mode(), cellule_runtime::node::NodeMode::Draining); + assert!(inventory.bindings().readers); + assert!(inventory.cells().is_empty()); + assert_eq!(inventory.readers_closed(), Some(false)); + let jobs = inventory.reader_jobs().unwrap(); + assert_eq!(jobs.running(), 0); + assert_eq!(jobs.unobserved(), 0); + assert_eq!(jobs.joining(), 0); + assert_eq!(inventory.readers().len(), 1); + assert_eq!(inventory.reader_enrollments().len(), 1); + let original = &inventory.reader_enrollments()[0]; + assert_eq!(original.spec, *fixture.original.spec()); + assert_eq!( + original.accepted.as_ref().unwrap().accepted_at_ms(), + fixture.original.accepted_at_ms() + ); + assert!(original.published); + assert!(original.opening_started && original.opening_joined); + assert!(!inventory.readers()[0].locally_joined()); + inventory.validate_enrollments(&roster).unwrap(); + fixture.read_original(17).await; + fixture.spare().await; + fixture.evacuate().await.unwrap(); + let (after_roster, after) = collect(&fixture, 1).await; + assert!(after.readers().is_empty()); + assert!(after.reader_enrollments().is_empty()); + let retired = after_roster + .enrollments() + .iter() + .find(|row| row.spec() == fixture.original.spec()) + .unwrap(); + assert_eq!(retired.status(), EnrollmentStatus::Retired); + assert!(retired.settlement_evidence().is_some()); + after.validate_enrollments(&after_roster).unwrap(); + assert!(inventory.validate_enrollments(&after_roster).is_err()); + drop(inventory); + drop(after); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn lost_retirement_capture_preserves_the_original_source_error() { + let fixture = Fixture::new().await; + fixture.spare().await; + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(true, true); + let manager = fixture.managers[1].clone(); + let original = fixture.original.clone(); + let operation = fixture.operation.clone(); + let peer = fixture.peer.clone(); + let task = tokio::spawn(async move { + manager + .evacuate( + &original, + &operation, + &peer, + Instant::now() + Duration::from_secs(3), + ) + .await + }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + resume.send(()).unwrap(); + assert!(task.await.unwrap().is_err()); + let original = fixture.managers[1] + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + let (roster, inventory) = collect(&fixture, 1).await; + inventory.validate_enrollments(&roster).unwrap(); + assert_eq!(inventory.readers().len(), 1); + assert!(inventory.readers()[0].locally_joined()); + let observed = &inventory.reader_enrollments()[0]; + assert!(!observed.published); + assert_eq!(observed.spec, original.spec); + assert_eq!(observed.accepted, original.accepted); + assert_eq!(observed.event, original.event); + assert!(Arc::ptr_eq( + observed.journal_error.as_ref().unwrap(), + original.journal_error.as_ref().unwrap() + )); + drop(inventory); + fixture.evacuate().await.unwrap(); + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/evacuation/mod.rs b/crates/cellule-host/minion/scenario/reader_tests/evacuation/mod.rs new file mode 100644 index 00000000..4446461a --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/evacuation/mod.rs @@ -0,0 +1,145 @@ +//! Actual managed boots and nonzero reader replacement under retained intent. +use super::super::startup; +use super::*; +use cellule_host::read_replicas::ReaderEvacuation; +use cellule_runtime::{ + client::{CellDescription, CellReadReplica}, + fleet::operations::{JournalTransition, MaintenanceEvent, MaintenanceOperation, OperationId}, + peer::{PeerPrincipal, PeerSigner, ReplicaPeerClient}, +}; +use ed25519_dalek::SigningKey; + +mod fixture; +mod inventory_tests; +mod persisted; +mod tests; +mod transport; + +struct Fixture { + layout: CellStorageLayout, + root: tempfile::TempDir, + journal: Arc, + description: CellDescription, + directory: NodeDirectory, + nodes: Vec>, + managers: Vec, + boots: Vec, + handle: CellHandle, + target: CellTarget, + original: EnrollmentRecord, + reader: CellReadReplica, + operation: MaintenanceOperation, + peer: ReplicaPeerClient, + transport: Arc, +} + +impl Fixture { + async fn spare(&self) { + self.boots[2] + .refresh_capacity( + 2, + self.journal.as_ref(), + Instant::now() + Duration::from_secs(3), + ) + .await + .unwrap(); + tokio::time::timeout(Duration::from_secs(3), async { + loop { + let selected = self + .directory + .select_readers( + self.target.cell_id(), + session(0), + self.description.code, + 1, + clock().unwrap(), + 10_000, + ) + .await + .unwrap(); + if selected.len() == 1 && selected[0].session() == session(2) { + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + let node = self + .directory + .load(session(2), clock().unwrap()) + .await + .unwrap() + .unwrap(); + self.peer + .activate( + &self.target, + &self.directory, + node.advertisement().clone(), + self.description, + ) + .await + .unwrap(); + } + + async fn evacuate(&self) -> cellule_runtime::Result { + self.managers[1] + .evacuate( + &self.original, + &self.operation, + &self.peer, + Instant::now() + Duration::from_secs(3), + ) + .await + } + + async fn original_row(&self) -> EnrollmentRecord { + self.journal + .load_enrollment(scope(), self.original.spec().key().unwrap()) + .await + .unwrap() + .unwrap() + } + + async fn read_original(&self, value: i64) { + let result = self + .reader + .query::(Some(self.reader.receipt().await), 0) + .await + .unwrap(); + assert_eq!(result.output, value); + } + + async fn finish(self) { + for node in self.nodes.iter().rev() { + node.shutdown().await.unwrap(); + assert_eq!(node.state(), NodeState::Stopped); + let stats = node.stats(); + assert_eq!(stats.active_cells(), 0); + assert_eq!(stats.retained_bytes(), 0); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); + } + assert_eq!( + self.original_row().await.status(), + EnrollmentStatus::Retired + ); + for boot in &self.boots { + assert!( + self.directory + .is_withdrawn(boot.spec.target.session) + .await + .unwrap() + ); + let row = self + .journal + .load_enrollment(scope(), boot.spec.key().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(row.status(), EnrollmentStatus::Retired); + } + self.journal.close().await.unwrap(); + } +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/evacuation/persisted/mod.rs b/crates/cellule-host/minion/scenario/reader_tests/evacuation/persisted/mod.rs new file mode 100644 index 00000000..5706512f --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/evacuation/persisted/mod.rs @@ -0,0 +1,63 @@ +//! Real native evacuation, immutable local transactions and fresh status proofs. +use super::*; +use cellule_host::fleet::{ + FleetReaderEvacuationJournal, FleetReaderEvacuationPublication, FleetReaderEvacuationVerifier, +}; +use cellule_runtime::{fleet::operations::ReaderEvacuationRecord, read_policy::ReadPolicyStore}; + +mod races; +mod tests; + +fn deadline() -> Instant { + Instant::now() + Duration::from_secs(5) +} +impl Fixture { + fn verifier(&self) -> FleetReaderEvacuationVerifier { + FleetReaderEvacuationVerifier::new( + self.directory.clone(), + CellAuthority::new(self.layout.clone()), + ReadPolicyStore::new(self.layout.clone()), + self.peer.clone(), + ) + } + async fn stored(&self, digest: Digest) -> ReaderEvacuationRecord { + self.journal + .load_reader_evacuation(scope(), digest) + .await + .unwrap() + .unwrap() + } + async fn latest(&self) -> ReaderEvacuationRecord { + let snapshot = self.journal.load_snapshot(scope()).await.unwrap(); + self.journal + .latest_reader_evacuation( + &snapshot, + self.operation.id(), + self.original.spec().key().unwrap(), + ) + .await + .unwrap() + .unwrap() + } + async fn client(&self) -> SqliteJournal { + SqliteJournal::open( + self.root.path().join("evacuation.sqlite"), + scope(), + FleetProfile::default(), + clock().unwrap(), + ) + .await + .unwrap() + } + async fn publish(&self, capture: &ReaderEvacuation) -> FleetReaderEvacuationPublication { + FleetReaderEvacuationPublication::publish( + capture, + self.journal.as_ref(), + &self.verifier(), + deadline(), + clock, + ) + .await + .unwrap() + } +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/evacuation/persisted/races.rs b/crates/cellule-host/minion/scenario/reader_tests/evacuation/persisted/races.rs new file mode 100644 index 00000000..6f0766f7 --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/evacuation/persisted/races.rs @@ -0,0 +1,120 @@ +use super::*; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reader_policy_publication_cancelled_waiter_leaves_complete_original_commit_owned() { + let fixture = Fixture::new().await; + fixture.spare().await; + let capture = fixture.evacuate().await.unwrap(); + let (record, _) = capture.durable_record().unwrap(); + let verifier = fixture.verifier(); + let (entered, resume) = fixture.journal.pause_next_reader_evacuation_reply(false); + let mut work = Box::pin(FleetReaderEvacuationPublication::publish( + &capture, + fixture.journal.as_ref(), + &verifier, + deadline(), + clock, + )); + tokio::select! {result=&mut work=>panic!("unexpected completion {}",result.is_ok()),result=entered=>result.unwrap()} + drop(work); + let _ = resume.send(()); + let client = fixture.client().await; + assert_eq!( + client + .load_reader_evacuation(scope(), record.digest().unwrap()) + .await + .unwrap() + .unwrap(), + record + ); + verifier + .recheck(&client, &record, deadline(), clock) + .await + .unwrap(); + assert_eq!(fixture.latest().await, record); + client.close().await.unwrap(); + drop(capture); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reader_policy_publication_stale_registry_and_corrupt_pages_cannot_publish_a_candidate() { + let fixture = Fixture::new().await; + fixture.spare().await; + let capture = fixture.evacuate().await.unwrap(); + let (record, pages) = capture.durable_record().unwrap(); + let mut bytes = record.to_bytes().unwrap(); + let digest = record.pages()[0]; + let position = bytes + .windows(32) + .rposition(|bytes| bytes == digest.as_bytes()) + .unwrap(); + bytes[position] ^= 1; + let changed = ReaderEvacuationRecord::from_bytes(&bytes).unwrap(); + assert!( + fixture + .journal + .persist_reader_evacuation(capture.snapshot(), &changed, &pages, clock().unwrap()) + .await + .is_err() + ); + assert!( + fixture + .journal + .load_reader_evacuation(scope(), changed.digest().unwrap()) + .await + .unwrap() + .is_none() + ); + let snapshot = fixture.journal.load_snapshot(scope()).await.unwrap(); + fixture + .journal + .set_scheduling( + snapshot.registry(), + !snapshot.registry().scheduling_enabled(), + ) + .await + .unwrap(); + assert!( + FleetReaderEvacuationPublication::publish( + &capture, + fixture.journal.as_ref(), + &fixture.verifier(), + deadline(), + clock + ) + .await + .is_err() + ); + assert!( + fixture + .journal + .load_reader_evacuation(scope(), record.digest().unwrap()) + .await + .unwrap() + .is_none() + ); + assert_eq!(fixture.original_row().await, *record.retired()); + drop(capture); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reader_policy_publication_rechecks_authority_after_suspended_native_status() { + let fixture = Fixture::new().await; + fixture.spare().await; + let capture = fixture.evacuate().await.unwrap(); + let result = fixture.publish(&capture).await; + let record = result.record().unwrap(); + let verifier = fixture.verifier(); + let (entered, resume) = fixture.transport.pause_probe(1); + let mut work = Box::pin(verifier.recheck(fixture.journal.as_ref(), record, deadline(), clock)); + tokio::select! {result=&mut work=>panic!("unexpected completion {}",result.is_ok()),result=entered=>result.unwrap()} + fixture.handle.drain().await.unwrap(); + resume.send(()).unwrap(); + assert!(work.await.is_err()); + assert_eq!(fixture.stored(record.digest().unwrap()).await, *record); + assert_eq!(fixture.original_row().await, *record.retired()); + drop(capture); + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/evacuation/persisted/tests.rs b/crates/cellule-host/minion/scenario/reader_tests/evacuation/persisted/tests.rs new file mode 100644 index 00000000..4fd515d5 --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/evacuation/persisted/tests.rs @@ -0,0 +1,266 @@ +use super::*; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reader_policy_publication_can_refresh_redundancy_during_closing_without_restamping_retirement() + { + use cellule_runtime::fleet::operations::{ + DrainEvidence, JournalTransition, MaintenanceEvent, MaintenancePhase, + }; + let fixture = Fixture::new().await; + fixture.spare().await; + let capture = fixture.evacuate().await.unwrap(); + let publication = fixture.publish(&capture).await; + let record = publication.record().unwrap(); + let retired = fixture.original_row().await; + let snapshot = fixture.journal.load_snapshot(scope()).await.unwrap(); + let snapshot = fixture + .journal + .compare_exchange( + &snapshot, + snapshot.head().controller().unwrap().epoch, + clock().unwrap(), + &JournalTransition::Maintenance(MaintenanceEvent::ReadyToClose(DrainEvidence { + node: node_id(1), + session: session(1), + remaining_cells: 0, + unresolved_attempts: 0, + relocated: true, + readers_settled: true, + followers_settled: true, + facilities_closed: false, + stopped: false, + withdrawn: false, + })), + ) + .await + .unwrap(); + assert_eq!( + snapshot.head().maintenance().unwrap().phase(), + MaintenancePhase::Closing + ); + let verifier = fixture.verifier(); + verifier + .recheck(fixture.journal.as_ref(), record, deadline(), clock) + .await + .unwrap(); + fixture.managers[0] + .set_target(&fixture.target, 1, 0) + .await + .unwrap() + .unwrap(); + let candidate = verifier + .refresh(fixture.journal.as_ref(), record, deadline(), clock) + .await + .unwrap(); + assert_eq!( + candidate.record().operation().phase(), + MaintenancePhase::Closing + ); + let updated = FleetReaderEvacuationPublication::publish_refreshed( + &candidate, + fixture.journal.as_ref(), + &verifier, + deadline(), + clock, + ) + .await + .unwrap(); + updated.confirmed().unwrap(); + assert_eq!(updated.record().unwrap().retired(), &retired); + assert_eq!(fixture.original_row().await, retired); + drop(capture); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reader_policy_publication_commits_complete_pages_and_rechecks_after_independent_reconstruction() + { + let fixture = Fixture::new().await; + fixture.spare().await; + let capture = fixture.evacuate().await.unwrap(); + let (record, pages) = capture.durable_record().unwrap(); + let before = capture.snapshot().registry().revision(); + let publication = fixture.publish(&capture).await; + let check = publication.confirmed().unwrap(); + assert_eq!(publication.record().unwrap(), &record); + assert_eq!(check.record_digest(), record.digest().unwrap()); + assert_eq!(check.snapshot().registry().revision(), before + 1); + assert_eq!(check.replacements()[0].node, node_id(2)); + assert_eq!(record.retired(), &fixture.original_row().await); + for page in &pages { + assert_eq!( + fixture + .journal + .load_reader_evacuation_page(scope(), page.digest().unwrap()) + .await + .unwrap() + .as_ref(), + Some(page) + ); + } + let client = fixture.client().await; + let loaded = client + .load_reader_evacuation(scope(), record.digest().unwrap()) + .await + .unwrap() + .unwrap(); + let verifier = fixture.verifier(); + verifier + .recheck(&client, &loaded, deadline(), clock) + .await + .unwrap(); + let repeated = fixture.publish(&capture).await; + assert_eq!(repeated.record().unwrap(), &record); + assert_eq!(repeated.confirmed().unwrap().snapshot(), check.snapshot()); + assert_eq!(loaded.interval(), capture.interval()); + client.close().await.unwrap(); + drop(capture); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reader_policy_publication_adopts_lost_commit_reply_without_restamping_native_history() { + let fixture = Fixture::new().await; + fixture.spare().await; + let capture = fixture.evacuate().await.unwrap(); + let (record, _) = capture.durable_record().unwrap(); + let verifier = fixture.verifier(); + let (entered, resume) = fixture.journal.pause_next_reader_evacuation_reply(true); + let mut work = Box::pin(FleetReaderEvacuationPublication::publish( + &capture, + fixture.journal.as_ref(), + &verifier, + deadline(), + clock, + )); + tokio::select! {result=&mut work=>panic!("unexpected completion {}",result.is_ok()),result=entered=>result.unwrap()} + assert_eq!(fixture.stored(record.digest().unwrap()).await, record); + resume.send(()).unwrap(); + let result = work.await.unwrap(); + let error = result.record().unwrap_err(); + assert!(Arc::ptr_eq(&error, &result.confirmed().err().unwrap())); + let Error::Facility { source, .. } = error.as_ref() else { + panic!("original source absent") + }; + assert!(source.downcast_ref::().is_some()); + let client = fixture.client().await; + let original = client + .load_reader_evacuation(scope(), record.digest().unwrap()) + .await + .unwrap() + .unwrap(); + verifier + .recheck(&client, &original, deadline(), clock) + .await + .unwrap(); + let replay = fixture.publish(&capture).await; + assert_eq!(replay.record().unwrap(), &record); + assert!(Arc::ptr_eq(&error, &result.record().unwrap_err())); + client.close().await.unwrap(); + drop(capture); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reader_policy_publication_refreshes_changed_policy_without_repeating_native_retirement() { + let fixture = Fixture::new().await; + fixture.spare().await; + let capture = fixture.evacuate().await.unwrap(); + let verifier = fixture.verifier(); + let (entered, resume) = fixture.journal.pause_next_reader_evacuation_reply(false); + let mut work = Box::pin(FleetReaderEvacuationPublication::publish( + &capture, + fixture.journal.as_ref(), + &verifier, + deadline(), + clock, + )); + tokio::select! {result=&mut work=>panic!("unexpected completion {}",result.is_ok()),result=entered=>result.unwrap()} + fixture.managers[0] + .set_target(&fixture.target, 1, 0) + .await + .unwrap() + .unwrap(); + resume.send(()).unwrap(); + let result = work.await.unwrap(); + let original = result.record().unwrap(); + assert!(result.confirmed().is_err()); + let retired = fixture.original_row().await; + let candidate = verifier + .refresh(fixture.journal.as_ref(), original, deadline(), clock) + .await + .unwrap(); + assert_eq!(candidate.record().desired_readers(), 0); + assert!(candidate.pages().is_empty()); + assert_eq!(candidate.record().retired(), &retired); + let updated = FleetReaderEvacuationPublication::publish_refreshed( + &candidate, + fixture.journal.as_ref(), + &verifier, + deadline(), + clock, + ) + .await + .unwrap(); + let updated = updated.record().unwrap(); + verifier + .recheck(fixture.journal.as_ref(), updated, deadline(), clock) + .await + .unwrap(); + assert_ne!(original.digest().unwrap(), updated.digest().unwrap()); + assert_eq!(fixture.latest().await, *updated); + assert_eq!(fixture.stored(original.digest().unwrap()).await, *original); + // Old exact replay cannot restore the superseded latest pointer or revision. + let snapshot = fixture.journal.load_snapshot(scope()).await.unwrap(); + let (_, pages) = capture.durable_record().unwrap(); + assert_eq!( + fixture + .journal + .persist_reader_evacuation(capture.snapshot(), original, &pages, clock().unwrap()) + .await + .unwrap(), + *original + ); + assert_eq!( + fixture.journal.load_snapshot(scope()).await.unwrap(), + snapshot + ); + assert_eq!(fixture.latest().await, *updated); + assert!( + verifier + .recheck(fixture.journal.as_ref(), original, deadline(), clock) + .await + .is_err() + ); + assert_eq!(fixture.original_row().await, retired); + drop(capture); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reader_policy_publication_missing_ready_replacement_preserves_committed_history_and_blocks_refresh() + { + let fixture = Fixture::new().await; + fixture.spare().await; + let capture = fixture.evacuate().await.unwrap(); + let result = fixture.publish(&capture).await; + let record = result.record().unwrap(); + let verifier = fixture.verifier(); + fixture.nodes[2].shutdown().await.unwrap(); + assert!( + verifier + .recheck(fixture.journal.as_ref(), record, deadline(), clock) + .await + .is_err() + ); + assert!( + verifier + .refresh(fixture.journal.as_ref(), record, deadline(), clock) + .await + .is_err() + ); + assert_eq!(fixture.stored(record.digest().unwrap()).await, *record); + assert_eq!(fixture.original_row().await, *record.retired()); + drop(capture); + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/evacuation/tests.rs b/crates/cellule-host/minion/scenario/reader_tests/evacuation/tests.rs new file mode 100644 index 00000000..3c048c6d --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/evacuation/tests.rs @@ -0,0 +1,461 @@ +use super::*; +use cellule_runtime::fleet::operations::EnrollmentEvent; +use std::sync::atomic::Ordering; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn replacement_withdrawal_during_initial_probe_preserves_original_reader() { + let fixture = Fixture::new().await; + fixture.spare().await; + let (captured, resume) = fixture.transport.pause_probe(1); + let manager = fixture.managers[1].clone(); + let original = fixture.original.clone(); + let operation = fixture.operation.clone(); + let peer = fixture.peer.clone(); + let task = tokio::spawn(async move { + manager + .evacuate( + &original, + &operation, + &peer, + Instant::now() + Duration::from_secs(3), + ) + .await + }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + fixture.nodes[2].shutdown().await.unwrap(); + assert!(fixture.directory.is_withdrawn(session(2)).await.unwrap()); + resume.send(()).unwrap(); + assert!(task.await.unwrap().is_err()); + assert_eq!(fixture.original_row().await, fixture.original); + assert!( + !fixture + .reader + .lifecycle_observation() + .await + .admission_closed() + ); + fixture.read_original(17).await; + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn replacement_withdrawal_during_final_probe_refuses_completion_after_retirement() { + let fixture = Fixture::new().await; + fixture.spare().await; + let (captured, resume) = fixture.transport.pause_probe(2); + let manager = fixture.managers[1].clone(); + let original = fixture.original.clone(); + let operation = fixture.operation.clone(); + let peer = fixture.peer.clone(); + let task = tokio::spawn(async move { + manager + .evacuate( + &original, + &operation, + &peer, + Instant::now() + Duration::from_secs(3), + ) + .await + }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + let retired = fixture.original_row().await; + assert_eq!(retired.status(), EnrollmentStatus::Retired); + assert!( + fixture + .reader + .lifecycle_observation() + .await + .locally_joined() + ); + fixture.nodes[2].shutdown().await.unwrap(); + resume.send(()).unwrap(); + assert!(task.await.unwrap().is_err()); + assert!(fixture.evacuate().await.is_err()); + assert_eq!(fixture.original_row().await, retired); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn draining_reader_without_spare_remains_open_and_durably_established() { + let fixture = Fixture::new().await; + assert!(matches!(fixture.evacuate().await, Err(Error::Capacity(_)))); + assert_eq!(fixture.transport.probes.load(Ordering::SeqCst), 0); + // Exercise the installed periodic repair loop beyond its real five-second + // tick; a changed placement must not silently prune this managed reader. + tokio::time::sleep(Duration::from_secs(6)).await; + assert_eq!(fixture.original_row().await, fixture.original); + assert!( + !fixture + .reader + .lifecycle_observation() + .await + .admission_closed() + ); + fixture.read_original(17).await; + assert!(fixture.nodes[1].is_management_ready()); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn ready_native_replacement_precedes_joined_reader_retirement() { + let fixture = Fixture::new().await; + fixture.spare().await; + let proof = fixture.evacuate().await.unwrap(); + assert_eq!(proof.original(), &fixture.original); + assert_eq!(proof.retired(), &fixture.original_row().await); + assert_eq!(proof.retired().status(), EnrollmentStatus::Retired); + assert_eq!(proof.policy().unwrap().desired_readers(), 1); + assert_eq!(proof.replacements().len(), 1); + assert_eq!(proof.replacements()[0].node, node_id(2)); + assert_eq!(proof.replacements()[0].session, session(2)); + assert!(proof.replacements()[0].receipt.commit_sequence >= proof.minimum().commit_sequence); + assert!(proof.interval().0 <= proof.interval().1); + assert_eq!(fixture.transport.probes.load(Ordering::SeqCst), 2); + assert!( + fixture + .reader + .lifecycle_observation() + .await + .locally_joined() + ); + assert!(matches!( + fixture + .reader + .query::(None, 0) + .await, + Err(Error::Fenced) + )); + let spare = fixture.managers[2] + .resolve(fixture.target.clone()) + .await + .unwrap(); + assert_eq!( + spare + .query::(Some(proof.minimum()), 0) + .await + .unwrap() + .output, + 17 + ); + assert!(fixture.nodes[1].is_management_ready()); + drop(proof); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn lost_retirement_reply_keeps_original_view_and_evidence_for_replay() { + let fixture = Fixture::new().await; + fixture.spare().await; + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(true, true); + let manager = fixture.managers[1].clone(); + let original = fixture.original.clone(); + let operation = fixture.operation.clone(); + let peer = fixture.peer.clone(); + let task = tokio::spawn(async move { + manager + .evacuate( + &original, + &operation, + &peer, + Instant::now() + Duration::from_secs(3), + ) + .await + }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + let retired = fixture.original_row().await; + assert_eq!(retired.status(), EnrollmentStatus::Retired); + assert!( + fixture + .reader + .lifecycle_observation() + .await + .locally_joined() + ); + resume.send(()).unwrap(); + assert!(task.await.unwrap().is_err()); + let pending = fixture.managers[1] + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert!(!pending.published); + assert!(pending.journal_error.is_some()); + let proof = fixture.evacuate().await.unwrap(); + assert_eq!(proof.retired(), &retired); + assert!( + fixture.managers[1] + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .is_none() + ); + drop(proof); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn cancelled_retirement_waiter_resumes_original_committed_closure() { + let fixture = Fixture::new().await; + fixture.spare().await; + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(true, false); + let manager = fixture.managers[1].clone(); + let original = fixture.original.clone(); + let operation = fixture.operation.clone(); + let peer = fixture.peer.clone(); + let task = tokio::spawn(async move { + manager + .evacuate( + &original, + &operation, + &peer, + Instant::now() + Duration::from_secs(3), + ) + .await + }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + let retired = fixture.original_row().await; + task.abort(); + assert!(matches!(task.await, Err(error) if error.is_cancelled())); + assert!( + fixture + .reader + .lifecycle_observation() + .await + .locally_joined() + ); + // The journal transaction committed before this reply-only pause. Cancelling + // its waiter drops that reply receiver, while the producer retains its event. + assert!(resume.send(()).is_err()); + let completion = fixture.managers[1] + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert!(!completion.published); + assert!(matches!( + completion.event, + Some(EnrollmentEvent::Retired(_)) + )); + let proof = fixture.evacuate().await.unwrap(); + assert_eq!(proof.retired(), &retired); + drop(proof); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn deadline_preserves_committed_retirement_and_original_error_source() { + let fixture = Fixture::new().await; + fixture.spare().await; + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(true, false); + let manager = fixture.managers[1].clone(); + let original = fixture.original.clone(); + let operation = fixture.operation.clone(); + let peer = fixture.peer.clone(); + let task = tokio::spawn(async move { + manager + .evacuate( + &original, + &operation, + &peer, + Instant::now() + Duration::from_secs(1), + ) + .await + }); + tokio::time::timeout(Duration::from_secs(1), captured) + .await + .unwrap() + .unwrap(); + let retired = fixture.original_row().await; + let error = match task.await.unwrap() { + Ok(_) => panic!("paused publication returned proof"), + Err(error) => error, + }; + assert!( + matches!(&error, Error::Facility { name: "reader-evacuation-deadline", source } if source.downcast_ref::().is_some()) + ); + assert!(resume.send(()).is_err()); + let completion = fixture.managers[1] + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert!(!completion.published); + assert!(matches!( + completion.event, + Some(EnrollmentEvent::Retired(_)) + )); + let proof = fixture.evacuate().await.unwrap(); + assert_eq!(proof.retired(), &retired); + drop(proof); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn policy_change_during_signed_status_refuses_close_then_zero_policy_can_settle() { + let fixture = Fixture::new().await; + fixture.spare().await; + let (captured, resume) = fixture.transport.pause_probe(1); + let manager = fixture.managers[1].clone(); + let original = fixture.original.clone(); + let operation = fixture.operation.clone(); + let peer = fixture.peer.clone(); + let task = tokio::spawn(async move { + manager + .evacuate( + &original, + &operation, + &peer, + Instant::now() + Duration::from_secs(3), + ) + .await + }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + fixture.managers[0] + .set_target(&fixture.target, 1, 0) + .await + .unwrap() + .unwrap(); + resume.send(()).unwrap(); + assert!(matches!(task.await.unwrap(), Err(Error::Fenced))); + assert_eq!(fixture.original_row().await, fixture.original); + fixture.read_original(17).await; + let proof = fixture.evacuate().await.unwrap(); + assert_eq!(proof.policy().unwrap().desired_readers(), 0); + assert!(proof.replacements().is_empty()); + drop(proof); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn saturated_metadata_refuses_before_closure_and_releases_all_temporary_credit() { + let fixture = Fixture::new().await; + fixture.spare().await; + let stats = fixture.nodes[1].stats(); + let credit = fixture.nodes[1] + .runtime() + .try_reserve_node_metadata_bytes(stats.retained_capacity_bytes() - stats.retained_bytes()) + .unwrap(); + assert!(matches!(fixture.evacuate().await, Err(Error::Capacity(_)))); + assert_eq!(fixture.transport.probes.load(Ordering::SeqCst), 0); + assert_eq!(fixture.original_row().await, fixture.original); + assert!( + !fixture + .reader + .lifecycle_observation() + .await + .admission_closed() + ); + drop(credit); + fixture.read_original(17).await; + let proof = fixture.evacuate().await.unwrap(); + drop(proof); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn peer_refresh_after_initial_probe_requires_replacement_to_cover_closed_prefix() { + let fixture = Fixture::new().await; + fixture.spare().await; + let (captured, resume) = fixture.transport.pause_probe(1); + let manager = fixture.managers[1].clone(); + let original = fixture.original.clone(); + let operation = fixture.operation.clone(); + let peer = fixture.peer.clone(); + let task = tokio::spawn(async move { + manager + .evacuate( + &original, + &operation, + &peer, + Instant::now() + Duration::from_secs(3), + ) + .await + }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + let now = clock().unwrap(); + fixture + .handle + .execute( + MutationIdentity { + request_id: RequestId::from_bytes([90; 16]), + issued_at_ms: now, + expires_at_ms: now + 30_000, + }, + Digest::from_bytes([91; 32]), + now, + 64, + 64, + |tx| { + tx.execute("UPDATE counter SET value = 18", [])?; + Ok(HandlerOutcome::Success(vec![])) + }, + ) + .await + .unwrap(); + let closed_prefix = fixture + .reader + .refresh(&fixture.root.path().join("peer-refresh.sqlite")) + .await + .unwrap(); + let cellule_runtime::fleet::operations::EnrollmentRole::Reader { position, .. } = + &fixture.original.spec().role + else { + panic!("original is not a reader"); + }; + assert!(closed_prefix.commit_sequence > position.root.commit_sequence); + resume.send(()).unwrap(); + assert!(matches!( + task.await.unwrap(), + Err(Error::ReplicaUnavailable) + )); + assert!( + fixture + .reader + .lifecycle_observation() + .await + .locally_joined() + ); + assert_eq!( + fixture.original_row().await.status(), + EnrollmentStatus::Retired + ); + fixture.managers[2] + .activate(fixture.target.clone(), session(0)) + .await + .unwrap(); + let proof = fixture.evacuate().await.unwrap(); + assert!(proof.minimum().commit_sequence >= closed_prefix.commit_sequence); + assert!(proof.replacements()[0].receipt.commit_sequence >= closed_prefix.commit_sequence); + let reader = fixture.managers[2] + .resolve(fixture.target.clone()) + .await + .unwrap(); + assert_eq!( + reader + .query::(Some(proof.minimum()), 0) + .await + .unwrap() + .output, + 18 + ); + drop(proof); + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/evacuation/transport.rs b/crates/cellule-host/minion/scenario/reader_tests/evacuation/transport.rs new file mode 100644 index 00000000..a148b688 --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/evacuation/transport.rs @@ -0,0 +1,165 @@ +use super::*; +use cellule_runtime::peer::{ + PeerAuthorizer, PeerDispatcher, PeerRoundTrip, ResidentPeerCellResolver, VerifiedPeerRequest, + wire, +}; +use std::{ + future::Future, + pin::Pin, + sync::{ + Mutex, + atomic::{AtomicUsize, Ordering}, + }, +}; +use tokio::sync::oneshot; + +#[derive(Clone)] +pub(super) struct NativePeers { + directory: NodeDirectory, + dispatchers: Vec>, + pub(super) probes: Arc, + pause: Arc>>, +} +struct Pause { + number: usize, + captured: oneshot::Sender<()>, + resume: oneshot::Receiver<()>, +} +struct Authorizer; +impl PeerAuthorizer for Authorizer { + fn authorize(&self, request: &VerifiedPeerRequest) -> cellule_runtime::Result<()> { + if request.origin_session() != session(0) + || request.principal().issuer != "managed-owner" + || request.principal().subject != "live-owner" + || !request + .principal() + .actions + .iter() + .any(|action| action == "replica-maintenance") + { + return Err(Error::PeerAuthorization( + "maintenance fixture principal differs", + )); + } + Ok(()) + } +} +impl NativePeers { + pub(super) fn new( + nodes: &[Arc], + managers: &[ReadReplicaManager], + layout: &CellStorageLayout, + directory: NodeDirectory, + ) -> Self { + let dispatchers = nodes + .iter() + .zip(managers) + .map(|(node, manager)| { + let registry = node.application().registry(); + Arc::new( + PeerDispatcher::new( + registry.clone(), + Arc::new(ResidentPeerCellResolver::new( + node.runtime().clone(), + layout.clone(), + registry, + )), + Arc::new(Authorizer), + ) + .with_replica_control(Arc::new(manager.clone())) + .with_replica_resolver(Arc::new(manager.clone())), + ) + }) + .collect(); + Self { + directory, + dispatchers, + probes: Arc::new(AtomicUsize::new(0)), + pause: Arc::new(Mutex::new(None)), + } + } + pub(super) fn pause_probe( + &self, + number: usize, + ) -> (oneshot::Receiver<()>, oneshot::Sender<()>) { + self.probes.store(0, Ordering::SeqCst); + let (captured, waiting) = oneshot::channel(); + let (resume, paused) = oneshot::channel(); + *self.pause.lock().unwrap() = Some(Pause { + number, + captured, + resume: paused, + }); + (waiting, resume) + } +} +impl PeerRoundTrip for NativePeers { + fn send( + &self, + _: CellTarget, + _: Vec, + _: u32, + ) -> Pin>> + Send + 'static>> { + Box::pin(async { Err(Error::Peer("fixture requires explicit replica routing")) }) + } + fn send_to_node( + &self, + target: CellTarget, + node: cellule_runtime::node::NodeAdvertisement, + request: Vec, + _: u32, + ) -> Pin>> + Send + 'static>> { + let peers = self.clone(); + Box::pin(async move { + let index = (0..3) + .find(|index| node.node() == node_id(*index) && node.session() == session(*index)) + .ok_or(Error::Fenced)?; + let now = clock()?; + peers + .directory + .load(node.session(), now) + .await? + .ok_or(Error::Fenced)?; + // Trusted in-process routing pins the actual enrolled sender's + // certificate/key, then uses the same verifier as an mTLS adapter. + let verified = peers + .directory + .verify_peer_request( + &request, + Digest::from_bytes([30; 32]), + SigningKey::from_bytes(&[1; 32]).verifying_key().to_bytes(), + now, + ) + .await?; + if verified.target() != &target { + return Err(Error::Fenced); + } + let status = matches!( + verified.operation(), + Some(wire::peer_request::Operation::Read(wire::ReadRequest { + operation: Some(wire::read_request::Operation::ReplicaStatus(true)), + .. + })) + ); + let reply = peers.dispatchers[index] + .dispatch_bytes(&verified, clock()?) + .await?; + let pause = if status { + let number = peers.probes.fetch_add(1, Ordering::SeqCst) + 1; + let mut pending = peers.pause.lock().unwrap(); + if pending.as_ref().is_some_and(|pause| pause.number == number) { + pending.take() + } else { + None + } + } else { + None + }; + if let Some(pause) = pause { + let _ = pause.captured.send(()); + let _ = pause.resume.await; + } + Ok(reply) + }) + } +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/failed/barriers.rs b/crates/cellule-host/minion/scenario/reader_tests/failed/barriers.rs new file mode 100644 index 00000000..bb742323 --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/failed/barriers.rs @@ -0,0 +1,106 @@ +use super::*; +use tokio::sync::oneshot; + +struct PausedProcesses { + processes: Processes, + pause: Mutex, oneshot::Receiver<()>)>>, +} +impl FleetFailedBootProcesses for PausedProcesses { + fn confirm_stopped<'a>( + &'a self, + request: &'a FleetFailedBootProcessRequest, + ) -> FleetAdapterFuture<'a, FleetFailedBootProcessEvidence> { + Box::pin(async move { + let evidence = self.processes.confirm_stopped(request).await?; + let pause = self.pause.lock().unwrap().take(); + if let Some((entered, resume)) = pause { + let _ = entered.send(()); + resume.await?; + } + Ok(evidence) + }) + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn failed_reader_closure_rechecks_complete_barrier_after_suspended_process_provider() { + let fixture = Fixture::new().await; + fixture.fence().await; + let capture = fixture.capture(0).await; + fixture.join_and_retain(capture.request()).await; + let (entered, waiting) = oneshot::channel(); + let (resume, paused) = oneshot::channel(); + let processes = PausedProcesses { + processes: Processes::new(fixture.process_path()), + pause: Mutex::new(Some((entered, paused))), + }; + let mut work = Box::pin(capture.publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(0), + deadline(), + clock, + )); + tokio::select! { result = &mut work => panic!("unexpected completion {}", result.is_ok()), result = waiting => result.unwrap() } + fixture + .journal + .publish_enrollment_result( + &fixture.readers[1], + EnrollmentEvent::Established(Digest::from_bytes([215; 32])), + clock().unwrap(), + ) + .await + .unwrap(); + resume.send(()).unwrap(); + assert!(matches!(work.await, Err(Error::Fenced))); + assert_eq!(fixture.row(0).await, fixture.readers[0]); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn failed_reader_closure_retains_committed_row_without_restamping_expired_collection() { + let fixture = Fixture::new().await; + fixture.fence().await; + let capture = fixture.capture(0).await; + fixture.join_and_retain(capture.request()).await; + let now = clock().unwrap(); + let expired = capture.interval().0 + 30_001; + let processes = Processes::new(fixture.process_path()); + let mut calls = 0; + let result = capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(0), + deadline(), + || { + calls += 1; + Ok(if calls == 4 { expired } else { now }) + }, + ) + .await + .unwrap(); + assert_eq!(calls, 4); + let committed = result.record().unwrap(); + assert_eq!(committed.status(), EnrollmentStatus::Retired); + assert!(matches!( + result.closure_error().unwrap().as_ref(), + Error::Deadline + )); + let replay = fixture.capture(0).await; + let result = replay + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &Processes::new(fixture.process_path()), + session(0), + deadline(), + clock, + ) + .await + .unwrap(); + assert_eq!(result.confirmed().unwrap().reader(), committed); + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/failed/faults.rs b/crates/cellule-host/minion/scenario/reader_tests/failed/faults.rs new file mode 100644 index 00000000..e93d8745 --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/failed/faults.rs @@ -0,0 +1,300 @@ +use super::*; + +struct Foreign(FleetFailedBootProcessEvidence); +impl FleetFailedBootProcesses for Foreign { + fn confirm_stopped<'a>( + &'a self, + _: &'a FleetFailedBootProcessRequest, + ) -> FleetAdapterFuture<'a, FleetFailedBootProcessEvidence> { + Box::pin(async { Ok(self.0.clone()) }) + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn failed_reader_closure_refuses_live_receiver_and_failed_source_as_receiver_proof() { + let fixture = Fixture::new().await; + assert!( + FleetFailedReaderRetirement::capture( + fixture.journal.as_ref(), + &fixture.directory, + &fixture.roster().await, + &fixture.boot, + &fixture.readers[0], + session(0), + deadline(), + clock + ) + .await + .is_err() + ); + let source = fixture + .roster() + .await + .enrollments() + .iter() + .find(|row| { + matches!(row.spec().role, EnrollmentRole::Node { .. }) + && row.spec().target.session == session(0) + }) + .unwrap() + .clone(); + let original = fixture + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .unwrap(); + fixture + .directory + .withdraw(&original, clock().unwrap()) + .await + .unwrap(); + assert!( + FleetFailedReaderRetirement::capture( + fixture.journal.as_ref(), + &fixture.directory, + &fixture.roster().await, + &source, + &fixture.readers[0], + session(1), + deadline(), + clock + ) + .await + .is_err() + ); + // Source advertisement failure has not joined the original receiver views. + assert!( + !fixture.views[0] + .lifecycle_observation() + .await + .locally_joined() + ); + assert_eq!(fixture.row(0).await, fixture.readers[0]); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn failed_reader_closure_refuses_foreign_process_payload_history_and_stale_barrier() { + let fixture = Fixture::new().await; + fixture.fence().await; + let capture = fixture.capture(0).await; + fixture.join_and_retain(capture.request()).await; + // A response bound to another independently enrolled receiver lifetime + // cannot certify this original request, even with a well-shaped witness. + let other = Fixture::new().await; + other.fence().await; + let other_capture = other.capture(0).await; + assert_ne!(other_capture.request().digest(), capture.request().digest()); + let foreign = Foreign( + FleetFailedBootProcessEvidence::new(other_capture.request(), Digest::from_bytes([211; 32])) + .unwrap(), + ); + assert!(matches!( + capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &foreign, + session(0), + deadline(), + clock + ) + .await, + Err(Error::Fenced) + )); + other.finish().await; + let roster = fixture.roster().await; + let mut spec = fixture.readers[0].spec().clone(); + let EnrollmentRole::Reader { position, .. } = &mut spec.role else { + panic!("reader expected") + }; + position.root.commit_sequence += 1; + let source = roster + .intents() + .iter() + .find(|intent| intent.node() == node_id(0)) + .unwrap(); + let target = roster + .intents() + .iter() + .find(|intent| intent.node() == node_id(1)) + .unwrap(); + let changed = EnrollmentRecord::pending(spec, Some(source), target, clock().unwrap()).unwrap(); + assert!( + FleetFailedReaderRetirement::capture( + fixture.journal.as_ref(), + &fixture.directory, + &roster, + &fixture.boot, + &changed, + session(0), + deadline(), + clock + ) + .await + .is_err() + ); + let changed = EnrollmentRecord::pending( + fixture.readers[0].spec().clone(), + Some(source), + target, + fixture.readers[0].accepted_at_ms() + 1, + ) + .unwrap(); + assert!( + FleetFailedReaderRetirement::capture( + fixture.journal.as_ref(), + &fixture.directory, + &roster, + &fixture.boot, + &changed, + session(0), + deadline(), + clock + ) + .await + .is_err() + ); + fixture + .journal + .publish_enrollment_result( + &fixture.readers[1], + EnrollmentEvent::Established(Digest::from_bytes([214; 32])), + clock().unwrap(), + ) + .await + .unwrap(); + let processes = Processes::new(fixture.process_path()); + assert!(matches!( + capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(0), + deadline(), + clock + ) + .await, + Err(Error::Fenced) + )); + assert_eq!(processes.reads.load(Ordering::Acquire), 0); + assert_eq!(fixture.row(0).await, fixture.readers[0]); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn failed_reader_closure_keeps_committed_row_when_final_process_read_fails_or_changes() { + for changed in [false, true] { + let fixture = Fixture::new().await; + fixture.fence().await; + let capture = fixture.capture(0).await; + fixture.join_and_retain(capture.request()).await; + let processes = Processes::new(fixture.process_path()); + if changed { + *processes.final_witness.lock().unwrap() = Some(Digest::from_bytes([212; 32])); + } else { + *processes.final_fault.lock().unwrap() = true; + } + let result = capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(0), + deadline(), + clock, + ) + .await + .unwrap(); + assert!(result.confirmed().is_err()); + let committed = result.record().unwrap(); + assert_eq!(committed.status(), EnrollmentStatus::Retired); + let error = result.closure_error().unwrap(); + assert!(Arc::ptr_eq(&error, &result.closure_error().unwrap())); + if !changed { + let Error::Facility { source, .. } = error.as_ref() else { + panic!("original failure absent") + }; + assert_eq!( + source.downcast_ref::().unwrap().kind(), + std::io::ErrorKind::ConnectionReset + ); + } + let replay = fixture.capture(0).await; + let replay = replay + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &Processes::new(fixture.process_path()), + session(0), + deadline(), + clock, + ) + .await + .unwrap(); + assert_eq!(replay.confirmed().unwrap().reader(), committed); + fixture.finish().await; + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn failed_reader_closure_refuses_regressing_clock_and_retirement_with_different_witness() { + let fixture = Fixture::new().await; + fixture.fence().await; + let capture = fixture.capture(0).await; + fixture.join_and_retain(capture.request()).await; + let processes = Processes::new(fixture.process_path()); + let mut calls = 0; + let now = clock().unwrap(); + assert!(matches!( + capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(0), + deadline(), + || { + calls += 1; + Ok(if calls == 1 { now } else { now - 1 }) + } + ) + .await, + Err(Error::Deadline) + )); + assert_eq!(fixture.row(0).await, fixture.readers[0]); + let result = capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(0), + deadline(), + clock, + ) + .await + .unwrap(); + let committed = result.confirmed().unwrap().reader().clone(); + let replay = fixture.capture(0).await; + let foreign = Foreign( + FleetFailedBootProcessEvidence::new(replay.request(), Digest::from_bytes([213; 32])) + .unwrap(), + ); + assert!( + replay + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &foreign, + session(0), + deadline(), + clock + ) + .await + .is_err() + ); + assert_eq!(fixture.row(0).await, committed); + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/failed/fixture.rs b/crates/cellule-host/minion/scenario/reader_tests/failed/fixture.rs new file mode 100644 index 00000000..22e52161 --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/failed/fixture.rs @@ -0,0 +1,346 @@ +use super::*; + +impl Fixture { + pub(super) async fn new() -> Self { + let root = tempfile::tempdir().unwrap(); + let journal = Arc::new( + SqliteJournal::open( + root.path().join("journal.sqlite"), + scope(), + FleetProfile::default(), + clock().unwrap(), + ) + .await + .unwrap(), + ); + let app = application::compile().unwrap(); + let code = app.registry().module_digests()[0]; + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + ObjectPath::from("failed-readers"), + [3; 16], + ); + let directory = NodeDirectory::new( + layout.clone(), + scope().fleet, + Digest::from_bytes([31; 32]), + app.registry().release_digest(), + ); + let now = clock().unwrap(); + let mut boots = Vec::new(); + for index in [0, 1] { + let intent = journal + .register_initial_intent( + &NodeIntent::initial(scope(), node_id(index), session(index)).unwrap(), + ) + .await + .unwrap(); + let ad = NodeAdvertisement::sign( + node_id(index), + session(index), + owner(index).endpoint, + scope().fleet, + Digest::from_bytes([30; 32]), + Digest::from_bytes([31; 32]), + app.registry().release_digest(), + &SigningKey::from_bytes(&[index as u8 + 1; 32]), + 1, + now, + now + 30_000, + vec![code], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + free_memory_bytes: 1 << 30, + free_disk_bytes: 1 << 30, + job_credits: 4, + log_protocol: 1, + ..Default::default() + }, + ) + .unwrap(); + boots.push( + startup::enroll( + journal.as_ref(), + &directory, + &startup::spec(&intent).unwrap(), + ad, + now, + ) + .await + .unwrap(), + ); + } + let limits = Limits { + max_database_bytes: 64 << 20, + max_capture_bytes: 16 << 20, + ..Default::default() + }; + let node = Arc::new( + CellNodeBuilder::new(app.clone()) + .with_runtime( + SqlWorkerPool::new(2, 8) + .unwrap() + .with_native_memory_limit(128 << 20) + .unwrap(), + 16 << 20, + ) + .with_replica_host(Host::default().with_local_disk_budget(DiskBudget::new(8 << 30))) + .with_session(session(1)) + .build() + .unwrap(), + ); + node.install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + let manager = node + .install_read_replicas( + layout.clone(), + directory.clone(), + root.path().join("readers"), + limits, + ) + .unwrap(); + node.install_node_lease(NodeLeaseGuard::new(clock().unwrap(), now + 30_000).unwrap()) + .unwrap(); + node.start().unwrap(); + let source = CellRuntime::new_with_replica_host( + SqlWorkerPool::new(2, 8).unwrap(), + 16 << 20, + session(0), + Host::default().with_local_disk_budget(DiskBudget::new(8 << 30)), + ) + .unwrap(); + let mut readers = Vec::new(); + let mut views = Vec::new(); + let mut handles = Vec::new(); + for partition in [1, 2] { + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + scope().application, + application::NAMESPACE, + &[partition], + ) + .unwrap(); + let incarnation = IncarnationId::from_bytes([partition; 16]); + let proof = CellCatalog::new(layout.clone(), target.tenant()) + .provision(CatalogEntry::new(&target, CatalogRole::Sql, code, 1).unwrap()) + .await + .unwrap(); + let authority = CellAuthority::new(layout.clone()); + let initial = authority + .create_initial(&proof, incarnation, owner(0)) + .await + .unwrap(); + let replica = CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + limits, + ) + .unwrap(); + let handle = source + .bootstrap( + proof, + replica, + authority, + initial, + root.path().join(format!("source-{partition}.sqlite")), + |tx| { + tx.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES (17)", + )?; + Ok(()) + }, + ) + .await + .unwrap(); + manager.set_target(&target, 0, 1).await.unwrap().unwrap(); + let prepared = manager + .prepare_source(target.clone(), session(0)) + .await + .unwrap(); + let spec = EnrollmentSpec { + scope: scope(), + request: Digest::from_bytes([partition + 90; 32]), + source: Some(EnrollmentEndpoint { + node: node_id(0), + session: session(0), + intent_revision: 1, + }), + target: EnrollmentEndpoint { + node: node_id(1), + session: session(1), + intent_revision: 1, + }, + role: EnrollmentRole::Reader { + target: target.clone(), + position: PublishedPosition { + incarnation, + epoch: prepared.epoch(), + root: prepared.root().clone(), + }, + }, + }; + // This fixture's application owns and joins native opening. Pending + // precedes the ordinary exact-source activation; no host producer + // is installed that could later publish a different retirement. + let FleetEnrollmentAcceptance::New(mut original) = journal + .accept_enrollment(&spec, clock().unwrap()) + .await + .unwrap() + else { + panic!("new reader expected") + }; + let receipt = manager.activate_source(prepared).await.unwrap(); + let reader = manager.resolve(target).await.unwrap(); + assert_eq!( + reader + .query::(Some(receipt), 0) + .await + .unwrap() + .output, + 17 + ); + if partition == 1 { + original = journal + .publish_enrollment_result( + &original, + EnrollmentEvent::Established(Digest::from_bytes([71; 32])), + clock().unwrap(), + ) + .await + .unwrap(); + } + readers.push(original); + views.push(reader); + handles.push(handle); + } + let snapshot = journal.load_snapshot(scope()).await.unwrap(); + journal + .bootstrap_registry(snapshot.registry()) + .await + .unwrap(); + // All source handles refer to runtime-owned actors. Keep one for exact + // receipt readback; runtime shutdown joins both original source actors. + Self { + root, + journal, + node, + directory, + source, + handle: handles.remove(0), + manager, + boot: boots.remove(1), + readers, + views, + } + } + pub(super) async fn roster(&self) -> FleetRoster { + let snapshot = self.journal.load_snapshot(scope()).await.unwrap(); + FleetRoster::collect(self.journal.as_ref(), &snapshot, deadline()) + .await + .unwrap() + } + pub(super) async fn fence(&self) { + let original = self + .directory + .load(session(1), clock().unwrap()) + .await + .unwrap() + .unwrap(); + self.directory + .withdraw(&original, clock().unwrap()) + .await + .unwrap(); + } + pub(super) async fn capture(&self, index: usize) -> FleetFailedReaderRetirement { + FleetFailedReaderRetirement::capture( + self.journal.as_ref(), + &self.directory, + &self.roster().await, + &self.boot, + &self.readers[index], + session(0), + deadline(), + clock, + ) + .await + .unwrap() + } + pub(super) fn process_path(&self) -> PathBuf { + self.root.path().join("joined-original-lifetime") + } + pub(super) async fn join_and_retain(&self, request: &FleetFailedBootProcessRequest) { + assert_eq!(request.boot().spec(), self.boot.spec()); + self.node.shutdown().await.unwrap(); + assert_eq!(self.node.state(), NodeState::Stopped); + for view in &self.views { + assert!(view.lifecycle_observation().await.locally_joined()); + assert!(matches!( + view.query::(None, 0).await, + Err(Error::RuntimeClosed) + )); + } + let EnrollmentRole::Reader { target, .. } = &self.readers[0].spec().role else { + panic!("reader expected") + }; + assert!( + self.manager + .activate(target.clone(), session(0)) + .await + .is_err() + ); + let stats = self.node.stats(); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.retained_bytes(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); + // There are no application external jobs in this fixture. All native + // accepted work is joined and the closed manager cannot restart it. + let mut hash = blake3::Hasher::new(); + hash.update(b"joined-original-native-lifetime\0"); + hash.update(request.digest().as_bytes()); + let mut bytes = request.digest().as_bytes().to_vec(); + bytes.extend_from_slice(hash.finalize().as_bytes()); + use std::io::Write; + let mut file = std::fs::File::create(self.process_path()).unwrap(); + file.write_all(&bytes).unwrap(); + file.sync_all().unwrap(); + } + pub(super) async fn row(&self, index: usize) -> EnrollmentRecord { + self.journal + .load_enrollment(scope(), self.readers[index].spec().key().unwrap()) + .await + .unwrap() + .unwrap() + } + pub(super) async fn reconstruct(&self) -> SqliteJournal { + SqliteJournal::open( + self.root.path().join("journal.sqlite"), + scope(), + FleetProfile::default(), + clock().unwrap(), + ) + .await + .unwrap() + } + pub(super) async fn finish(self) { + self.node.shutdown().await.unwrap(); + let bytes = self + .handle + .query(64, 8, |connection| { + let value: i64 = + connection.query_row("SELECT value FROM counter", [], |row| row.get(0))?; + Ok(value.to_be_bytes().to_vec()) + }) + .await + .unwrap(); + assert_eq!(bytes, 17_i64.to_be_bytes()); + self.source.shutdown().await.unwrap(); + assert_eq!(self.source.stats().resident_bytes(), 0); + assert_eq!(self.source.stats().retained_bytes(), 0); + assert_eq!(self.source.stats().worker_jobs(), 0); + assert_eq!(self.source.stats().local_disk_reserved_bytes(), 0); + self.journal.close().await.unwrap(); + } +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/failed/mod.rs b/crates/cellule-host/minion/scenario/reader_tests/failed/mod.rs new file mode 100644 index 00000000..d15b11b0 --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/failed/mod.rs @@ -0,0 +1,91 @@ +//! Application-owned enrollment around real native readers and joined lifetimes. +//! This proves in-process closure; OS crash/provider qualification is separate. +use super::super::startup; +use super::*; +use cellule_host::fleet::{ + FleetAdapterFuture, FleetEnrollmentAcceptance, FleetFailedBootProcessEvidence, + FleetFailedBootProcessRequest, FleetFailedBootProcesses, FleetFailedBootRetirement, + FleetFailedReaderRetirement, FleetRoster, +}; +use cellule_runtime::{ + client::CellReadReplica, + fleet::operations::{ + EnrollmentEndpoint, EnrollmentEvent, EnrollmentRole, EnrollmentSpec, PublishedPosition, + }, + node::{NodeAdvertisement, NodeCapacity, NodeFailureDomain}, +}; +use ed25519_dalek::SigningKey; +use std::sync::{ + Mutex, + atomic::{AtomicUsize, Ordering}, +}; + +mod barriers; +mod faults; +mod fixture; +mod tests; + +fn deadline() -> Instant { + Instant::now() + Duration::from_secs(5) +} + +struct Fixture { + root: tempfile::TempDir, + journal: Arc, + node: Arc, + directory: NodeDirectory, + source: CellRuntime, + handle: CellHandle, + manager: ReadReplicaManager, + boot: EnrollmentRecord, + readers: Vec, + views: Vec, +} + +struct Processes { + path: PathBuf, + reads: AtomicUsize, + final_fault: Mutex, + final_witness: Mutex>, +} +impl Processes { + fn new(path: PathBuf) -> Self { + Self { + path, + reads: AtomicUsize::new(0), + final_fault: Mutex::new(false), + final_witness: Mutex::new(None), + } + } +} +impl FleetFailedBootProcesses for Processes { + fn confirm_stopped<'a>( + &'a self, + request: &'a FleetFailedBootProcessRequest, + ) -> FleetAdapterFuture<'a, FleetFailedBootProcessEvidence> { + Box::pin(async move { + if self.reads.fetch_add(1, Ordering::AcqRel) == 1 { + if *self.final_fault.lock().unwrap() { + return Err(std::io::Error::new( + std::io::ErrorKind::ConnectionReset, + "original native lifetime evidence read lost", + ) + .into()); + } + if let Some(witness) = *self.final_witness.lock().unwrap() { + return Ok(FleetFailedBootProcessEvidence::new(request, witness)?); + } + } + let bytes = std::fs::read(&self.path)?; + if bytes.len() != 64 || bytes[..32] != *request.digest().as_bytes() { + return Err( + std::io::Error::other("original native lifetime request differs").into(), + ); + } + Ok(FleetFailedBootProcessEvidence::new( + request, + Digest::from_bytes(bytes[32..].try_into()?), + )?) + }) + } +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/failed/tests.rs b/crates/cellule-host/minion/scenario/reader_tests/failed/tests.rs new file mode 100644 index 00000000..7193f66c --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/failed/tests.rs @@ -0,0 +1,302 @@ +use super::*; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn failed_reader_closure_requires_joined_original_native_lifetime_before_boot_closure() { + let fixture = Fixture::new().await; + fixture.fence().await; + let capture = fixture.capture(0).await; + let processes = Processes::new(fixture.process_path()); + assert!( + capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(0), + deadline(), + clock + ) + .await + .is_err() + ); + assert_eq!(fixture.row(0).await, fixture.readers[0]); + assert!( + !fixture.views[0] + .lifecycle_observation() + .await + .locally_joined() + ); + assert!( + FleetFailedBootRetirement::capture( + fixture.journal.as_ref(), + &fixture.directory, + &fixture.roster().await, + &fixture.boot, + session(0), + deadline(), + clock + ) + .await + .is_err() + ); + fixture.join_and_retain(capture.request()).await; + let request = capture.request().digest(); + for index in [0, 1] { + let capture = fixture.capture(index).await; + assert_eq!(capture.request().digest(), request); + let publication = capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(0), + deadline(), + clock, + ) + .await + .unwrap(); + let closure = publication.confirmed().unwrap(); + let retired = closure.reader(); + assert_eq!(retired.spec(), fixture.readers[index].spec()); + assert_eq!( + retired.accepted_at_ms(), + fixture.readers[index].accepted_at_ms() + ); + assert_eq!( + retired.established_evidence(), + fixture.readers[index].established_evidence() + ); + assert_eq!(retired.status(), EnrollmentStatus::Retired); + assert_eq!(closure.process().request_digest(), request); + let replay = fixture.capture(index).await; + let replay = replay + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &Processes::new(fixture.process_path()), + session(0), + deadline(), + clock, + ) + .await + .unwrap(); + assert_eq!(replay.confirmed().unwrap().reader(), retired); + assert_eq!(replay.confirmed().unwrap().digest(), closure.digest()); + if index == 0 { + assert!( + FleetFailedBootRetirement::capture( + fixture.journal.as_ref(), + &fixture.directory, + &fixture.roster().await, + &fixture.boot, + session(0), + deadline(), + clock + ) + .await + .is_err() + ); + } + } + let boot = FleetFailedBootRetirement::capture( + fixture.journal.as_ref(), + &fixture.directory, + &fixture.roster().await, + &fixture.boot, + session(0), + deadline(), + clock, + ) + .await + .unwrap(); + assert_eq!(boot.request().digest(), request); + boot.publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(0), + deadline(), + clock, + ) + .await + .unwrap() + .confirmed() + .unwrap(); + assert!( + !fixture + .roster() + .await + .required_boots() + .iter() + .any(|boot| boot.session == session(1)) + ); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn failed_reader_closure_adopts_lost_reply_after_independent_adapter_reconstruction() { + let fixture = Fixture::new().await; + fixture.fence().await; + let capture = fixture.capture(1).await; + fixture.join_and_retain(capture.request()).await; + let processes = Processes::new(fixture.process_path()); + let (entered, resume) = fixture.journal.pause_next_enrollment_reply(true, true); + let mut work = Box::pin(capture.publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(0), + deadline(), + clock, + )); + tokio::select! { result=&mut work=>panic!("unexpected completion {}",result.is_ok()), result=entered=>result.unwrap() } + let committed = fixture.row(1).await; + resume.send(()).unwrap(); + let publication = work.await.unwrap(); + assert!(publication.confirmed().is_err()); + let error = publication.record().unwrap_err(); + let Error::Facility { source, .. } = error.as_ref() else { + panic!("original publication source absent") + }; + assert!(source.downcast_ref::().is_some()); + assert!(Arc::ptr_eq(&error, &publication.record().unwrap_err())); + fixture.journal.close().await.unwrap(); + let journal = fixture.reconstruct().await; + let snapshot = journal.load_snapshot(scope()).await.unwrap(); + let roster = FleetRoster::collect(&journal, &snapshot, deadline()) + .await + .unwrap(); + let replay = FleetFailedReaderRetirement::capture( + &journal, + &fixture.directory, + &roster, + &fixture.boot, + &fixture.readers[1], + session(0), + deadline(), + clock, + ) + .await + .unwrap(); + let replay = replay + .publish( + &journal, + &fixture.directory, + &Processes::new(fixture.process_path()), + session(0), + deadline(), + clock, + ) + .await + .unwrap(); + assert_eq!(replay.confirmed().unwrap().reader(), &committed); + assert!(committed.established_evidence().is_none()); + journal.close().await.unwrap(); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn failed_reader_closure_cancelled_waiter_is_joined_by_original_backend_owner() { + let fixture = Fixture::new().await; + fixture.fence().await; + let capture = fixture.capture(0).await; + fixture.join_and_retain(capture.request()).await; + let processes = Processes::new(fixture.process_path()); + let (entered, resume) = fixture.journal.pause_next_enrollment_reply(true, false); + let mut work = Box::pin(capture.publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(0), + deadline(), + clock, + )); + tokio::select! { result=&mut work=>panic!("unexpected completion {}",result.is_ok()), result=entered=>result.unwrap() } + let committed = fixture.row(0).await; + drop(work); + let _ = resume.send(()); + fixture.journal.close().await.unwrap(); + let journal = fixture.reconstruct().await; + let snapshot = journal.load_snapshot(scope()).await.unwrap(); + let roster = FleetRoster::collect(&journal, &snapshot, deadline()) + .await + .unwrap(); + let capture = FleetFailedReaderRetirement::capture( + &journal, + &fixture.directory, + &roster, + &fixture.boot, + &fixture.readers[0], + session(0), + deadline(), + clock, + ) + .await + .unwrap(); + let result = capture + .publish( + &journal, + &fixture.directory, + &Processes::new(fixture.process_path()), + session(0), + deadline(), + clock, + ) + .await + .unwrap(); + assert_eq!(result.confirmed().unwrap().reader(), &committed); + journal.close().await.unwrap(); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn failed_reader_closure_retains_original_establishment_completed_after_acceptance() { + let fixture = Fixture::new().await; + fixture.fence().await; + let initial = fixture.capture(1).await; + fixture.join_and_retain(initial.request()).await; + let established = fixture + .journal + .publish_enrollment_result( + &fixture.readers[1], + EnrollmentEvent::Established(Digest::from_bytes([72; 32])), + clock().unwrap(), + ) + .await + .unwrap(); + assert!( + initial + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &Processes::new(fixture.process_path()), + session(0), + deadline(), + clock + ) + .await + .is_err() + ); + let recaptured = fixture.capture(1).await; + assert_eq!(recaptured.request().digest(), initial.request().digest()); + let result = recaptured + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &Processes::new(fixture.process_path()), + session(0), + deadline(), + clock, + ) + .await + .unwrap(); + assert_eq!( + result.confirmed().unwrap().reader().established_evidence(), + established.established_evidence() + ); + assert_eq!( + result.confirmed().unwrap().reader().accepted_at_ms(), + fixture.readers[1].accepted_at_ms() + ); + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/inventory/mod.rs b/crates/cellule-host/minion/scenario/reader_tests/inventory/mod.rs new file mode 100644 index 00000000..0e439e2e --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/inventory/mod.rs @@ -0,0 +1,341 @@ +//! Inventory is read-only through the original producer's paused native/journal work. +use super::*; +use cellule_host::read_replicas::{ReaderEnrollmentInventoryCursor, ReaderEnrollmentInventoryPage}; +use cellule_runtime::{fleet::operations::EnrollmentEvent, node::NodeMode}; + +mod variable_rows; + +fn capture(fixture: &ReaderFixture, limit: usize) -> ReaderEnrollmentInventoryPage { + fixture + .manager + .fleet_reader_enrollments_page(None, limit, clock().unwrap()) + .unwrap() + .unwrap() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn paused_acceptance_exposes_original_request_without_waiting_or_settling() { + let fixture = ReaderFixture::new().await; + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(false, true); + let manager = fixture.manager.clone(); + let target = fixture.target.clone(); + let opening = tokio::spawn(async move { manager.activate(target, session(0)).await }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + let pending = fixture.rows().await; + let before = fixture.node.stats().retained_bytes(); + let now = clock().unwrap(); + let page = fixture + .manager + .fleet_reader_enrollments_page(None, 128, now) + .unwrap() + .unwrap(); + assert_eq!(page.session(), session(1)); + assert_eq!(page.observed_at_ms(), now); + assert_eq!(page.total_enrollments(), 1); + assert_eq!(page.entries().len(), 1); + assert_eq!(page.entries()[0].spec, *pending[0].spec()); + assert!(page.entries()[0].accepted.is_none()); + assert!(!page.entries()[0].opening_started); + assert!(!page.entries()[0].opening_joined); + assert!(page.entries()[0].event.is_none()); + assert_eq!(page.jobs().retained(), 1); + assert_eq!(page.jobs().running(), 1); + assert_eq!(page.jobs().unobserved(), 0); + assert_eq!(page.jobs().joining(), 0); + assert!(page.next().is_none()); + assert_eq!(fixture.node.stats().retained_bytes(), before + (1 << 20)); + let completion = tokio::time::timeout( + Duration::from_secs(3), + fixture + .manager + .enrollment_completion(fixture.target.cell_id()), + ) + .await + .unwrap() + .unwrap() + .unwrap(); + assert_eq!(completion.spec, page.entries()[0].spec); + assert_eq!(fixture.rows().await, pending); + assert_eq!(fixture.node.stats().worker_jobs(), 0); + drop(page); + assert_eq!(fixture.node.stats().retained_bytes(), before); + resume.send(()).unwrap(); + assert!(opening.await.unwrap().is_err()); + let page = capture(&fixture, 1); + let completion = fixture + .manager + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert!(Arc::ptr_eq( + page.entries()[0].journal_error.as_ref().unwrap(), + completion.journal_error.as_ref().unwrap() + )); + assert_eq!(page.jobs().running(), 0); + assert_eq!(page.jobs().unobserved(), 0); + assert!(page.jobs().protocol_failure().is_some()); + drop(page); + fixture + .manager + .remove(fixture.target.cell_id()) + .await + .unwrap(); + fixture.finish_status(EnrollmentStatus::Refused).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn paused_establishment_exposes_joined_native_open_and_original_pending_reply() { + let fixture = ReaderFixture::new().await; + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(true, false); + let manager = fixture.manager.clone(); + let target = fixture.target.clone(); + let opening = tokio::spawn(async move { manager.activate(target, session(0)).await }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + let original = fixture.rows().await; + let page = capture(&fixture, 1); + let progress = &page.entries()[0]; + assert_eq!(progress.spec, *original[0].spec()); + assert_eq!( + progress.accepted.as_ref().unwrap().status(), + EnrollmentStatus::Pending + ); + assert!(progress.opening_started && progress.opening_joined); + assert!(!progress.published); + assert!( + matches!(progress.event, Some(EnrollmentEvent::Established(e)) if Some(e) == original[0].established_evidence()) + ); + assert_eq!(page.jobs().running(), 1); + assert_eq!(fixture.rows().await, original); + drop(page); + resume.send(()).unwrap(); + opening.await.unwrap().unwrap(); + fixture.read().await; + let page = capture(&fixture, 1); + assert!(page.entries()[0].published); + assert_eq!(page.jobs().running(), 0); + drop(page); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn pages_traverse_all_original_readers_and_reject_progress_and_mode_changes() { + let fixture = ReaderFixture::new().await; + let (second, second_handle) = fixture.additional_cell(2).await; + let (third, third_handle) = fixture.additional_cell(3).await; + for target in [&fixture.target, &second] { + fixture + .manager + .activate(target.clone(), session(0)) + .await + .unwrap(); + } + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(true, false); + let manager = fixture.manager.clone(); + let target = third.clone(); + let opening = tokio::spawn(async move { manager.activate(target, session(0)).await }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + let rows = fixture.rows().await; + let before = fixture.node.stats().retained_bytes(); + let first = capture(&fixture, 1); + assert_eq!(first.total_enrollments(), 3); + let cursor = + ReaderEnrollmentInventoryCursor::from_bytes(&first.next().unwrap().to_bytes()).unwrap(); + let rest = fixture + .manager + .fleet_reader_enrollments_page(Some(cursor), 128, clock().unwrap()) + .unwrap() + .unwrap(); + assert_eq!(rest.topology(), first.topology()); + assert_eq!(rest.entries().len(), 2); + assert!(rest.next().is_none()); + let mut observed = first + .entries() + .iter() + .chain(rest.entries()) + .map(|r| r.spec.clone()) + .collect::>(); + let cells = first + .entries() + .iter() + .chain(rest.entries()) + .map(|r| r.source.description().cell) + .collect::>(); + assert!( + cells + .windows(2) + .all(|pair| pair[0].as_bytes() < pair[1].as_bytes()) + ); + observed.sort_by_key(|r| *r.key().unwrap().as_bytes()); + let mut expected = rows.iter().map(|r| r.spec().clone()).collect::>(); + expected.sort_by_key(|r| *r.key().unwrap().as_bytes()); + assert_eq!(observed, expected); + assert_eq!(fixture.node.stats().retained_bytes(), before + (2 << 20)); + drop(rest); + drop(first); + assert_eq!(fixture.node.stats().retained_bytes(), before); + resume.send(()).unwrap(); + opening.await.unwrap().unwrap(); + assert!( + fixture + .manager + .fleet_reader_enrollments_page(Some(cursor), 128, clock().unwrap()) + .is_err() + ); + fixture.manager.remove(second.cell_id()).await.unwrap(); + assert!( + fixture + .manager + .fleet_reader_enrollments_page(Some(cursor), 128, clock().unwrap()) + .is_err() + ); + let first = capture(&fixture, 1); + let cursor = first.next().unwrap(); + drop(first); + fixture.node.runtime().node_admission().cordon().unwrap(); + assert!( + fixture + .manager + .fleet_reader_enrollments_page(Some(cursor), 128, clock().unwrap()) + .is_err() + ); + let cordoned = capture(&fixture, 128); + assert_eq!(cordoned.mode(), NodeMode::Cordoned); + assert_eq!(cordoned.total_enrollments(), 2); + drop(cordoned); + fixture.manager.shutdown().await.unwrap(); + let closed = capture(&fixture, 1); + assert_eq!(closed.total_enrollments(), 0); + assert!(closed.jobs().draining()); + assert_eq!(closed.jobs().retained(), 0); + assert_eq!(closed.jobs().unobserved(), 0); + drop(closed); + fixture.node.shutdown().await.unwrap(); + assert!(matches!( + fixture + .manager + .fleet_reader_enrollments_page(None, 1, clock().unwrap()), + Err(Error::RuntimeClosed) + )); + second_handle.drain().await.unwrap(); + third_handle.drain().await.unwrap(); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn invalid_bounds_stale_keys_and_memory_refusal_preserve_obligations() { + let fixture = ReaderFixture::new().await; + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(true, false); + let manager = fixture.manager.clone(); + let target = fixture.target.clone(); + let opening = tokio::spawn(async move { manager.activate(target, session(0)).await }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + let original = fixture.rows().await; + let before = fixture.node.stats().retained_bytes(); + for (limit, now) in [(0, 0), (129, 0), (1, -1), (usize::MAX, 0)] { + assert!( + fixture + .manager + .fleet_reader_enrollments_page(None, limit, now) + .is_err() + ); + } + let page = capture(&fixture, 1); + let mut bytes = [9; 64]; + bytes[..32].copy_from_slice(page.topology().as_bytes()); + drop(page); + let cursor = ReaderEnrollmentInventoryCursor::from_bytes(&bytes).unwrap(); + assert!( + fixture + .manager + .fleet_reader_enrollments_page(Some(cursor), 128, clock().unwrap()) + .is_err() + ); + assert_eq!(fixture.node.stats().retained_bytes(), before); + let mut held = Vec::new(); + while let Ok(reservation) = fixture.node.runtime().try_reserve_node_bytes(1 << 20) { + held.push(reservation); + } + let occupied = fixture.node.stats().retained_bytes(); + assert!(matches!( + fixture + .manager + .fleet_reader_enrollments_page(None, 128, clock().unwrap()), + Err(Error::Capacity(_)) + )); + assert_eq!(fixture.node.stats().retained_bytes(), occupied); + assert_eq!(fixture.rows().await, original); + drop(held); + assert_eq!(fixture.node.stats().retained_bytes(), before); + drop(capture(&fixture, 1)); + resume.send(()).unwrap(); + opening.await.unwrap().unwrap(); + fixture.read().await; + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn accepted_preparation_without_a_request_is_visible_as_running_work() { + use std::sync::{ + Mutex, + atomic::{AtomicBool, Ordering}, + }; + let armed = Arc::new(AtomicBool::new(false)); + let arm = armed.clone(); + let entered = Arc::new(tokio::sync::Notify::new()); + let signal = entered.clone(); + let (release, receive) = std::sync::mpsc::channel(); + let gate = Mutex::new(Some(receive)); + let fixture = ReaderFixture::with_store( + Store::new(Arc::new(InMemory::new())).with_read_request_observer(Arc::new(move |_| { + if arm.swap(false, Ordering::AcqRel) { + let receiver = gate.lock().unwrap().take().unwrap(); + signal.notify_one(); + let _ = receiver.recv(); + } + })), + ) + .await; + armed.store(true, Ordering::Release); + let manager = fixture.manager.clone(); + let target = fixture.target.clone(); + let opening = tokio::spawn(async move { manager.activate(target, session(0)).await }); + let reached = tokio::time::timeout(Duration::from_secs(3), entered.notified()).await; + if reached.is_err() { + let _ = release.send(()); + panic!("original reader preparation read was not captured"); + } + let page = fixture + .manager + .fleet_reader_enrollments_page(None, 1, clock().unwrap()); + let completion = fixture + .manager + .enrollment_completion(fixture.target.cell_id()) + .await; + let rows = fixture.rows().await; + release.send(()).unwrap(); + let page = page.unwrap().unwrap(); + assert_eq!(page.total_enrollments(), 0); + assert!(page.entries().is_empty()); + assert_eq!(page.jobs().retained(), 1); + assert_eq!(page.jobs().running(), 1); + assert_eq!(page.jobs().unobserved(), 0); + assert!(completion.unwrap().is_none()); + assert!(rows.is_empty()); + drop(page); + opening.await.unwrap().unwrap(); + fixture.read().await; + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/inventory/variable_rows.rs b/crates/cellule-host/minion/scenario/reader_tests/inventory/variable_rows.rs new file mode 100644 index 00000000..7dd0fe48 --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/inventory/variable_rows.rs @@ -0,0 +1,165 @@ +//! Real maximum-width partitions paginate before cloned payloads fill admission. +use super::*; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn byte_budget_yields_a_continuation_over_128_native_long_partition_readers() { + // This fixture explicitly admits 128 native views, with 256 MiB of retained + // byte credit and a four-GiB native ceiling. Each original responsibility still + // reserves 192 KiB. Ordinary cases retain eight slots, 16 MiB of retained + // credit and their existing 128-MiB receiver ceiling. + let fixture = + ReaderFixture::with_capacity(Store::new(Arc::new(InMemory::new())), 128, 4 << 30).await; + fixture + .manager + .activate(fixture.target.clone(), session(0)) + .await + .unwrap(); + let mut handles = Vec::new(); + for key in 2..128 { + fixture.leases.renew().await; + let (target, handle) = fixture.additional_partition(key, 1024).await; + fixture.manager.activate(target, session(0)).await.unwrap(); + handles.push(handle); + } + // A byte-pagination qualification must have admission headroom after the + // real classifier dwell, rather than depending on 128 opens finishing before + // pressure sampling. The two-GiB fixture crossed the 600-permille recovery + // threshold and refused later views on slower machines. + let initial = fixture + .node + .runtime() + .operational_sample() + .unwrap() + .unwrap(); + tokio::time::timeout(Duration::from_secs(3), async { + loop { + fixture.leases.renew().await; + let sample = fixture + .node + .runtime() + .operational_sample() + .unwrap() + .unwrap(); + assert_eq!(sample.pressure, NodePressure::Normal, "{sample:?}"); + if sample.observed_at_ms - initial.observed_at_ms >= 1_500 { + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + let (target, handle) = fixture.additional_partition(128, 1024).await; + handles.push(handle); + // Pause the final original acceptance, so page accounting has a stable owner + // without imposing the capture timeout on a native opening of the last view. + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(false, false); + let manager = fixture.manager.clone(); + let cell = target.cell_id(); + let opening = tokio::spawn(async move { manager.activate(target, session(0)).await }); + let reached = tokio::time::timeout(Duration::from_secs(3), captured).await; + if reached.is_err() { + let _ = resume.send(()); + let result = tokio::time::timeout(Duration::from_secs(3), opening).await; + let progress = fixture.manager.enrollment_completion(cell).await.unwrap(); + let state = progress.map(|r| { + ( + r.opening_started, + r.opening_joined, + r.execution_error, + r.journal_error, + ) + }); + let gauges = ( + fixture.node.stats().retained_bytes(), + fixture.node.stats().resident_bytes(), + fixture.node.stats().local_disk_reserved_bytes(), + ); + fixture.node.shutdown().await.unwrap(); + for handle in handles { + handle.drain().await.unwrap(); + } + fixture.finish().await; + panic!( + "last original acceptance was not reached: result={result:?}, progress={state:?}, gauges={gauges:?}" + ); + } + let rows = fixture.rows().await; + assert_eq!(rows.len(), 128); + let before = fixture.node.stats().retained_bytes(); + let first = capture(&fixture, 128); + assert_eq!(first.total_enrollments(), 128); + assert!(!first.entries().is_empty()); + assert!(first.entries().len() < 128); + let cursor = first.next().unwrap(); + let second = fixture + .manager + .fleet_reader_enrollments_page(Some(cursor), 128, clock().unwrap()) + .unwrap() + .unwrap(); + assert_eq!(second.topology(), first.topology()); + assert!(second.next().is_none()); + let observed = first + .entries() + .iter() + .chain(second.entries()) + .collect::>(); + assert_eq!(observed.len(), 128); + assert!( + observed + .windows(2) + .all(|pair| pair[0].source.description().cell.as_bytes() + < pair[1].source.description().cell.as_bytes()) + ); + for original in &rows { + assert_eq!( + observed + .iter() + .filter(|r| r.spec == *original.spec()) + .count(), + 1 + ); + } + assert_eq!(fixture.node.stats().retained_bytes(), before + (2 << 20)); + drop(observed); + drop(second); + drop(first); + assert_eq!(fixture.node.stats().retained_bytes(), before); + assert_eq!(fixture.rows().await, rows); + resume.send(()).unwrap(); + opening.await.unwrap().unwrap(); + let native = fixture + .manager + .fleet_readers_page(None, 128, clock().unwrap()) + .await + .unwrap(); + assert_eq!(native.total_views(), 128); + assert_eq!(native.entries().len(), 128); + drop(native); + for row in &rows { + fixture.leases.renew().await; + let cellule_runtime::fleet::operations::EnrollmentRole::Reader { target, .. } = + &row.spec().role + else { + panic!("non-reader in native reader fixture"); + }; + let reader = fixture.manager.resolve(target.clone()).await.unwrap(); + assert_eq!( + reader + .query::(Some(reader.receipt().await), 0) + .await + .unwrap() + .output, + 17 + ); + } + fixture.node.shutdown().await.unwrap(); + assert_eq!(fixture.node.stats().resident_bytes(), 0); + assert_eq!(fixture.node.stats().retained_bytes(), 0); + assert_eq!(fixture.node.stats().worker_jobs(), 0); + assert_eq!(fixture.node.stats().local_disk_reserved_bytes(), 0); + for handle in handles { + handle.drain().await.unwrap(); + } + fixture.finish().await; +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/leases.rs b/crates/cellule-host/minion/scenario/reader_tests/leases.rs new file mode 100644 index 00000000..fb4bb876 --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/leases.rs @@ -0,0 +1,154 @@ +//! Caller-driven fixture heartbeats retain exact boot identity and real expiry. +use super::*; +use cellule_runtime::node::{NodeAdvertisement, NodeCapacity, NodeFailureDomain}; +use ed25519_dalek::SigningKey; + +pub(super) struct ReaderLeases { + directory: NodeDirectory, + code: Digest, + release: Digest, + receiver: NodeLeaseGuard, +} + +fn advertisement( + index: usize, + code: Digest, + release: Digest, + now: i64, + lifetime_ms: i64, +) -> NodeAdvertisement { + NodeAdvertisement::sign( + node_id(index), + session(index), + owner(index).endpoint, + scope().fleet, + Digest::from_bytes([30; 32]), + Digest::from_bytes([31; 32]), + release, + &SigningKey::from_bytes(&[index as u8 + 1; 32]), + 1, + now, + now + lifetime_ms, + vec![code], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + free_memory_bytes: 1 << 30, + free_disk_bytes: 1 << 30, + job_credits: 4, + log_protocol: 1, + ..NodeCapacity::default() + }, + ) + .unwrap() +} + +impl ReaderLeases { + pub(super) async fn create( + directory: NodeDirectory, + code: Digest, + release: Digest, + lifetime_ms: i64, + ) -> Self { + let now = clock().unwrap(); + for index in [0, 1] { + directory + .create(advertisement(index, code, release, now, lifetime_ms), now) + .await + .unwrap(); + } + Self { + directory, + code, + release, + receiver: NodeLeaseGuard::new(clock().unwrap(), now + lifetime_ms).unwrap(), + } + } + + pub(super) fn receiver_guard(&self) -> &NodeLeaseGuard { + &self.receiver + } + + pub(super) async fn renew(&self) { + for index in [0, 1] { + let now = clock().unwrap(); + // Load live evidence before refresh. Expiry or withdrawal cannot be + // repaired by recreating an advertisement for this same boot. + let observed = self + .directory + .load(session(index), now) + .await + .unwrap() + .unwrap(); + if observed.advertisement().expires_at_ms() - now > 15_000 { + continue; + } + let next = advertisement(index, self.code, self.release, now, 30_000); + let refreshed = self.directory.refresh(&observed, next, now).await.unwrap(); + if index == 1 { + // Local credit advances only after the authoritative CAS reply. + self.receiver + .renew(clock().unwrap(), refreshed.advertisement().expires_at_ms()) + .unwrap(); + } + } + } + + pub(super) fn fence(&self) { + self.receiver.fence(); + } +} + +#[tokio::test] +async fn reader_fixture_renews_signed_boots_and_guard_past_original_expiry() { + let app = application::compile().unwrap(); + let code = app.registry().module_digests()[0]; + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + ObjectPath::from("reader-heartbeat"), + [3; 16], + ); + let directory = NodeDirectory::new( + layout, + scope().fleet, + Digest::from_bytes([31; 32]), + app.registry().release_digest(), + ); + let leases = ReaderLeases::create( + directory.clone(), + code, + app.registry().release_digest(), + 2_000, + ) + .await; + let original = directory + .load(session(1), clock().unwrap()) + .await + .unwrap() + .unwrap(); + tokio::time::sleep(Duration::from_millis(10)).await; + leases.renew().await; + tokio::time::timeout(Duration::from_secs(3), async { + while clock().unwrap() < original.advertisement().expires_at_ms() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + for index in [0, 1] { + let current = directory + .load(session(index), clock().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(current.advertisement().node(), node_id(index)); + assert_eq!(current.advertisement().session(), session(index)); + assert_eq!(current.advertisement().generation(), 2); + } + leases.receiver_guard().check().unwrap(); + leases.fence(); + assert!(matches!( + leases.receiver_guard().check(), + Err(Error::Fenced) + )); +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/mod.rs b/crates/cellule-host/minion/scenario/reader_tests/mod.rs new file mode 100644 index 00000000..902b3f33 --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/mod.rs @@ -0,0 +1,773 @@ +//! Real canonical reader production against the durable reference transaction domain. +use super::*; +use cellule_host::read_replicas::ReadReplicaManager; +use cellule_runtime::{ + CellRuntime, Error, + fleet::operations::{EnrollmentRecord, EnrollmentStatus}, + node::NodeDirectory, + peer::PeerReplicaResolver, +}; + +mod evacuation; +mod failed; +mod inventory; +mod leases; +mod reconciliation; + +struct ReaderFixture { + root: tempfile::TempDir, + node: Arc, + manager: ReadReplicaManager, + source: CellRuntime, + handle: CellHandle, + target: CellTarget, + journal: Arc, + layout: CellStorageLayout, + limits: Limits, + leases: leases::ReaderLeases, +} +impl ReaderFixture { + async fn new() -> Self { + Self::with_store(Store::new(Arc::new(InMemory::new()))).await + } + async fn with_store(store: Store) -> Self { + Self::with_capacity(store, 8, 128 << 20).await + } + async fn with_capacity(store: Store, active_cells: usize, native_memory_bytes: usize) -> Self { + let root = tempfile::tempdir().unwrap(); + let app = application::compile().unwrap(); + let code = app.registry().module_digests()[0]; + let layout = CellStorageLayout::new(store, ObjectPath::from("enrolled-readers"), [3; 16]); + let directory = NodeDirectory::new( + layout.clone(), + scope().fleet, + Digest::from_bytes([31; 32]), + app.registry().release_digest(), + ); + let now = clock().unwrap(); + let leases = leases::ReaderLeases::create( + directory.clone(), + code, + app.registry().release_digest(), + 30_000, + ) + .await; + let journal = Arc::new( + SqliteJournal::open( + root.path().join("journal.sqlite"), + scope(), + FleetProfile::default(), + now, + ) + .await + .unwrap(), + ); + for index in [0, 1] { + journal + .register_initial_intent( + &NodeIntent::initial(scope(), node_id(index), session(index)).unwrap(), + ) + .await + .unwrap(); + } + let limits = Limits { + max_database_bytes: 64 << 20, + max_capture_bytes: 16 << 20, + ..Limits::default() + }; + let node = Arc::new( + CellNodeBuilder::new(app) + .with_runtime( + SqlWorkerPool::new(2, active_cells) + .unwrap() + .with_native_memory_limit(native_memory_bytes) + .unwrap(), + active_cells << 21, + ) + .with_replica_host(Host::default().with_local_disk_budget(DiskBudget::new(8 << 30))) + .with_session(session(1)) + .build() + .unwrap(), + ); + node.install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + let manager = node + .install_read_replicas( + layout.clone(), + directory, + root.path().join("readers"), + limits, + ) + .unwrap(); + node.install_fleet_reader_enrollment(scope(), node_id(1), journal.clone()) + .unwrap(); + assert!( + node.install_fleet_reader_enrollment(scope(), node_id(1), journal.clone()) + .is_err() + ); + node.install_node_lease(leases.receiver_guard().clone()) + .unwrap(); + let source_workers = SqlWorkerPool::new(2, active_cells).unwrap(); + let source_workers = if active_cells == 8 { + source_workers + } else { + source_workers + .with_native_memory_limit(native_memory_bytes) + .unwrap() + }; + let source = CellRuntime::new_with_replica_host( + source_workers, + 16 << 20, + session(0), + Host::default().with_local_disk_budget(DiskBudget::new(8 << 30)), + ) + .unwrap(); + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + scope().application, + application::NAMESPACE, + &[1], + ) + .unwrap(); + let incarnation = IncarnationId::from_bytes([1; 16]); + let catalog = CellCatalog::new(layout.clone(), target.tenant()); + let proof = catalog + .provision(CatalogEntry::new(&target, CatalogRole::Sql, code, 1).unwrap()) + .await + .unwrap(); + let authority = CellAuthority::new(layout.clone()); + let observed = authority + .create_initial(&proof, incarnation, owner(0)) + .await + .unwrap(); + let replica = CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + limits, + ) + .unwrap(); + let handle = source + .bootstrap( + proof, + replica, + authority, + observed, + root.path().join("source.sqlite"), + |tx| { + tx.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES (17)", + )?; + Ok(()) + }, + ) + .await + .unwrap(); + manager.set_target(&target, 0, 1).await.unwrap().unwrap(); + Self { + root, + node, + manager, + source, + handle, + target, + journal, + layout, + limits, + leases, + } + } + async fn additional_cell(&self, partition: u8) -> (CellTarget, CellHandle) { + self.additional_partition(partition, 1).await + } + async fn additional_partition(&self, partition: u8, width: usize) -> (CellTarget, CellHandle) { + let target = CellTarget::new( + self.target.tenant(), + scope().application, + application::NAMESPACE, + &vec![partition; width], + ) + .unwrap(); + let code = application::compile().unwrap().registry().module_digests()[0]; + let proof = CellCatalog::new(self.layout.clone(), target.tenant()) + .provision(CatalogEntry::new(&target, CatalogRole::Sql, code, 1).unwrap()) + .await + .unwrap(); + let incarnation = IncarnationId::from_bytes([partition; 16]); + let authority = CellAuthority::new(self.layout.clone()); + let observed = authority + .create_initial(&proof, incarnation, owner(0)) + .await + .unwrap(); + let replica = CellReplica::new( + self.layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + self.limits, + ) + .unwrap(); + let handle = self + .source + .bootstrap( + proof, + replica, + authority, + observed, + self.root.path().join(format!("source-{partition}.sqlite")), + |tx| { + tx.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES (17)", + )?; + Ok(()) + }, + ) + .await + .unwrap(); + self.manager + .set_target(&target, 0, 1) + .await + .unwrap() + .unwrap(); + (target, handle) + } + async fn rows(&self) -> Vec { + tokio::time::timeout(Duration::from_secs(3), async { + loop { + let version = self + .journal + .load_snapshot(scope()) + .await + .unwrap() + .registry(); + match self.journal.enrollments_page(version, None, 128).await { + Ok(page) => return page.entries().to_vec(), + Err(error) + if matches!( + error + .downcast_ref::( + ), + Some(cellule_runtime::fleet::operations::OperationError::Conflict) + ) => + { + // The owned producer may publish between the header and + // exact-version page. Retry a fresh consistent scan only. + tokio::task::yield_now().await; + } + Err(error) => panic!("reader registry scan failed: {error}"), + } + } + }) + .await + .unwrap() + } + async fn read(&self) { + let reader = self.manager.resolve(self.target.clone()).await.unwrap(); + let observed = reader + .query::(Some(reader.receipt().await), 0) + .await + .unwrap(); + assert_eq!(observed.output, 17); + } + async fn finish(self) { + self.finish_status(EnrollmentStatus::Retired).await; + } + async fn finish_status(self, expected: EnrollmentStatus) { + self.node.shutdown().await.unwrap(); + assert!(self.rows().await.iter().all(|row| row.status() == expected)); + assert_eq!(self.node.stats().retained_bytes(), 0); + assert_eq!(self.node.stats().local_disk_reserved_bytes(), 0); + self.handle.drain().await.unwrap(); + self.source.shutdown().await.unwrap(); + self.leases.fence(); + assert_eq!(self.source.stats().resident_bytes(), 0); + assert_eq!(self.source.stats().retained_bytes(), 0); + assert_eq!(self.source.stats().worker_jobs(), 0); + assert_eq!(self.source.stats().local_disk_reserved_bytes(), 0); + self.journal.close().await.unwrap(); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reader_producer_journals_before_open_and_owns_a_cancelled_activation() { + let fixture = ReaderFixture::new().await; + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(false, false); + let manager = fixture.manager.clone(); + let target = fixture.target.clone(); + let activation = tokio::spawn(async move { manager.activate(target, session(0)).await }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + let pending = fixture.rows().await; + assert_eq!(pending.len(), 1); + assert_eq!(pending[0].status(), EnrollmentStatus::Pending); + assert_eq!(fixture.node.stats().worker_jobs(), 0); + assert!( + fixture + .manager + .resolve(fixture.target.clone()) + .await + .is_err() + ); + activation.abort(); + assert!(activation.await.unwrap_err().is_cancelled()); + resume.send(()).unwrap(); + tokio::time::timeout(Duration::from_secs(3), async { + loop { + if fixture.rows().await[0].status() == EnrollmentStatus::Established { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + fixture.read().await; + fixture + .manager + .activate(fixture.target.clone(), session(0)) + .await + .unwrap(); + let established = fixture.rows().await; + assert_eq!(established.len(), 1); + assert_eq!(established[0].spec(), pending[0].spec()); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reader_producer_lost_acceptance_cannot_repeat_open_and_is_refused_by_joined_removal() { + let fixture = ReaderFixture::new().await; + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(false, true); + let manager = fixture.manager.clone(); + let target = fixture.target.clone(); + let activation = tokio::spawn(async move { manager.activate(target, session(0)).await }); + captured.await.unwrap(); + resume.send(()).unwrap(); + assert!(activation.await.unwrap().is_err()); + let pending = fixture.rows().await; + assert_eq!(pending[0].status(), EnrollmentStatus::Pending); + assert!( + fixture + .manager + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .unwrap() + .accepted + .is_none() + ); + assert!( + fixture + .manager + .activate(fixture.target.clone(), session(0)) + .await + .is_err() + ); + assert_eq!(fixture.rows().await, pending); + assert!( + fixture + .manager + .resolve(fixture.target.clone()) + .await + .is_err() + ); + fixture + .manager + .remove(fixture.target.cell_id()) + .await + .unwrap(); + assert_eq!(fixture.rows().await[0].status(), EnrollmentStatus::Refused); + assert!( + fixture + .manager + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .is_none() + ); + fixture.finish_status(EnrollmentStatus::Refused).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reader_producer_replays_original_establishment_and_retirement_after_lost_replies() { + let fixture = ReaderFixture::new().await; + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(true, true); + let manager = fixture.manager.clone(); + let target = fixture.target.clone(); + let activation = tokio::spawn(async move { manager.activate(target, session(0)).await }); + captured.await.unwrap(); + let original = fixture.rows().await[0].clone(); + assert_eq!(original.status(), EnrollmentStatus::Established); + resume.send(()).unwrap(); + assert!(activation.await.unwrap().is_err()); + let completion = fixture + .manager + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert!(!completion.published); + assert!(completion.journal_error.is_some()); + fixture + .manager + .activate(fixture.target.clone(), session(0)) + .await + .unwrap(); + assert_eq!(fixture.rows().await, vec![original.clone()]); + fixture.read().await; + let peer = fixture + .manager + .resolve(fixture.target.clone()) + .await + .unwrap(); + fixture.journal.lose_next_commit_reply(); + assert!( + fixture + .manager + .remove(fixture.target.cell_id()) + .await + .is_err() + ); + let retired = fixture.rows().await[0].clone(); + assert_eq!(retired.status(), EnrollmentStatus::Retired); + assert_eq!( + retired.established_evidence(), + original.established_evidence() + ); + assert!(matches!( + peer.query::(None, 0).await, + Err(Error::Fenced) + )); + assert_eq!( + fixture + .manager + .fleet_readers_page(None, 128, clock().unwrap()) + .await + .unwrap() + .entries() + .len(), + 1 + ); + fixture + .manager + .remove(fixture.target.cell_id()) + .await + .unwrap(); + assert_eq!(fixture.rows().await, vec![retired.clone()]); + let restarted = SqliteJournal::open( + fixture.root.path().join("journal.sqlite"), + scope(), + FleetProfile::default(), + clock().unwrap(), + ) + .await + .unwrap(); + assert_eq!( + restarted + .load_enrollment(scope(), retired.spec().key().unwrap()) + .await + .unwrap(), + Some(retired) + ); + restarted.close().await.unwrap(); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reader_producer_owns_native_open_through_cancelled_waiter_and_shutdown_deadline() { + use std::sync::{ + Mutex, + atomic::{AtomicBool, Ordering}, + }; + let runtime = Arc::new(Mutex::new(None::)); + let observed = runtime.clone(); + let entered = Arc::new(tokio::sync::Notify::new()); + let signal = entered.clone(); + let armed = Arc::new(AtomicBool::new(false)); + let once = armed.clone(); + let (release, receive) = std::sync::mpsc::channel(); + let gate = Mutex::new(Some(receive)); + let fixture = ReaderFixture::with_store( + Store::new(Arc::new(InMemory::new())).with_read_request_observer(Arc::new(move |kind| { + if kind == cellule_store::StorageReadKind::Range + && observed + .lock() + .unwrap() + .as_ref() + .is_some_and(|runtime| runtime.stats().worker_jobs() == 1) + && !once.swap(true, Ordering::AcqRel) + { + let receiver = gate.lock().unwrap().take().unwrap(); + signal.notify_one(); + let _ = receiver.recv(); + } + })), + ) + .await; + *runtime.lock().unwrap() = Some(fixture.node.runtime().clone()); + let manager = fixture.manager.clone(); + let target = fixture.target.clone(); + let activation = tokio::spawn(async move { manager.activate(target, session(0)).await }); + let captured = tokio::time::timeout(Duration::from_secs(3), entered.notified()).await; + if captured.is_err() { + let _ = release.send(()); + let result = activation.await.unwrap(); + panic!("reader native VFS pause was not reached: {result:?}"); + } + let pending = fixture.rows().await; + assert_eq!(pending[0].status(), EnrollmentStatus::Pending); + assert_eq!(fixture.node.stats().worker_jobs(), 1); + let page = fixture + .manager + .fleet_reader_enrollments_page(None, 128, clock().unwrap()) + .unwrap() + .unwrap(); + assert_eq!(page.entries()[0].spec, *pending[0].spec()); + assert_eq!(page.entries()[0].accepted.as_ref().unwrap(), &pending[0]); + assert!(page.entries()[0].opening_started); + assert!(!page.entries()[0].opening_joined); + assert!(page.entries()[0].event.is_none()); + assert_eq!(page.jobs().running(), 1); + drop(page); + activation.abort(); + assert!(activation.await.unwrap_err().is_cancelled()); + let drained = fixture + .node + .shutdown_until(std::time::Instant::now() + Duration::from_millis(30)) + .await; + let before = fixture.node.stats(); + let state = fixture.node.state(); + let _ = release.send(()); + fixture.node.shutdown().await.unwrap(); + runtime.lock().unwrap().take(); + assert!(drained.is_err()); + assert_eq!(state, NodeState::Draining); + assert_eq!(before.worker_jobs(), 1); + assert_eq!(fixture.rows().await[0].status(), EnrollmentStatus::Retired); + assert!( + fixture + .manager + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .is_none() + ); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reader_producer_cancelled_removal_retains_fenced_inventory_and_original_retirement() { + let fixture = ReaderFixture::new().await; + fixture + .manager + .activate(fixture.target.clone(), session(0)) + .await + .unwrap(); + let peer = fixture + .manager + .resolve(fixture.target.clone()) + .await + .unwrap(); + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(true, false); + let manager = fixture.manager.clone(); + let cell = fixture.target.cell_id(); + let removal = tokio::spawn(async move { manager.remove(cell).await }); + captured.await.unwrap(); + let original = fixture.rows().await[0].clone(); + assert_eq!(original.status(), EnrollmentStatus::Retired); + removal.abort(); + assert!(removal.await.unwrap_err().is_cancelled()); + let _ = resume.send(()); + assert!(matches!( + peer.query::(None, 0).await, + Err(Error::Fenced) + )); + assert_eq!(fixture.node.stats().worker_jobs(), 0); + assert_eq!(fixture.node.stats().local_disk_reserved_bytes(), 0); + assert_eq!( + fixture + .manager + .fleet_readers_page(None, 128, clock().unwrap()) + .await + .unwrap() + .total_views(), + 1 + ); + let completion = fixture + .manager + .enrollment_completion(cell) + .await + .unwrap() + .unwrap(); + assert!(!completion.published); + fixture.manager.remove(cell).await.unwrap(); + assert_eq!(fixture.rows().await, vec![original]); + assert_eq!( + fixture + .manager + .fleet_readers_page(None, 128, clock().unwrap()) + .await + .unwrap() + .total_views(), + 0 + ); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reader_producer_cordon_refusal_preserves_native_and_lost_retirement_errors_separately() { + let fixture = ReaderFixture::new().await; + let (accepted, resume_acceptance) = fixture.journal.pause_next_enrollment_reply(false, false); + let manager = fixture.manager.clone(); + let target = fixture.target.clone(); + let activation = tokio::spawn(async move { manager.activate(target, session(0)).await }); + accepted.await.unwrap(); + assert_eq!(fixture.rows().await[0].status(), EnrollmentStatus::Pending); + let (published, resume_publication) = fixture.journal.pause_next_enrollment_reply(true, true); + fixture.node.runtime().node_admission().cordon().unwrap(); + resume_acceptance.send(()).unwrap(); + published.await.unwrap(); + assert_eq!(fixture.rows().await[0].status(), EnrollmentStatus::Retired); + let page = fixture + .manager + .fleet_reader_enrollments_page(None, 128, clock().unwrap()) + .unwrap() + .unwrap(); + assert!(page.entries()[0].opening_started && page.entries()[0].opening_joined); + assert!(matches!( + page.entries()[0].execution_error.as_deref(), + Some(Error::CellDraining) + )); + assert!(matches!( + page.entries()[0].event, + Some(cellule_runtime::fleet::operations::EnrollmentEvent::Retired(_)) + )); + assert!(!page.entries()[0].published); + assert_eq!(page.jobs().running(), 1); + drop(page); + resume_publication.send(()).unwrap(); + assert!(activation.await.unwrap().is_err()); + let completion = fixture + .manager + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert!(matches!( + completion.execution_error.as_deref(), + Some(Error::CellDraining) + )); + assert!(completion.journal_error.is_some()); + assert!(!completion.published); + assert_eq!(fixture.node.stats().worker_jobs(), 0); + assert!( + fixture + .manager + .resolve(fixture.target.clone()) + .await + .is_err() + ); + fixture + .manager + .remove(fixture.target.cell_id()) + .await + .unwrap(); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reader_producer_atomic_cordon_refusal_closes_without_creating_a_retirement_request() { + use cellule_runtime::fleet::operations::{ + JournalTransition, MaintenanceOperation, OperationId, + }; + let fixture = ReaderFixture::new().await; + let (captured, resume) = fixture.journal.pause_before_enrollment_acceptance(); + let manager = fixture.manager.clone(); + let target = fixture.target.clone(); + let activation = tokio::spawn(async move { manager.activate(target, session(0)).await }); + captured.await.unwrap(); + let page = fixture + .manager + .fleet_reader_enrollments_page(None, 128, clock().unwrap()) + .unwrap() + .unwrap(); + assert_eq!(page.total_enrollments(), 1); + assert!(page.entries()[0].accepted.is_none()); + assert!(!page.entries()[0].opening_started); + assert_eq!(page.jobs().running(), 1); + assert!(fixture.rows().await.is_empty()); + drop(page); + let old = fixture.journal.load_snapshot(scope()).await.unwrap(); + let now = clock().unwrap(); + let controller = fixture + .journal + .claim_controller( + scope(), + old.head().revision(), + SessionId::from_bytes([202; 16]), + now, + ) + .await + .unwrap(); + fixture + .journal + .compare_exchange( + &controller, + controller.head().controller().unwrap().epoch, + now, + &JournalTransition::BeginMaintenance( + MaintenanceOperation::new( + OperationId::from_bytes([203; 16]).unwrap(), + Digest::from_bytes([204; 32]), + node_id(1), + session(1), + 2, + now, + now + 60_000, + ) + .unwrap(), + ), + ) + .await + .unwrap(); + fixture.node.runtime().node_admission().cordon().unwrap(); + resume.send(()).unwrap(); + assert!(activation.await.unwrap().is_err()); + assert!(fixture.rows().await.is_empty()); + assert!( + fixture + .manager + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .unwrap() + .accepted + .is_none() + ); + fixture + .manager + .remove(fixture.target.cell_id()) + .await + .unwrap(); + let exclusion = fixture.rows().await; + assert_eq!(exclusion.len(), 1); + assert_eq!(exclusion[0].status(), EnrollmentStatus::Refused); + let delayed = fixture + .journal + .accept_enrollment(exclusion[0].spec(), clock().unwrap()) + .await + .unwrap(); + assert!( + matches!(delayed, cellule_host::fleet::FleetEnrollmentAcceptance::Existing(row) + if row == exclusion[0]) + ); + assert!( + fixture + .manager + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .is_none() + ); + fixture.finish_status(EnrollmentStatus::Refused).await; +} diff --git a/crates/cellule-host/minion/scenario/reader_tests/reconciliation.rs b/crates/cellule-host/minion/scenario/reader_tests/reconciliation.rs new file mode 100644 index 00000000..4793040b --- /dev/null +++ b/crates/cellule-host/minion/scenario/reader_tests/reconciliation.rs @@ -0,0 +1,278 @@ +//! The installed manager repairs original obligations without another hint or drain. +use super::*; +use cellule_host::fleet::FleetEnrollmentAcceptance; + +async fn wait_for_repair(fixture: &ReaderFixture, cell: CellId, removed: bool) { + tokio::time::timeout(Duration::from_secs(12), async { + loop { + let completion = fixture.manager.enrollment_completion(cell).await.unwrap(); + if (removed && completion.is_none()) + || (!removed && completion.is_some_and(|c| c.published)) + { + let page = fixture + .manager + .fleet_reader_enrollments_page(None, 128, clock().unwrap()) + .unwrap() + .unwrap(); + if page.jobs().retained() == 0 { + return; + } + } + fixture.leases.renew().await; + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn periodic_reconciliation_fences_lost_acceptance_without_reopening_or_shutdown() { + let fixture = ReaderFixture::new().await; + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(false, true); + let manager = fixture.manager.clone(); + let target = fixture.target.clone(); + let opening = tokio::spawn(async move { manager.activate(target, session(0)).await }); + captured.await.unwrap(); + let original = fixture.rows().await[0].clone(); + assert_eq!(original.status(), EnrollmentStatus::Pending); + resume.send(()).unwrap(); + assert!(opening.await.unwrap().is_err()); + wait_for_repair(&fixture, fixture.target.cell_id(), true).await; + let exclusion = fixture.rows().await[0].clone(); + assert_eq!(exclusion.spec(), original.spec()); + assert_eq!(exclusion.accepted_at_ms(), original.accepted_at_ms()); + assert_eq!(exclusion.status(), EnrollmentStatus::Refused); + assert!(exclusion.established_evidence().is_none()); + assert_eq!(fixture.node.stats().worker_jobs(), 0); + assert_eq!(fixture.node.stats().local_disk_reserved_bytes(), 0); + assert!( + fixture + .manager + .resolve(fixture.target.clone()) + .await + .is_err() + ); + let delayed = fixture + .journal + .accept_enrollment(original.spec(), clock().unwrap()) + .await + .unwrap(); + assert!(matches!(delayed, FleetEnrollmentAcceptance::Existing(row) if row == exclusion)); + fixture.finish_status(EnrollmentStatus::Refused).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn periodic_reconciliation_fences_lost_acceptance_after_local_lease_fencing() { + let fixture = ReaderFixture::new().await; + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(false, true); + let manager = fixture.manager.clone(); + let target = fixture.target.clone(); + let opening = tokio::spawn(async move { manager.activate(target, session(0)).await }); + captured.await.unwrap(); + let original = fixture.rows().await[0].clone(); + resume.send(()).unwrap(); + assert!(opening.await.unwrap().is_err()); + fixture.leases.fence(); + assert!(matches!( + fixture.leases.receiver_guard().check(), + Err(Error::Fenced) + )); + // Observe original local progress directly: native inventory/read admission + // must remain fenced. No new hint, lease renewal, removal or shutdown may + // supply the periodic producer's nonexecution proof. + let repaired = tokio::time::timeout(Duration::from_secs(12), async { + loop { + if fixture + .manager + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .is_none() + { + return; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await; + let exclusion = fixture.rows().await[0].clone(); + let inventory = fixture + .manager + .fleet_reader_enrollments_page(None, 128, clock().unwrap()); + let activation = fixture + .manager + .activate(fixture.target.clone(), session(0)) + .await; + let before = fixture.node.stats(); + let shutdown = fixture.node.shutdown().await; + fixture.handle.drain().await.unwrap(); + fixture.source.shutdown().await.unwrap(); + fixture.journal.close().await.unwrap(); + assert!( + repaired.is_ok(), + "fenced producer did not reconcile its original request" + ); + shutdown.unwrap(); + assert_eq!(exclusion.spec(), original.spec()); + assert_eq!(exclusion.accepted_at_ms(), original.accepted_at_ms()); + assert_eq!(exclusion.status(), EnrollmentStatus::Refused); + assert!(exclusion.established_evidence().is_none()); + assert!(matches!(inventory, Err(Error::Fenced))); + assert!(activation.is_err()); + assert_eq!(before.worker_jobs(), 0); + assert_eq!(before.local_disk_reserved_bytes(), 0); + assert_eq!(fixture.node.stats().retained_bytes(), 0); + assert_eq!(fixture.source.stats().retained_bytes(), 0); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn periodic_reconciliation_republishes_original_opening_and_preserves_its_error() { + let fixture = ReaderFixture::new().await; + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(true, true); + let manager = fixture.manager.clone(); + let target = fixture.target.clone(); + let opening = tokio::spawn(async move { manager.activate(target, session(0)).await }); + captured.await.unwrap(); + let original = fixture.rows().await[0].clone(); + resume.send(()).unwrap(); + assert!(opening.await.unwrap().is_err()); + let retained = fixture + .manager + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert!(!retained.published); + assert!(retained.opening_started && retained.opening_joined); + wait_for_repair(&fixture, fixture.target.cell_id(), false).await; + let repaired = fixture + .manager + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(repaired.spec, retained.spec); + assert_eq!(repaired.accepted, retained.accepted); + assert_eq!(repaired.event, retained.event); + assert!(Arc::ptr_eq( + repaired.journal_error.as_ref().unwrap(), + retained.journal_error.as_ref().unwrap(), + )); + assert_eq!(fixture.rows().await, vec![original]); + fixture.read().await; + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn periodic_reconciliation_resumes_cancelled_removal_without_restamping_retirement() { + let fixture = ReaderFixture::new().await; + fixture + .manager + .activate(fixture.target.clone(), session(0)) + .await + .unwrap(); + let peer = fixture + .manager + .resolve(fixture.target.clone()) + .await + .unwrap(); + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(true, false); + let manager = fixture.manager.clone(); + let cell = fixture.target.cell_id(); + let removal = tokio::spawn(async move { manager.remove(cell).await }); + captured.await.unwrap(); + let original = fixture.rows().await[0].clone(); + assert_eq!(original.status(), EnrollmentStatus::Retired); + removal.abort(); + assert!(removal.await.unwrap_err().is_cancelled()); + let _ = resume.send(()); + wait_for_repair(&fixture, cell, true).await; + let page = fixture + .manager + .fleet_readers_page(None, 128, clock().unwrap()) + .await + .unwrap(); + assert_eq!(page.total_views(), 0); + drop(page); + assert_eq!(fixture.rows().await, vec![original]); + assert!(peer.lifecycle_observation().await.locally_joined()); + assert!(matches!( + peer.query::(None, 0).await, + Err(Error::Fenced) + )); + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn periodic_reconciliation_cannot_refuse_a_still_owned_pending_opening() { + let fixture = ReaderFixture::new().await; + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(false, false); + let manager = fixture.manager.clone(); + let target = fixture.target.clone(); + let opening = tokio::spawn(async move { manager.activate(target, session(0)).await }); + captured.await.unwrap(); + let original = fixture.rows().await[0].clone(); + opening.abort(); + assert!(opening.await.unwrap_err().is_cancelled()); + // Let the real five-second reconciliation tick encounter this original + // record. Its lane is held by the retained acceptance owner, not the waiter. + tokio::time::sleep(Duration::from_millis(5_100)).await; + assert_eq!(fixture.rows().await, vec![original.clone()]); + let page = fixture + .manager + .fleet_reader_enrollments_page(None, 128, clock().unwrap()) + .unwrap() + .unwrap(); + assert_eq!(page.jobs().running(), 1); + assert_eq!(page.entries()[0].spec, *original.spec()); + assert!(!page.entries()[0].opening_started); + assert!(page.entries()[0].event.is_none()); + drop(page); + resume.send(()).unwrap(); + wait_for_repair(&fixture, fixture.target.cell_id(), false).await; + let established = fixture.rows().await[0].clone(); + assert_eq!(established.spec(), original.spec()); + assert_eq!(established.accepted_at_ms(), original.accepted_at_ms()); + assert_eq!(established.status(), EnrollmentStatus::Established); + fixture.read().await; + fixture.finish().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn periodic_reconciliation_retains_requests_when_inventory_credit_is_refused() { + let fixture = ReaderFixture::new().await; + let (captured, resume) = fixture.journal.pause_next_enrollment_reply(false, true); + let manager = fixture.manager.clone(); + let target = fixture.target.clone(); + let opening = tokio::spawn(async move { manager.activate(target, session(0)).await }); + captured.await.unwrap(); + let original = fixture.rows().await[0].clone(); + resume.send(()).unwrap(); + assert!(opening.await.unwrap().is_err()); + // Activation joins its exact finite task before returning. Charge all + // remaining credit; its epilogue cannot later reopen scan admission. + let stats = fixture.node.stats(); + let held = fixture + .node + .runtime() + .try_reserve_node_bytes(stats.retained_capacity_bytes() - stats.retained_bytes() - 128) + .unwrap(); + tokio::time::sleep(Duration::from_millis(5_100)).await; + assert_eq!(fixture.rows().await, vec![original.clone()]); + assert!( + fixture + .manager + .enrollment_completion(fixture.target.cell_id()) + .await + .unwrap() + .is_some() + ); + drop(held); + wait_for_repair(&fixture, fixture.target.cell_id(), true).await; + let exclusion = fixture.rows().await[0].clone(); + assert_eq!(exclusion.spec(), original.spec()); + assert_eq!(exclusion.accepted_at_ms(), original.accepted_at_ms()); + assert_eq!(exclusion.status(), EnrollmentStatus::Refused); + fixture.finish_status(EnrollmentStatus::Refused).await; +} diff --git a/crates/cellule-host/minion/scenario/recovered_followers/failed_boot/mod.rs b/crates/cellule-host/minion/scenario/recovered_followers/failed_boot/mod.rs new file mode 100644 index 00000000..6945c1af --- /dev/null +++ b/crates/cellule-host/minion/scenario/recovered_followers/failed_boot/mod.rs @@ -0,0 +1,153 @@ +//! Provider-bound process evidence; this child is a lifetime stand-in, not a +//! multi-process Cell/provider qualification or an acknowledged-tail workload. +use super::*; +use cellule_host::fleet::{ + FleetAdapterFuture, FleetFailedBootProcessEvidence, FleetFailedBootProcessRequest, + FleetFailedBootProcesses, FleetFailedBootRetirement, +}; +use std::{ + process::{Child, Command}, + sync::Mutex, +}; + +mod process_tests; +mod tests; +mod writer_tests; + +pub(super) struct Process { + child: Child, + evidence_path: PathBuf, +} +impl Process { + pub(super) fn start(path: PathBuf) -> Self { + Self { + child: Command::new("sleep").arg("60").spawn().unwrap(), + evidence_path: path, + } + } + pub(super) fn stop_and_retain(&mut self, request: &FleetFailedBootProcessRequest) { + assert!(self.child.try_wait().unwrap().is_none()); + self.child.kill().unwrap(); + let status = self.child.wait().unwrap(); + let mut hash = blake3::Hasher::new(); + hash.update(request.digest().as_bytes()); + hash.update(&self.child.id().to_be_bytes()); + hash.update(status.to_string().as_bytes()); + let evidence = FleetFailedBootProcessEvidence::new( + request, + Digest::from_bytes(*hash.finalize().as_bytes()), + ) + .unwrap(); + let mut bytes = evidence.request_digest().as_bytes().to_vec(); + bytes.extend_from_slice(evidence.witness().as_bytes()); + use std::io::Write; + let mut file = std::fs::File::create(&self.evidence_path).unwrap(); + file.write_all(&bytes).unwrap(); + file.sync_all().unwrap(); + } +} +impl Drop for Process { + fn drop(&mut self) { + let _ = self.child.kill(); + let _ = self.child.wait(); + } +} + +pub(super) struct Processes { + path: PathBuf, + reads: AtomicUsize, + final_fault: Mutex>, + final_witness: Mutex>, +} +impl Processes { + pub(super) fn new(path: PathBuf) -> Self { + Self { + path, + reads: AtomicUsize::new(0), + final_fault: Mutex::new(None), + final_witness: Mutex::new(None), + } + } +} +impl FleetFailedBootProcesses for Processes { + fn confirm_stopped<'a>( + &'a self, + request: &'a FleetFailedBootProcessRequest, + ) -> FleetAdapterFuture<'a, FleetFailedBootProcessEvidence> { + Box::pin(async move { + let read = self.reads.fetch_add(1, Ordering::AcqRel); + if read == 1 { + if let Some(kind) = self.final_fault.lock().unwrap().take() { + return Err( + std::io::Error::new(kind, "original process evidence read lost").into(), + ); + } + if let Some(witness) = self.final_witness.lock().unwrap().take() { + return Ok(FleetFailedBootProcessEvidence::new(request, witness)?); + } + } + let bytes = std::fs::read(&self.path)?; + if bytes.len() != 64 || bytes[..32] != *request.digest().as_bytes() { + return Err( + std::io::Error::other("original process evidence request differs").into(), + ); + } + let witness = Digest::from_bytes(bytes[32..].try_into()?); + Ok(FleetFailedBootProcessEvidence::new(request, witness)?) + }) + } +} + +struct ForeignEvidence(FleetFailedBootProcessEvidence); +impl FleetFailedBootProcesses for ForeignEvidence { + fn confirm_stopped<'a>( + &'a self, + _: &'a FleetFailedBootProcessRequest, + ) -> FleetAdapterFuture<'a, FleetFailedBootProcessEvidence> { + Box::pin(async { Ok(self.0.clone()) }) + } +} + +impl Fixture { + async fn failed_boot(&self) -> EnrollmentRecord { + let spec = + startup::spec(&NodeIntent::initial(scope(), node_id(0), session(0)).unwrap()).unwrap(); + self.journal + .load_enrollment(scope(), spec.key().unwrap()) + .await + .unwrap() + .unwrap() + } + async fn settle_followers(&self) { + self.retire().await; + self.capture() + .await + .publish( + self.journal.as_ref(), + &self.directory, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .unwrap() + .confirmed() + .unwrap(); + } + async fn failed_capture(&self, original: &EnrollmentRecord) -> FleetFailedBootRetirement { + FleetFailedBootRetirement::capture( + self.journal.as_ref(), + &self.directory, + &self.roster().await, + original, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .unwrap() + } + fn process_path(&self) -> PathBuf { + self._root.path().join("process-closure") + } +} diff --git a/crates/cellule-host/minion/scenario/recovered_followers/failed_boot/process_tests.rs b/crates/cellule-host/minion/scenario/recovered_followers/failed_boot/process_tests.rs new file mode 100644 index 00000000..2063e4a1 --- /dev/null +++ b/crates/cellule-host/minion/scenario/recovered_followers/failed_boot/process_tests.rs @@ -0,0 +1,435 @@ +//! Original child lifetime evidence through recovery stages. This uses no Cell +//! suffix or external jobs and does not qualify an OS-crashed CellNode provider. +use super::*; +use tokio::sync::oneshot; + +async fn fenced_request(fixture: &Fixture) -> FleetFailedBootProcessRequest { + FleetFailedBootProcessRequest::capture_fenced( + fixture.journal.as_ref(), + &fixture.directory, + &fixture.roster().await, + &fixture.failed_boot().await, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .unwrap() +} + +#[tokio::test] +async fn original_process_can_join_before_recovery_and_retire_with_the_same_identity() { + let fixture = Fixture::with_process_observation(true).await; + let request = fixture.process_request.as_ref().unwrap(); + let original_interval = request.interval(); + let original = fixture.failed_boot().await; + assert!(request.canonical().is_none()); + assert_eq!(fenced_request(&fixture).await.digest(), request.digest()); + assert!( + FleetFailedBootRetirement::capture_retained( + fixture.journal.as_ref(), + &fixture.directory, + &fixture.roster().await, + request, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .is_err() + ); + fixture.settle_followers().await; + let recaptured = fenced_request(&fixture).await; + assert_eq!(recaptured.digest(), request.digest()); + assert_eq!(recaptured.fence(), request.fence()); + let legacy = fixture.failed_capture(&original).await; + assert!(legacy.request().canonical().is_some()); + assert_ne!(legacy.request().digest(), request.digest()); + let processes = Processes::new(fixture.process_path()); + let capture = FleetFailedBootRetirement::capture_retained( + fixture.journal.as_ref(), + &fixture.directory, + &fixture.roster().await, + request, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .unwrap(); + assert_eq!(capture.request(), request); + assert_eq!(capture.request().interval(), original_interval); + assert_ne!(capture.snapshot(), request.snapshot()); + let publication = capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .unwrap(); + let closure = publication.confirmed().unwrap(); + assert_eq!(closure.process().request_digest(), request.digest()); + assert_eq!(closure.canonical().fence(), *request.fence()); + assert_eq!( + closure.canonical().log().unwrap().phase(), + cellule_runtime::node::log_state::NodeLogPhase::Retired + ); + let retired = closure.boot().clone(); + let closure_digest = closure.digest(); + fixture.journal.close().await.unwrap(); + let independent = fixture.reconstruct().await; + let snapshot = independent.load_snapshot(scope()).await.unwrap(); + let roster = FleetRoster::collect(&independent, &snapshot, deadline()) + .await + .unwrap(); + let recaptured = FleetFailedBootProcessRequest::capture_fenced( + &independent, + &fixture.directory, + &roster, + &original, + session(1), + deadline(), + || Ok(CHECK + 10), + ) + .await + .unwrap(); + assert_eq!(recaptured.digest(), request.digest()); + let confirmation = recaptured + .confirm( + &independent, + &fixture.directory, + &Processes::new(fixture.process_path()), + session(1), + deadline(), + || Ok(CHECK + 10), + ) + .await + .unwrap(); + assert_eq!(confirmation.process(), closure.process()); + let replay = FleetFailedBootRetirement::capture_retained( + &independent, + &fixture.directory, + &roster, + &recaptured, + session(1), + deadline(), + || Ok(CHECK + 10), + ) + .await + .unwrap(); + let repeated = replay + .publish( + &independent, + &fixture.directory, + &Processes::new(fixture.process_path()), + session(1), + deadline(), + || Ok(CHECK + 10), + ) + .await + .unwrap(); + assert_eq!(repeated.confirmed().unwrap().boot(), &retired); + assert_eq!(repeated.confirmed().unwrap().digest(), closure_digest); + assert_eq!(fixture.transport.retirements.load(Ordering::Acquire), 2); + independent.close().await.unwrap(); +} + +#[tokio::test] +async fn a_permanent_fence_and_sealed_log_do_not_prove_original_process_termination() { + let fixture = Fixture::new().await; + let request = fenced_request(&fixture).await; + let process = Process::start(fixture.process_path()); + let mut child = process; + let result = request + .confirm( + fixture.journal.as_ref(), + &fixture.directory, + &Processes::new(fixture.process_path()), + session(1), + deadline(), + || Ok(CHECK), + ) + .await; + assert!(matches!(result, Err(Error::Facility { .. }))); + assert!(child.child.try_wait().unwrap().is_none()); + assert_eq!(fixture.failed_boot().await, *request.boot()); + assert_eq!(fixture.transport.retirements.load(Ordering::Acquire), 0); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn fenced_process_confirmation_preserves_provider_error_and_rejects_changed_witness() { + let fixture = Fixture::with_process_observation(true).await; + let request = fixture.process_request.as_ref().unwrap(); + let processes = Processes::new(fixture.process_path()); + *processes.final_fault.lock().unwrap() = Some(std::io::ErrorKind::ConnectionReset); + let error = request + .confirm( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .err() + .unwrap(); + let Error::Facility { source, .. } = error else { + panic!("provider source error required") + }; + assert_eq!( + source.downcast_ref::().unwrap().kind(), + std::io::ErrorKind::ConnectionReset + ); + let processes = Processes::new(fixture.process_path()); + *processes.final_witness.lock().unwrap() = Some(Digest::from_bytes([201; 32])); + assert!(matches!( + request + .confirm( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK) + ) + .await, + Err(Error::Control("original failed process evidence changed")) + )); + assert_eq!(fixture.failed_boot().await, *request.boot()); + assert_eq!(fixture.transport.retirements.load(Ordering::Acquire), 0); + fixture.journal.close().await.unwrap(); +} + +struct PausedProcesses { + processes: Processes, + pause: Mutex, oneshot::Receiver<()>)>>, +} +impl FleetFailedBootProcesses for PausedProcesses { + fn confirm_stopped<'a>( + &'a self, + request: &'a FleetFailedBootProcessRequest, + ) -> FleetAdapterFuture<'a, FleetFailedBootProcessEvidence> { + Box::pin(async move { + let evidence = self.processes.confirm_stopped(request).await?; + let pause = self.pause.lock().unwrap().take(); + if let Some((entered, resume)) = pause { + let _ = entered.send(()); + resume.await?; + } + Ok(evidence) + }) + } +} + +#[tokio::test] +async fn fenced_process_confirmation_rechecks_registry_after_a_suspended_provider() { + let fixture = Fixture::with_process_observation(true).await; + let request = fixture.process_request.as_ref().unwrap(); + let (entered, waiting) = oneshot::channel(); + let (resume, paused) = oneshot::channel(); + let processes = PausedProcesses { + processes: Processes::new(fixture.process_path()), + pause: Mutex::new(Some((entered, paused))), + }; + let mut work = Box::pin(request.confirm( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK), + )); + tokio::select! { result = &mut work => panic!("unexpected completion {}", result.is_ok()), result = waiting => result.unwrap() } + fixture + .journal + .publish_enrollment_result( + &fixture.originals[1], + EnrollmentEvent::Established(Digest::from_bytes([212; 32])), + CHECK, + ) + .await + .unwrap(); + resume.send(()).unwrap(); + assert!(matches!(work.await, Err(Error::FleetOperation(source)) + if matches!(source.as_ref(), cellule_runtime::fleet::operations::OperationError::Conflict))); + assert_eq!(processes.processes.reads.load(Ordering::Acquire), 1); + assert_eq!(fixture.failed_boot().await, *request.boot()); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn fenced_process_identity_survives_native_log_progress_during_confirmation() { + let fixture = Fixture::with_process_observation(true).await; + let request = fixture.process_request.as_ref().unwrap(); + let (entered, waiting) = oneshot::channel(); + let (resume, paused) = oneshot::channel(); + let processes = PausedProcesses { + processes: Processes::new(fixture.process_path()), + pause: Mutex::new(Some((entered, paused))), + }; + let original = fixture.roster().await; + let mut work = Box::pin(request.confirm( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK), + )); + tokio::select! { result = &mut work => panic!("unexpected completion {}", result.is_ok()), result = waiting => result.unwrap() } + fixture.retire().await; + resume.send(()).unwrap(); + let confirmed = work.await.unwrap(); + assert_eq!(confirmed.snapshot(), original.snapshot()); + assert_eq!(confirmed.fence(), request.fence()); + assert_eq!(confirmed.process().request_digest(), request.digest()); + // Native log closure cannot settle the original Pending/Established journal + // rows; a process confirmation cannot bypass that separate boot barrier. + assert!( + FleetFailedBootRetirement::capture_retained( + fixture.journal.as_ref(), + &fixture.directory, + &fixture.roster().await, + request, + session(1), + deadline(), + || Ok(CHECK) + ) + .await + .is_err() + ); + assert_eq!(fixture.failed_boot().await, *request.boot()); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn fenced_process_confirmation_rejects_foreign_evidence_and_regressing_clocks() { + let fixture = Fixture::with_process_observation(true).await; + let request = fixture.process_request.as_ref().unwrap(); + fixture.settle_followers().await; + let terminal = fixture.failed_capture(request.boot()).await; + let foreign = ForeignEvidence( + FleetFailedBootProcessEvidence::new(terminal.request(), Digest::from_bytes([213; 32])) + .unwrap(), + ); + assert!(matches!( + request + .confirm( + fixture.journal.as_ref(), + &fixture.directory, + &foreign, + session(1), + deadline(), + || Ok(CHECK) + ) + .await, + Err(Error::Fenced) + )); + let mut times = [CHECK, CHECK + 1, CHECK].into_iter(); + assert!(matches!( + request + .confirm( + fixture.journal.as_ref(), + &fixture.directory, + &Processes::new(fixture.process_path()), + session(1), + deadline(), + || Ok(times.next().unwrap_or(CHECK)) + ) + .await, + Err(Error::Deadline) + )); + assert!(matches!( + request + .confirm( + fixture.journal.as_ref(), + &fixture.directory, + &Processes::new(fixture.process_path()), + session(1), + deadline(), + || Ok(request.interval().1 - 1) + ) + .await, + Err(Error::Deadline) + )); + assert_eq!(fixture.failed_boot().await, *request.boot()); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn retained_fenced_boot_retirement_adopts_lost_reply_with_unchanged_process_evidence() { + let fixture = Fixture::with_process_observation(true).await; + let request = fixture.process_request.as_ref().unwrap(); + fixture.settle_followers().await; + let capture = FleetFailedBootRetirement::capture_retained( + fixture.journal.as_ref(), + &fixture.directory, + &fixture.roster().await, + request, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .unwrap(); + let (entered, resume) = fixture.journal.pause_next_enrollment_reply(true, true); + let processes = Processes::new(fixture.process_path()); + let mut work = Box::pin(capture.publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK), + )); + tokio::select! { result = &mut work => panic!("unexpected completion {}", result.is_ok()), result = entered => result.unwrap() } + assert_eq!( + fixture.failed_boot().await.status(), + EnrollmentStatus::Retired + ); + resume.send(()).unwrap(); + let result = work.await.unwrap(); + assert!(result.record().is_err()); + assert_eq!( + fixture.failed_boot().await.status(), + EnrollmentStatus::Retired + ); + let replay = FleetFailedBootRetirement::capture_retained( + fixture.journal.as_ref(), + &fixture.directory, + &fixture.roster().await, + request, + session(1), + deadline(), + || Ok(CHECK + 1), + ) + .await + .unwrap(); + let result = replay + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &Processes::new(fixture.process_path()), + session(1), + deadline(), + || Ok(CHECK + 1), + ) + .await + .unwrap(); + assert_eq!( + result.confirmed().unwrap().process().request_digest(), + request.digest() + ); + assert_eq!( + result.confirmed().unwrap().boot(), + &fixture.failed_boot().await + ); + assert_eq!(replay.request().interval(), request.interval()); + fixture.journal.close().await.unwrap(); +} diff --git a/crates/cellule-host/minion/scenario/recovered_followers/failed_boot/tests.rs b/crates/cellule-host/minion/scenario/recovered_followers/failed_boot/tests.rs new file mode 100644 index 00000000..d871768e --- /dev/null +++ b/crates/cellule-host/minion/scenario/recovered_followers/failed_boot/tests.rs @@ -0,0 +1,720 @@ +use super::*; + +#[tokio::test] +async fn failed_boot_closure_requires_joined_original_process_and_replays_after_reconstruction() { + let fixture = Fixture::new().await; + fixture.settle_followers().await; + let original = fixture.failed_boot().await; + let capture = fixture.failed_capture(&original).await; + let mut process = Process::start(fixture.process_path()); + let processes = Processes::new(fixture.process_path()); + let result = capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK), + ) + .await; + assert!(matches!(result, Err(Error::Facility { .. }))); + assert!(process.child.try_wait().unwrap().is_none()); + assert_eq!(fixture.failed_boot().await, original); + process.stop_and_retain(capture.request()); + let publication = capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .unwrap(); + let closure = publication.confirmed().unwrap(); + assert_eq!(closure.boot().spec(), original.spec()); + assert_eq!(closure.boot().accepted_at_ms(), original.accepted_at_ms()); + assert_eq!( + closure.boot().established_evidence(), + original.established_evidence() + ); + assert_eq!(closure.canonical().node(), node_id(0)); + assert_eq!(closure.canonical().session(), session(0)); + assert_eq!( + closure.canonical().log().unwrap().phase(), + cellule_runtime::node::log_state::NodeLogPhase::Retired + ); + assert!( + !fixture + .roster() + .await + .required_boots() + .iter() + .any(|boot| boot.session == session(0)) + ); + // This remains a recovery tombstone, not a clean local drain/withdrawal. + assert!(!fixture.directory.is_withdrawn(session(0)).await.unwrap()); + fixture.journal.close().await.unwrap(); + let journal = fixture.reconstruct().await; + let processes = Processes::new(fixture.process_path()); + let snapshot = journal.load_snapshot(scope()).await.unwrap(); + let roster = FleetRoster::collect(&journal, &snapshot, deadline()) + .await + .unwrap(); + let replay = FleetFailedBootRetirement::capture( + &journal, + &fixture.directory, + &roster, + &original, + session(1), + deadline(), + || Ok(CHECK + 10), + ) + .await + .unwrap(); + assert_eq!(replay.request().digest(), capture.request().digest()); + let repeated = replay + .publish( + &journal, + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK + 10), + ) + .await + .unwrap(); + assert_eq!(repeated.confirmed().unwrap().boot(), closure.boot()); + assert_eq!(repeated.confirmed().unwrap().digest(), closure.digest()); + assert_eq!(repeated.confirmed().unwrap().snapshot(), closure.snapshot()); + assert_eq!(fixture.transport.retirements.load(Ordering::Acquire), 2); + journal.close().await.unwrap(); +} + +#[tokio::test] +async fn failed_boot_closure_refuses_unretired_logs_unresolved_roles_and_duplicate_boots() { + let fixture = Fixture::new().await; + let original = fixture.failed_boot().await; + // Even Sealed, inactive recovery is not terminal leader-log retirement. + assert!( + fixture + .directory + .closed_session(node_id(0), session(0), session(1), CHECK) + .await + .is_err() + ); + fixture.retire().await; + assert!( + FleetFailedBootRetirement::capture( + fixture.journal.as_ref(), + &fixture.directory, + &fixture.roster().await, + &original, + session(1), + deadline(), + || Ok(CHECK) + ) + .await + .is_err() + ); + fixture + .capture() + .await + .publish( + fixture.journal.as_ref(), + &fixture.directory, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .unwrap() + .confirmed() + .unwrap(); + let mut duplicate = original.spec().clone(); + duplicate.request = Digest::from_bytes([218; 32]); + fixture + .journal + .accept_enrollment(&duplicate, CHECK) + .await + .unwrap(); + assert!( + FleetFailedBootRetirement::capture( + fixture.journal.as_ref(), + &fixture.directory, + &fixture.roster().await, + &original, + session(1), + deadline(), + || Ok(CHECK) + ) + .await + .is_err() + ); + assert_eq!( + fixture.failed_boot().await.status(), + EnrollmentStatus::Established + ); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn failed_boot_closure_retains_lost_retirement_reply_and_original_times() { + let fixture = Fixture::new().await; + fixture.settle_followers().await; + let original = fixture.failed_boot().await; + let capture = fixture.failed_capture(&original).await; + let mut process = Process::start(fixture.process_path()); + process.stop_and_retain(capture.request()); + let processes = Processes::new(fixture.process_path()); + let (entered, resume) = fixture.journal.pause_next_enrollment_reply(true, true); + let mut work = Box::pin(capture.publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK), + )); + tokio::select! { result = &mut work => panic!("unexpected completion {}", result.is_ok()), result = entered => result.unwrap() } + let committed = fixture.failed_boot().await; + assert_eq!(committed.status(), EnrollmentStatus::Retired); + resume.send(()).unwrap(); + let result = work.await.unwrap(); + assert!(result.confirmed().is_err()); + let error = result.record().unwrap_err(); + let Error::Facility { source, .. } = error.as_ref() else { + panic!("original source absent") + }; + assert!(source.downcast_ref::().is_some()); + assert!(Arc::ptr_eq(&error, &result.record().unwrap_err())); + fixture.journal.close().await.unwrap(); + let journal = fixture.reconstruct().await; + let snapshot = journal.load_snapshot(scope()).await.unwrap(); + let roster = FleetRoster::collect(&journal, &snapshot, deadline()) + .await + .unwrap(); + let replay = FleetFailedBootRetirement::capture( + &journal, + &fixture.directory, + &roster, + &original, + session(1), + deadline(), + || Ok(CHECK + 10), + ) + .await + .unwrap(); + let processes = Processes::new(fixture.process_path()); + let replay = replay + .publish( + &journal, + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK + 10), + ) + .await + .unwrap(); + assert_eq!(replay.confirmed().unwrap().boot(), &committed); + assert!(Arc::ptr_eq(&error, &result.record().unwrap_err())); + journal.close().await.unwrap(); +} + +#[tokio::test] +async fn failed_boot_closure_retains_publication_when_final_process_confirmation_fails_or_changes() +{ + for changed in [false, true] { + let fixture = Fixture::new().await; + fixture.settle_followers().await; + let original = fixture.failed_boot().await; + let capture = fixture.failed_capture(&original).await; + let mut process = Process::start(fixture.process_path()); + process.stop_and_retain(capture.request()); + let processes = Processes::new(fixture.process_path()); + if changed { + *processes.final_witness.lock().unwrap() = Some(Digest::from_bytes([217; 32])); + } else { + *processes.final_fault.lock().unwrap() = Some(std::io::ErrorKind::ConnectionReset); + } + let result = capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .unwrap(); + assert!(result.confirmed().is_err()); + let committed = result.record().unwrap().clone(); + assert_eq!(committed.status(), EnrollmentStatus::Retired); + let error = result.closure_error().unwrap(); + assert!(Arc::ptr_eq(&error, &result.closure_error().unwrap())); + if !changed { + let Error::Facility { source, .. } = error.as_ref() else { + panic!("original process failure absent") + }; + assert_eq!( + source.downcast_ref::().unwrap().kind(), + std::io::ErrorKind::ConnectionReset + ); + } + let replay = fixture.failed_capture(&original).await; + let replay = replay + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK + 1), + ) + .await + .unwrap(); + assert_eq!(replay.confirmed().unwrap().boot(), &committed); + fixture.journal.close().await.unwrap(); + } +} + +#[tokio::test] +async fn failed_boot_closure_final_roster_rejects_delayed_original_session_responsibility() { + let fixture = Fixture::new().await; + fixture.settle_followers().await; + let original = fixture.failed_boot().await; + let capture = fixture.failed_capture(&original).await; + let mut process = Process::start(fixture.process_path()); + process.stop_and_retain(capture.request()); + let processes = Processes::new(fixture.process_path()); + let (entered, resume) = fixture.journal.pause_next_enrollment_reply(true, false); + let mut work = Box::pin(capture.publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK), + )); + tokio::select! { result = &mut work => panic!("unexpected completion {}", result.is_ok()), result = entered => result.unwrap() } + let mut delayed = fixture.originals[0].spec().clone(); + delayed.request = Digest::from_bytes([216; 32]); + let FleetEnrollmentAcceptance::New(pending) = fixture + .journal + .accept_enrollment(&delayed, CHECK) + .await + .unwrap() + else { + panic!("new late responsibility expected") + }; + resume.send(()).unwrap(); + let result = work.await.unwrap(); + assert!(result.record().is_ok()); + assert!(result.confirmed().is_err()); + assert!( + fixture + .roster() + .await + .required_boots() + .iter() + .any(|boot| boot.session == session(0)) + ); + assert!( + FleetFailedBootRetirement::capture( + fixture.journal.as_ref(), + &fixture.directory, + &fixture.roster().await, + &original, + session(1), + deadline(), + || Ok(CHECK) + ) + .await + .is_err() + ); + // The unit case never dispatched this delayed native CAS: join and publish + // that exact nonexecution exclusion through the existing atomic path. + fixture + .journal + .refuse_unexecuted_enrollment(pending.spec(), Digest::from_bytes([215; 32]), CHECK + 1) + .await + .unwrap(); + let replay = fixture.failed_capture(&original).await; + replay + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK + 1), + ) + .await + .unwrap() + .confirmed() + .unwrap(); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn failed_boot_closure_refuses_stale_snapshot_foreign_process_evidence_and_clock_regression() +{ + let fixture = Fixture::new().await; + fixture.settle_followers().await; + let original = fixture.failed_boot().await; + let capture = fixture.failed_capture(&original).await; + assert!( + FleetFailedBootProcessEvidence::new(capture.request(), Digest::from_bytes([0; 32])) + .is_err() + ); + let other = fixture + .roster() + .await + .enrollments() + .iter() + .find(|row| { + matches!(row.spec().role, EnrollmentRole::Node { .. }) + && row.spec().target.session == session(2) + }) + .unwrap() + .clone(); + let observed = fixture + .directory + .load(session(2), CHECK) + .await + .unwrap() + .unwrap(); + fixture.directory.withdraw(&observed, CHECK).await.unwrap(); + let other = FleetFailedBootRetirement::capture( + fixture.journal.as_ref(), + &fixture.directory, + &fixture.roster().await, + &other, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .unwrap(); + // Inject a response naming another actual canonical boot request. A public + // evidence constructor cannot bypass the receiver's original digest check. + let foreign = ForeignEvidence( + FleetFailedBootProcessEvidence::new(other.request(), Digest::from_bytes([212; 32])) + .unwrap(), + ); + assert!(matches!( + capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &foreign, + session(1), + deadline(), + || Ok(CHECK) + ) + .await, + Err(Error::Fenced) + )); + assert_eq!(fixture.failed_boot().await, original); + let mut process = Process::start(fixture.process_path()); + process.stop_and_retain(capture.request()); + let processes = Processes::new(fixture.process_path()); + let bytes = std::fs::read(fixture.process_path()).unwrap(); + let mut foreign = bytes.clone(); + foreign[0] ^= 1; + std::fs::write(fixture.process_path(), foreign).unwrap(); + assert!( + capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK) + ) + .await + .is_err() + ); + assert_eq!(fixture.failed_boot().await, original); + std::fs::write(fixture.process_path(), bytes).unwrap(); + let mut calls = 0; + let clock = || { + calls += 1; + Ok(if calls == 1 { CHECK } else { CHECK - 1 }) + }; + assert!(matches!( + capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + clock + ) + .await, + Err(Error::Deadline) + )); + assert_eq!(fixture.failed_boot().await, original); + fixture + .journal + .register_initial_intent( + &NodeIntent::initial( + scope(), + NodeId::from_bytes([214; 16]), + SessionId::from_bytes([213; 16]), + ) + .unwrap(), + ) + .await + .unwrap(); + assert!( + capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK) + ) + .await + .is_err() + ); + assert_eq!(fixture.failed_boot().await, original); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn failed_boot_closure_cancelled_publication_joins_backend_before_original_replay() { + let fixture = Fixture::new().await; + fixture.settle_followers().await; + let original = fixture.failed_boot().await; + let capture = fixture.failed_capture(&original).await; + let mut process = Process::start(fixture.process_path()); + process.stop_and_retain(capture.request()); + let processes = Processes::new(fixture.process_path()); + let (entered, resume) = fixture.journal.pause_next_enrollment_reply(true, false); + let mut work = Box::pin(capture.publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK), + )); + tokio::select! { result = &mut work => panic!("unexpected completion {}", result.is_ok()), result = entered => result.unwrap() } + drop(work); + assert!(resume.send(()).is_err()); + fixture.journal.close().await.unwrap(); + let journal = fixture.reconstruct().await; + let committed = journal + .load_enrollment(scope(), original.spec().key().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(committed.status(), EnrollmentStatus::Retired); + let snapshot = journal.load_snapshot(scope()).await.unwrap(); + let roster = FleetRoster::collect(&journal, &snapshot, deadline()) + .await + .unwrap(); + let capture = FleetFailedBootRetirement::capture( + &journal, + &fixture.directory, + &roster, + &original, + session(1), + deadline(), + || Ok(CHECK + 10), + ) + .await + .unwrap(); + let processes = Processes::new(fixture.process_path()); + let replay = capture + .publish( + &journal, + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK + 10), + ) + .await + .unwrap(); + assert_eq!(replay.confirmed().unwrap().boot(), &committed); + assert_eq!(fixture.transport.retirements.load(Ordering::Acquire), 2); + journal.close().await.unwrap(); +} + +#[tokio::test] +async fn failed_boot_closure_replay_preserves_new_boot_foreign_roles_on_the_same_physical_node() { + let fixture = Fixture::new().await; + fixture.settle_followers().await; + let original = fixture.failed_boot().await; + let capture = fixture.failed_capture(&original).await; + let mut process = Process::start(fixture.process_path()); + process.stop_and_retain(capture.request()); + let processes = Processes::new(fixture.process_path()); + let publication = capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .unwrap(); + let closure = publication.confirmed().unwrap(); + let old = NodeIntent::initial(scope(), node_id(0), session(0)).unwrap(); + let next = fixture + .journal + .rebind_active_intent(&old, SessionId::from_bytes([50; 16]), 2) + .await + .unwrap(); + let ad = NodeAdvertisement::sign( + next.node(), + next.session(), + "https://replacement.invalid".into(), + scope().fleet, + Digest::from_bytes([30; 32]), + Digest::from_bytes([31; 32]), + Digest::from_bytes([32; 32]), + &SigningKey::from_bytes(&[50; 32]), + 1, + CHECK, + CHECK + 10_000, + vec![Digest::from_bytes([33; 32])], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + free_memory_bytes: 1 << 20, + free_disk_bytes: 1 << 20, + follower_free_bytes: 1 << 20, + job_credits: 4, + log_protocol: 1, + ..Default::default() + }, + ) + .unwrap(); + startup::enroll( + fixture.journal.as_ref(), + &fixture.directory, + &startup::spec(&next).unwrap(), + ad, + CHECK, + ) + .await + .unwrap(); + let leader = fixture + .directory + .load(session(1), CHECK) + .await + .unwrap() + .unwrap(); + let prepared = fixture + .directory + .prepare_log_enrollment(&leader, 8, 1, 3, CHECK) + .await + .unwrap() + .unwrap(); + assert!( + prepared + .followers() + .iter() + .any(|ad| ad.node() == node_id(0) && ad.session() == next.session()) + ); + let attempt = fixture + .directory + .prepare_log_enrollment_attempt(&prepared, CHECK) + .await + .unwrap(); + let mut accepted = Vec::new(); + for (index, member) in prepared.followers().iter().enumerate() { + let spec = EnrollmentSpec { + scope: scope(), + request: Digest::from_bytes([index as u8 + 200; 32]), + source: Some(EnrollmentEndpoint { + node: node_id(1), + session: session(1), + intent_revision: 1, + }), + target: EnrollmentEndpoint { + node: member.node(), + session: member.session(), + intent_revision: if member.node() == next.node() { 2 } else { 1 }, + }, + role: EnrollmentRole::Follower { log_epoch: 8 }, + }; + let FleetEnrollmentAcceptance::New(row) = fixture + .journal + .accept_enrollment(&spec, CHECK) + .await + .unwrap() + else { + panic!("new replacement enrollment expected") + }; + accepted.push(row); + } + fixture + .directory + .commit_log_enrollment(&attempt, CHECK) + .await + .unwrap(); + for row in accepted { + fixture + .journal + .publish_enrollment_result( + &row, + EnrollmentEvent::Established(Digest::from_bytes([198; 32])), + CHECK, + ) + .await + .unwrap(); + } + // The original process evidence stays immutable. The different boot's + // current foreign responsibility is retained, never retired by old replay. + let replay = fixture.failed_capture(&original).await; + assert_eq!(replay.request().digest(), capture.request().digest()); + let replay = replay + .publish( + fixture.journal.as_ref(), + &fixture.directory, + &processes, + session(1), + deadline(), + || Ok(CHECK + 1), + ) + .await + .unwrap(); + assert_eq!(replay.confirmed().unwrap().boot(), closure.boot()); + assert_eq!(replay.confirmed().unwrap().digest(), closure.digest()); + let roster = fixture.roster().await; + assert!( + roster + .required_boots() + .iter() + .any(|boot| boot.session == next.session()) + ); + assert!( + !roster + .required_boots() + .iter() + .any(|boot| boot.session == session(0)) + ); + assert!( + roster + .enrollments() + .iter() + .any(|row| row.spec().target.session == next.session() + && matches!(row.spec().role, EnrollmentRole::Follower { log_epoch: 8 }) + && row.status() == EnrollmentStatus::Established) + ); + fixture.journal.close().await.unwrap(); +} diff --git a/crates/cellule-host/minion/scenario/recovered_followers/failed_boot/writer_tests.rs b/crates/cellule-host/minion/scenario/recovered_followers/failed_boot/writer_tests.rs new file mode 100644 index 00000000..cf67b544 --- /dev/null +++ b/crates/cellule-host/minion/scenario/recovered_followers/failed_boot/writer_tests.rs @@ -0,0 +1,864 @@ +//! Complete metadata capture over authenticated closed test catalogs. The child +//! proves a lifetime stand-in; these cases supply no OS-crashed CellNode, prefix +//! availability or external-job qualification. +use super::*; +use cellule_host::fleet::{ + FleetJournalSnapshot, FleetOriginalCatalogSet, FleetOriginalCatalogSource, + FleetOriginalCatalogs, FleetOriginalWriterCapture, FleetOriginalWriterInventory, + FleetOriginalWriterJournal, +}; +use cellule_runtime::control::Control; +use cellule_runtime::control::{Transition, authority::CellAuthority}; +use cellule_runtime::fleet::operations::{ + JournalTransition, MaintenanceOperation, OperationId, OriginalWriterInventoryRecord, +}; +use cellule_runtime::identity::NamespaceId; + +struct Catalogs { + sources: Vec, + reads: AtomicUsize, + change_on_second: bool, + error_on_second: bool, +} +impl FleetOriginalCatalogs for Catalogs { + fn catalogs<'a>( + &'a self, + request: &'a FleetFailedBootProcessRequest, + operation: &'a MaintenanceOperation, + _: &'a FleetJournalSnapshot, + ) -> FleetAdapterFuture<'a, FleetOriginalCatalogSet> { + Box::pin(async move { + let read = self.reads.fetch_add(1, Ordering::AcqRel); + if read == 1 && self.error_on_second { + return Err(std::io::Error::new( + std::io::ErrorKind::PermissionDenied, + "original catalog authentication failed", + ) + .into()); + } + let witness = Digest::from_bytes( + [if read == 1 && self.change_on_second { + 51 + } else { + 50 + }; 32], + ); + Ok(FleetOriginalCatalogSet::new( + request, + operation.clone(), + witness, + self.sources.clone(), + )?) + }) + } +} +struct WriterFixture { + base: Fixture, + request: FleetFailedBootProcessRequest, + catalogs: Catalogs, + expected: Vec, + layouts: Vec, +} +impl WriterFixture { + async fn new() -> Self { + Self::with_originals(33).await + } + async fn with_originals(originals_per_catalog: u64) -> Self { + let base = Fixture::new().await; + let request = FleetFailedBootProcessRequest::capture_fenced( + base.journal.as_ref(), + &base.directory, + &base.roster().await, + &base.failed_boot().await, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .unwrap(); + let mut sources = Vec::new(); + let mut layouts = Vec::new(); + let mut expected = Vec::new(); + for index in 0..2 { + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + ObjectPath::from(format!("original-catalog-{index}")), + [index + 3; 16], + ); + let tenant = TenantId::from_bytes([index + 1; 16]); + let catalog = CellCatalog::new(layout.clone(), tenant); + let authority = CellAuthority::new(layout.clone()); + // Accepted original metadata commits before process joining. Each + // original is later removed by the sole authority takeover path. + for n in 0_u64..originals_per_catalog { + let target = CellTarget::new( + tenant, + catalog.application(), + NamespaceId::from_bytes([9; 16]), + &n.to_be_bytes(), + ) + .unwrap(); + let proof = catalog + .provision( + CatalogEntry::new( + &target, + CatalogRole::Sql, + Digest::from_bytes([8; 32]), + 1, + ) + .unwrap(), + ) + .await + .unwrap(); + let observed = authority + .create_initial(&proof, IncarnationId::from_bytes([10; 16]), owner(0)) + .await + .unwrap(); + expected.push(observed.value().clone()); + authority + .transition( + &observed, + observed.value().takeover(owner(1)).unwrap(), + Transition::Takeover, + ) + .await + .unwrap(); + } + // An unused bootstrap entry is part of complete traversal. + let target = CellTarget::new( + tenant, + catalog.application(), + NamespaceId::from_bytes([9; 16]), + b"unused", + ) + .unwrap(); + catalog + .provision( + CatalogEntry::new(&target, CatalogRole::Sql, Digest::from_bytes([8; 32]), 1) + .unwrap(), + ) + .await + .unwrap(); + layouts.push(layout.clone()); + sources.push( + FleetOriginalCatalogSource::new( + Digest::from_bytes([index + 60; 32]), + tenant, + layout, + ) + .unwrap(), + ); + } + let mut child = Process::start(base.process_path()); + child.stop_and_retain(&request); + let now = CHECK; + let before = base.journal.load_snapshot(scope()).await.unwrap(); + let snapshot = base + .journal + .claim_controller(scope(), before.head().revision(), session(1), now) + .await + .unwrap(); + let operation = MaintenanceOperation::new( + OperationId::from_bytes([80; 16]).unwrap(), + Digest::from_bytes([81; 32]), + node_id(0), + session(0), + 2, + now, + now + 60_000, + ) + .unwrap(); + base.journal + .compare_exchange( + &snapshot, + snapshot.head().controller().unwrap().epoch, + now, + &JournalTransition::BeginMaintenance(operation), + ) + .await + .unwrap(); + Self { + base, + request, + catalogs: Catalogs { + sources, + reads: AtomicUsize::new(0), + change_on_second: false, + error_on_second: false, + }, + expected, + layouts, + } + } + async fn capture(&self) -> cellule_runtime::Result { + FleetOriginalWriterCapture::capture( + self.base.journal.as_ref(), + &self.base.directory, + &Processes::new(self.base.process_path()), + &self.catalogs, + &self.request, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + } +} +#[tokio::test] +async fn complete_original_writers_survive_takeover_atomic_publication_and_reconstruction() { + let fixture = WriterFixture::new().await; + let capture = fixture.capture().await.unwrap(); + assert_eq!(capture.record().owner_count(), 66); + assert_eq!(capture.pages().len(), 2); + assert_eq!( + capture + .record() + .catalogs() + .iter() + .map(|scope| scope.cells) + .sum::(), + 68 + ); + let original_interval = capture.record().basis().interval; + let original_digest = capture.record().digest().unwrap(); + let stored = capture + .publish( + fixture.base.journal.as_ref(), + &fixture.base.directory, + &Processes::new(fixture.base.process_path()), + &fixture.catalogs, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .unwrap(); + assert_eq!(stored, *capture.record()); + let snapshot = fixture.base.journal.load_snapshot(scope()).await.unwrap(); + assert_eq!( + snapshot.registry().revision(), + stored.basis().registry.revision() + 1 + ); + let replay = capture + .publish( + fixture.base.journal.as_ref(), + &fixture.base.directory, + &Processes::new(fixture.base.process_path()), + &fixture.catalogs, + session(1), + deadline(), + || Ok(CHECK + 60_000), + ) + .await + .unwrap(); + assert_eq!(replay.digest().unwrap(), original_digest); + assert_eq!( + fixture.base.journal.load_snapshot(scope()).await.unwrap(), + snapshot + ); + fixture.base.journal.close().await.unwrap(); + let independent = fixture.base.reconstruct().await; + let snapshot = independent.load_snapshot(scope()).await.unwrap(); + let recovered = FleetOriginalWriterInventory::load( + &independent, + &snapshot, + stored.basis().operation.id(), + fixture.request.digest(), + deadline(), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(recovered.record(), &stored); + assert_eq!(recovered.pages(), capture.pages()); + assert_eq!(recovered.record().basis().interval, original_interval); + let mut actual = recovered + .writers() + .map(|row| row.control.clone()) + .collect::>(); + let mut expected = fixture.expected; + actual.sort_by_key(|row| *row.cell.as_bytes()); + expected.sort_by_key(|row| *row.cell.as_bytes()); + assert_eq!(actual, expected); + independent.close().await.unwrap(); +} +#[tokio::test] +async fn changed_or_unauthenticated_original_catalog_configuration_cannot_publish() { + for source_error in [false, true] { + let mut fixture = WriterFixture::new().await; + fixture.catalogs.change_on_second = !source_error; + fixture.catalogs.error_on_second = source_error; + let before = fixture.base.journal.load_snapshot(scope()).await.unwrap(); + let error = fixture.capture().await.err().unwrap(); + if source_error { + let Error::Facility { source, .. } = error else { + panic!("provider cause required") + }; + assert_eq!( + source.downcast_ref::().unwrap().kind(), + std::io::ErrorKind::PermissionDenied + ); + } else { + assert!(matches!( + error, + Error::Control("original catalog configuration changed") + )); + } + assert_eq!( + fixture.base.journal.load_snapshot(scope()).await.unwrap(), + before + ); + assert!( + fixture + .base + .journal + .original_writers( + &before, + before.head().maintenance().unwrap().id(), + fixture.request.digest() + ) + .await + .unwrap() + .is_none() + ); + fixture.base.journal.close().await.unwrap(); + } +} +#[tokio::test] +async fn original_set_lost_commit_reply_is_adopted_without_recollecting_or_restamping() { + let fixture = WriterFixture::new().await; + let capture = fixture.capture().await.unwrap(); + let expected = fixture.base.journal.load_snapshot(scope()).await.unwrap(); + fixture.base.journal.lose_next_commit_reply(); + assert!( + fixture + .base + .journal + .persist_original_writers(&expected, capture.record(), capture.pages(), CHECK) + .await + .is_err() + ); + let snapshot = fixture.base.journal.load_snapshot(scope()).await.unwrap(); + assert_ne!(snapshot.registry(), expected.registry()); + let retained = fixture + .base + .journal + .original_writers( + &snapshot, + capture.record().basis().operation.id(), + fixture.request.digest(), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(retained, *capture.record()); + fixture + .base + .journal + .persist_original_writers(&expected, capture.record(), capture.pages(), CHECK + 60_000) + .await + .unwrap(); + assert_eq!( + fixture.base.journal.load_snapshot(scope()).await.unwrap(), + snapshot + ); + fixture.base.journal.close().await.unwrap(); +} +#[tokio::test] +async fn changed_barrier_and_incomplete_pages_cannot_commit_an_original_set() { + let fixture = WriterFixture::new().await; + let capture = fixture.capture().await.unwrap(); + let expected = fixture.base.journal.load_snapshot(scope()).await.unwrap(); + assert!( + fixture + .base + .journal + .persist_original_writers(&expected, capture.record(), &capture.pages()[..1], CHECK) + .await + .is_err() + ); + assert_eq!( + fixture.base.journal.load_snapshot(scope()).await.unwrap(), + expected + ); + fixture + .base + .journal + .set_scheduling(expected.registry(), true) + .await + .unwrap(); + assert_ne!( + fixture + .base + .journal + .load_snapshot(scope()) + .await + .unwrap() + .registry(), + expected.registry() + ); + assert!( + capture + .publish( + fixture.base.journal.as_ref(), + &fixture.base.directory, + &Processes::new(fixture.base.process_path()), + &fixture.catalogs, + session(1), + deadline(), + || Ok(CHECK) + ) + .await + .is_err() + ); + let current = fixture.base.journal.load_snapshot(scope()).await.unwrap(); + assert!( + fixture + .base + .journal + .original_writers( + ¤t, + capture.record().basis().operation.id(), + fixture.request.digest() + ) + .await + .unwrap() + .is_none() + ); + fixture.base.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn missing_original_owner_history_is_a_typed_blocker_not_an_empty_set() { + let fixture = WriterFixture::new().await; + let original = &fixture.expected[0]; + let layout = &fixture.layouts[0]; + layout + .store() + .delete(&layout.owner_observation_path( + original.cell.as_bytes(), + original.incarnation.as_bytes(), + original.epoch, + )) + .await + .unwrap(); + let error = fixture.capture().await.err().unwrap(); + assert!( + matches!(error,Error::OwnerHistoryIncomplete { cell,incarnation,epoch } if cell==original.cell && incarnation==original.incarnation && epoch==original.epoch) + ); + let snapshot = fixture.base.journal.load_snapshot(scope()).await.unwrap(); + assert!( + fixture + .base + .journal + .original_writers( + &snapshot, + snapshot.head().maintenance().unwrap().id(), + fixture.request.digest() + ) + .await + .unwrap() + .is_none() + ); + fixture.base.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn catalog_and_process_changes_before_first_publication_refuse() { + for catalog_change in [false, true] { + let fixture = WriterFixture::new().await; + let capture = fixture.capture().await.unwrap(); + let before = fixture.base.journal.load_snapshot(scope()).await.unwrap(); + if catalog_change { + let catalog = + CellCatalog::new(fixture.layouts[0].clone(), TenantId::from_bytes([1; 16])); + let target = CellTarget::new( + catalog.tenant(), + catalog.application(), + NamespaceId::from_bytes([9; 16]), + b"later", + ) + .unwrap(); + catalog + .provision( + CatalogEntry::new(&target, CatalogRole::Sql, Digest::from_bytes([8; 32]), 1) + .unwrap(), + ) + .await + .unwrap(); + } else { + let mut bytes = std::fs::read(fixture.base.process_path()).unwrap(); + bytes[32] ^= 1; + std::fs::write(fixture.base.process_path(), bytes).unwrap(); + } + assert!( + capture + .publish( + fixture.base.journal.as_ref(), + &fixture.base.directory, + &Processes::new(fixture.base.process_path()), + &fixture.catalogs, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .is_err() + ); + assert_eq!( + fixture.base.journal.load_snapshot(scope()).await.unwrap(), + before + ); + assert!( + fixture + .base + .journal + .original_writers( + &before, + capture.record().basis().operation.id(), + fixture.request.digest(), + ) + .await + .unwrap() + .is_none() + ); + fixture.base.journal.close().await.unwrap(); + } +} + +#[tokio::test] +async fn independent_publications_choose_one_immutable_original_set() { + for identical in [false, true] { + let fixture = WriterFixture::new().await; + let capture = fixture.capture().await.unwrap(); + let before = fixture.base.journal.load_snapshot(scope()).await.unwrap(); + let independent = fixture.base.reconstruct().await; + let mut basis = capture.record().basis().clone(); + if !identical { + basis.catalog_witness = Digest::from_bytes([91; 32]); + } + let (other, pages) = OriginalWriterInventoryRecord::new( + basis, + capture.record().catalogs().to_vec(), + capture + .pages() + .iter() + .flat_map(|page| page.entries().iter().cloned()) + .collect(), + ) + .unwrap(); + let (a, b) = tokio::join!( + fixture.base.journal.persist_original_writers( + &before, + capture.record(), + capture.pages(), + CHECK + ), + independent.persist_original_writers(&before, &other, &pages, CHECK), + ); + assert_eq!( + usize::from(a.is_ok()) + usize::from(b.is_ok()), + if identical { 2 } else { 1 } + ); + let current = independent.load_snapshot(scope()).await.unwrap(); + assert_eq!( + current.registry().revision(), + before.registry().revision() + 1 + ); + let retained = independent + .original_writers( + ¤t, + capture.record().basis().operation.id(), + fixture.request.digest(), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(retained, a.or(b).unwrap()); + independent.close().await.unwrap(); + fixture.base.journal.close().await.unwrap(); + } +} + +#[tokio::test] +async fn canceled_publication_waiter_preserves_the_accepted_original_commit() { + let fixture = WriterFixture::new().await; + let capture = fixture.capture().await.unwrap(); + let before = fixture.base.journal.load_snapshot(scope()).await.unwrap(); + let (committed, resume) = fixture.base.journal.pause_next_original_writer_reply(); + let journal = Arc::clone(&fixture.base.journal); + let expected = before.clone(); + let record = capture.record().clone(); + let pages = capture.pages().to_vec(); + let waiter = tokio::spawn(async move { + journal + .persist_original_writers(&expected, &record, &pages, CHECK) + .await + }); + committed.await.unwrap(); + waiter.abort(); + assert!(waiter.await.unwrap_err().is_cancelled()); + drop(resume); + fixture.base.journal.close().await.unwrap(); + let independent = fixture.base.reconstruct().await; + let current = independent.load_snapshot(scope()).await.unwrap(); + assert_eq!( + current.registry().revision(), + before.registry().revision() + 1 + ); + let retained = FleetOriginalWriterInventory::load( + &independent, + ¤t, + capture.record().basis().operation.id(), + fixture.request.digest(), + deadline(), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(retained.record(), capture.record()); + assert_eq!(retained.pages(), capture.pages()); + assert_eq!(retained.writers().count(), 66); + independent.close().await.unwrap(); +} + +#[tokio::test] +async fn original_writer_reload_distinguishes_absence_and_rejects_stale_barriers() { + let fixture = WriterFixture::new().await; + let capture = fixture.capture().await.unwrap(); + let before = fixture.base.journal.load_snapshot(scope()).await.unwrap(); + let operation = capture.record().basis().operation.id(); + assert!( + FleetOriginalWriterInventory::load( + fixture.base.journal.as_ref(), + &before, + operation, + fixture.request.digest(), + deadline(), + ) + .await + .unwrap() + .is_none() + ); + fixture + .base + .journal + .persist_original_writers(&before, capture.record(), capture.pages(), CHECK) + .await + .unwrap(); + let current = fixture.base.journal.load_snapshot(scope()).await.unwrap(); + let error = FleetOriginalWriterInventory::load( + fixture.base.journal.as_ref(), + &before, + operation, + fixture.request.digest(), + deadline(), + ) + .await + .err() + .unwrap(); + let Error::Facility { source, .. } = error else { + panic!("original journal conflict required") + }; + assert!(matches!( + source.downcast_ref::(), + Some(cellule_runtime::fleet::operations::OperationError::Conflict) + )); + let loaded = FleetOriginalWriterInventory::load( + fixture.base.journal.as_ref(), + ¤t, + operation, + fixture.request.digest(), + deadline(), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(loaded.record(), capture.record()); + assert_eq!(loaded.pages(), capture.pages()); + assert_eq!( + fixture.base.journal.load_snapshot(scope()).await.unwrap(), + current + ); + fixture.base.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn original_writer_reload_never_returns_a_partial_missing_or_corrupt_set() { + for corrupt in [false, true] { + let fixture = WriterFixture::new().await; + let capture = fixture.capture().await.unwrap(); + let before = fixture.base.journal.load_snapshot(scope()).await.unwrap(); + fixture + .base + .journal + .persist_original_writers(&before, capture.record(), capture.pages(), CHECK) + .await + .unwrap(); + fixture.base.journal.close().await.unwrap(); + // Damage the final page offline. The first page remains valid, so a + // streamed or prematurely returned inventory would lose original owners. + let db = rusqlite::Connection::open(&fixture.base.path).unwrap(); + let digest = capture.record().pages()[1]; + let changed = if corrupt { + db.execute( + "UPDATE original_writer_pages SET body=?1 WHERE key=?2", + rusqlite::params![vec![0_u8], digest.as_bytes().as_slice()], + ) + .unwrap() + } else { + db.execute( + "DELETE FROM original_writer_pages WHERE key=?1", + [digest.as_bytes().as_slice()], + ) + .unwrap() + }; + assert_eq!(changed, 1); + drop(db); + let independent = fixture.base.reconstruct().await; + let current = independent.load_snapshot(scope()).await.unwrap(); + let error = FleetOriginalWriterInventory::load( + &independent, + ¤t, + capture.record().basis().operation.id(), + fixture.request.digest(), + deadline(), + ) + .await + .err() + .unwrap(); + if corrupt { + let Error::Facility { source, .. } = error else { + panic!("original page decoder cause required") + }; + assert!( + source + .downcast_ref::() + .is_some() + ); + } else { + assert!(matches!(error, Error::Fenced)); + } + assert_eq!(independent.load_snapshot(scope()).await.unwrap(), current); + independent.close().await.unwrap(); + } +} + +#[tokio::test] +async fn original_writer_reload_preserves_sql_source_errors() { + let fixture = WriterFixture::new().await; + let capture = fixture.capture().await.unwrap(); + let before = fixture.base.journal.load_snapshot(scope()).await.unwrap(); + fixture + .base + .journal + .persist_original_writers(&before, capture.record(), capture.pages(), CHECK) + .await + .unwrap(); + fixture.base.journal.close().await.unwrap(); + let independent = fixture.base.reconstruct().await; + let current = independent.load_snapshot(scope()).await.unwrap(); + // Open creates the schema; remove it afterwards to force the actual reader + // to return its source failure rather than a missing-page observation. + let db = rusqlite::Connection::open(&fixture.base.path).unwrap(); + db.execute("DROP TABLE original_writer_pages", []).unwrap(); + drop(db); + let error = FleetOriginalWriterInventory::load( + &independent, + ¤t, + capture.record().basis().operation.id(), + fixture.request.digest(), + deadline(), + ) + .await + .err() + .unwrap(); + let Error::Facility { source, .. } = error else { + panic!("original SQL source cause required") + }; + assert!(source.downcast_ref::().is_some()); + assert_eq!(independent.load_snapshot(scope()).await.unwrap(), current); + independent.close().await.unwrap(); +} + +#[tokio::test] +async fn original_writer_reload_refuses_expired_deadlines_and_invalid_keys() { + let fixture = WriterFixture::new().await; + let current = fixture.base.journal.load_snapshot(scope()).await.unwrap(); + let operation = current.head().maintenance().unwrap().id(); + let error = FleetOriginalWriterInventory::load( + fixture.base.journal.as_ref(), + ¤t, + operation, + fixture.request.digest(), + Instant::now(), + ) + .await + .err() + .unwrap(); + assert!(matches!(error, Error::Deadline)); + let error = FleetOriginalWriterInventory::load( + fixture.base.journal.as_ref(), + ¤t, + operation, + Digest::from_bytes([0; 32]), + deadline(), + ) + .await + .err() + .unwrap(); + assert!(matches!( + error, + Error::Control("invalid original writer inventory key") + )); + assert_eq!( + fixture.base.journal.load_snapshot(scope()).await.unwrap(), + current + ); + fixture.base.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn original_writer_reload_retains_an_explicit_complete_empty_capture() { + // Both authenticated original catalogs contain only unused bootstrap entries; + // no original owner has existed. This is a real complete scan, not omission + // of populated catalogs or removal of entries from a retained manifest. + let fixture = WriterFixture::with_originals(0).await; + let capture = fixture.capture().await.unwrap(); + assert_eq!(capture.record().owner_count(), 0); + assert!(capture.pages().is_empty()); + assert_eq!(capture.record().catalogs().len(), 2); + assert_eq!( + capture + .record() + .catalogs() + .iter() + .map(|row| row.cells) + .sum::(), + 2 + ); + let before = fixture.base.journal.load_snapshot(scope()).await.unwrap(); + fixture + .base + .journal + .persist_original_writers(&before, capture.record(), capture.pages(), CHECK) + .await + .unwrap(); + fixture.base.journal.close().await.unwrap(); + let independent = fixture.base.reconstruct().await; + let current = independent.load_snapshot(scope()).await.unwrap(); + let retained = FleetOriginalWriterInventory::load( + &independent, + ¤t, + capture.record().basis().operation.id(), + fixture.request.digest(), + deadline(), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(retained.record(), capture.record()); + assert!(retained.pages().is_empty()); + assert_eq!(retained.writers().count(), 0); + independent.close().await.unwrap(); +} diff --git a/crates/cellule-host/minion/scenario/recovered_followers/mod.rs b/crates/cellule-host/minion/scenario/recovered_followers/mod.rs new file mode 100644 index 00000000..2c2b15d6 --- /dev/null +++ b/crates/cellule-host/minion/scenario/recovered_followers/mod.rs @@ -0,0 +1,394 @@ +//! Actual cold recovered ensemble and durable original enrollment publication. +//! This fixture has no Cell suffix; runtime lifecycle tests cover pinned tails. +use super::*; +use bytes::Bytes; +use cellule_host::fleet::{ + FleetEnrollmentAcceptance, FleetRecoveredFollowerRetirement, FleetRoster, +}; +use cellule_runtime::{ + Error, + fleet::operations::{ + EnrollmentEndpoint, EnrollmentEvent, EnrollmentRecord, EnrollmentRole, EnrollmentSpec, + EnrollmentStatus, + }, + follower::{FollowerReceipt, FollowerStore}, + node::{ + NodeAdvertisement, NodeCapacity, NodeDirectory, NodeFailureDomain, SealedNodeLog, + log_recovery::{ + NodeLogRecovery, RecoveryCoordinator, retirement::retire_recovered_members, + }, + log_transport::{ + AppendRequest, LocalFollowerTransport, LocalRecoveredFollowerTransport, + NodeLogTransport, RecoveredNodeLogTransport, RecoveredRetireRequest, RetireRequest, + SealRequest, TailRequest, + }, + }, + recovery::manifest::RecoveryManifestStore, +}; +use ed25519_dalek::SigningKey; +use futures_util::future::BoxFuture; +use std::sync::atomic::{AtomicUsize, Ordering}; + +#[cfg(unix)] +mod failed_boot; +mod tests; + +const NOW: i64 = 1_000_000; +const CHECK: i64 = NOW + 10_005; +fn deadline() -> Instant { + Instant::now() + Duration::from_secs(5) +} + +struct Members { + peers: Vec<(NodeId, LocalRecoveredFollowerTransport)>, + retirements: AtomicUsize, +} +impl Members { + fn peer(&self, member: NodeId) -> &LocalRecoveredFollowerTransport { + &self + .peers + .iter() + .find(|(node, _)| *node == member) + .unwrap() + .1 + } +} +impl NodeLogTransport for Members { + fn append<'a>( + &'a self, + member: NodeId, + request: AppendRequest, + ) -> BoxFuture<'a, cellule_runtime::Result> { + self.peer(member).append(member, request) + } + fn seal<'a>( + &'a self, + member: NodeId, + request: SealRequest, + ) -> BoxFuture<'a, cellule_runtime::Result> { + self.peer(member).seal(member, request) + } + fn retire<'a>( + &'a self, + member: NodeId, + request: RetireRequest, + ) -> BoxFuture<'a, cellule_runtime::Result> { + self.peer(member).retire(member, request) + } + fn tail<'a>( + &'a self, + member: NodeId, + request: TailRequest, + ) -> BoxFuture<'a, cellule_runtime::Result>> { + self.peer(member).tail(member, request) + } +} +impl RecoveredNodeLogTransport for Members { + fn retire_recovered<'a>( + &'a self, + member: NodeId, + request: RecoveredRetireRequest, + ) -> BoxFuture<'a, cellule_runtime::Result> { + self.retirements.fetch_add(1, Ordering::AcqRel); + self.peer(member).retire_recovered(member, request) + } +} + +struct Fixture { + directory: NodeDirectory, + journal: Arc, + path: PathBuf, + transport: Arc, + sealed: SealedNodeLog, + originals: Vec, + stores: Vec, + #[cfg(unix)] + process_request: Option, + _root: tempfile::TempDir, +} +impl Fixture { + async fn new() -> Self { + Self::with_process_observation(false).await + } + + async fn with_process_observation(observe: bool) -> Self { + #[cfg(not(unix))] + assert!(!observe, "process lifetime stand-in requires Unix"); + let root = tempfile::tempdir().unwrap(); + let path = root.path().join("journal.sqlite"); + let journal = Arc::new( + SqliteJournal::open(path.clone(), scope(), FleetProfile::default(), NOW) + .await + .unwrap(), + ); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + ObjectPath::from("recovered-enrollment"), + [3; 16], + ); + let directory = NodeDirectory::new( + layout.clone(), + scope().fleet, + Digest::from_bytes([31; 32]), + Digest::from_bytes([32; 32]), + ); + let mut intents = Vec::new(); + for index in 0..3 { + let intent = journal + .register_initial_intent( + &NodeIntent::initial(scope(), node_id(index), session(index)).unwrap(), + ) + .await + .unwrap(); + let issued = if index == 0 { NOW } else { NOW + 9_000 }; + let ad = NodeAdvertisement::sign( + node_id(index), + session(index), + owner(index).endpoint, + scope().fleet, + Digest::from_bytes([30; 32]), + Digest::from_bytes([31; 32]), + Digest::from_bytes([32; 32]), + &SigningKey::from_bytes(&[index as u8 + 1; 32]), + 1, + issued, + issued + 10_000, + vec![Digest::from_bytes([33; 32])], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + free_memory_bytes: 1 << 20, + free_disk_bytes: 1 << 20, + follower_free_bytes: 1 << 20, + job_credits: 4, + log_protocol: 1, + ..Default::default() + }, + ) + .unwrap(); + startup::enroll( + journal.as_ref(), + &directory, + &startup::spec(&intent).unwrap(), + ad, + issued, + ) + .await + .unwrap(); + intents.push(intent); + } + let source = directory + .load(session(0), NOW + 9_001) + .await + .unwrap() + .unwrap(); + let prepared = directory + .prepare_log_enrollment(&source, 4, 1, 3, NOW + 9_001) + .await + .unwrap() + .unwrap(); + assert_eq!(prepared.followers().len(), 2); + let attempt = directory + .prepare_log_enrollment_attempt(&prepared, NOW + 9_001) + .await + .unwrap(); + let mut originals = Vec::new(); + for (index, member) in prepared.followers().iter().enumerate() { + let target = intents + .iter() + .find(|intent| intent.node() == member.node()) + .unwrap(); + let spec = EnrollmentSpec { + scope: scope(), + request: Digest::from_bytes([index as u8 + 90; 32]), + source: Some(EnrollmentEndpoint { + node: node_id(0), + session: session(0), + intent_revision: 1, + }), + target: EnrollmentEndpoint { + node: member.node(), + session: member.session(), + intent_revision: target.revision(), + }, + role: EnrollmentRole::Follower { log_epoch: 4 }, + }; + let FleetEnrollmentAcceptance::New(row) = + journal.accept_enrollment(&spec, NOW + 9_001).await.unwrap() + else { + panic!("new enrollment expected") + }; + originals.push(row); + } + directory + .commit_log_enrollment(&attempt, NOW + 9_001) + .await + .unwrap(); + // Keep the second establishment reply unresolved. Canonical retirement + // must cover Pending as well as the known Established original request. + originals[0] = journal + .publish_enrollment_result( + &originals[0], + EnrollmentEvent::Established(Digest::from_bytes([70; 32])), + NOW + 9_002, + ) + .await + .unwrap(); + let version = journal.load_snapshot(scope()).await.unwrap().registry(); + journal.bootstrap_registry(version).await.unwrap(); + let mut stores = Vec::new(); + let mut peers = Vec::new(); + for (index, member) in prepared.log().members().iter().copied().enumerate() { + let store = FollowerStore::open( + root.path().join(format!("follower-{index}")), + Limits::default(), + DiskBudget::new(1 << 30), + ) + .unwrap(); + peers.push(( + member, + LocalRecoveredFollowerTransport::new( + LocalFollowerTransport::new(member, store.clone()), + directory.clone(), + session(1), + || Ok(CHECK), + ) + .unwrap(), + )); + stores.push(store); + } + let transport = Arc::new(Members { + peers, + retirements: AtomicUsize::new(0), + }); + #[cfg(unix)] + let mut process = + observe.then(|| failed_boot::Process::start(root.path().join("process-closure"))); + let fenced = directory + .claim_expired(session(0), session(1), NOW + 10_001) + .await + .unwrap(); + #[cfg(unix)] + let process_request = if let Some(process) = &mut process { + assert_eq!( + fenced.log().unwrap().phase(), + cellule_runtime::node::log_state::NodeLogPhase::Recovering + ); + let snapshot = journal.load_snapshot(scope()).await.unwrap(); + let roster = FleetRoster::collect(journal.as_ref(), &snapshot, deadline()) + .await + .unwrap(); + let original = journal + .load_enrollment(scope(), startup::spec(&intents[0]).unwrap().key().unwrap()) + .await + .unwrap() + .unwrap(); + assert!( + cellule_host::fleet::FleetFailedBootProcessRequest::capture( + journal.as_ref(), + &directory, + &roster, + &original, + session(1), + deadline(), + || Ok(NOW + 10_001) + ) + .await + .is_err() + ); + let request = cellule_host::fleet::FleetFailedBootProcessRequest::capture_fenced( + journal.as_ref(), + &directory, + &roster, + &original, + session(1), + deadline(), + || Ok(NOW + 10_001), + ) + .await + .unwrap(); + assert!(request.canonical().is_none()); + process.stop_and_retain(&request); + let confirmation = request + .confirm( + journal.as_ref(), + &directory, + &failed_boot::Processes::new(root.path().join("process-closure")), + session(1), + deadline(), + || Ok(NOW + 10_001), + ) + .await + .unwrap(); + assert_eq!(confirmation.fence(), request.fence()); + assert_eq!(confirmation.snapshot(), roster.snapshot()); + assert_eq!(confirmation.process().request_digest(), request.digest()); + assert_eq!(confirmation.interval(), (NOW + 10_001, NOW + 10_001)); + Some(request) + } else { + None + }; + let recovery = + NodeLogRecovery::from_fenced(transport.clone(), &fenced, Limits::default()).unwrap(); + let completed = RecoveryCoordinator::new( + recovery, + RecoveryManifestStore::new(layout, Limits::default()), + ) + .recover_and_seal(&directory, fenced, Vec::new(), NOW + 10_002) + .await + .unwrap(); + assert!(completed.controls.is_empty()); + Self { + directory, + journal, + path, + transport, + sealed: completed.sealed, + originals, + stores, + #[cfg(unix)] + process_request, + _root: root, + } + } + async fn retire(&self) { + let proof = retire_recovered_members(self.transport.clone(), &self.sealed) + .await + .unwrap() + .confirmed() + .unwrap(); + self.directory + .retire_recovered_log(&proof, session(1), CHECK) + .await + .unwrap(); + } + async fn roster(&self) -> FleetRoster { + let snapshot = self.journal.load_snapshot(scope()).await.unwrap(); + FleetRoster::collect(self.journal.as_ref(), &snapshot, deadline()) + .await + .unwrap() + } + async fn capture(&self) -> FleetRecoveredFollowerRetirement { + FleetRecoveredFollowerRetirement::capture( + self.journal.as_ref(), + &self.directory, + &self.roster().await, + &self.sealed, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .unwrap() + } + async fn reconstruct(&self) -> SqliteJournal { + SqliteJournal::open( + self.path.clone(), + scope(), + FleetProfile::default(), + CHECK + 10, + ) + .await + .unwrap() + } +} diff --git a/crates/cellule-host/minion/scenario/recovered_followers/tests.rs b/crates/cellule-host/minion/scenario/recovered_followers/tests.rs new file mode 100644 index 00000000..a13735f7 --- /dev/null +++ b/crates/cellule-host/minion/scenario/recovered_followers/tests.rs @@ -0,0 +1,469 @@ +use super::*; + +#[tokio::test] +async fn recovered_publication_cancelled_waiter_joins_backend_work_then_recaptures_original_records() + { + let fixture = Fixture::new().await; + fixture.retire().await; + let capture = fixture.capture().await; + let (entered, resume) = fixture.journal.pause_next_enrollment_reply(true, false); + let mut work = Box::pin(capture.publish( + fixture.journal.as_ref(), + &fixture.directory, + session(1), + deadline(), + || Ok(CHECK), + )); + tokio::select! { + result = &mut work => panic!("publication completed before held reply: {}", result.is_ok()), + result = entered => result.unwrap(), + } + drop(work); + assert!(resume.send(()).is_err()); + // The reference adapter joins every accepted blocking transaction before + // closing its SQLite handle, even though its caller no longer awaits it. + fixture.journal.close().await.unwrap(); + let journal = fixture.reconstruct().await; + let snapshot = journal.load_snapshot(scope()).await.unwrap(); + let roster = FleetRoster::collect(&journal, &snapshot, deadline()) + .await + .unwrap(); + let committed = roster + .enrollments() + .iter() + .filter(|row| { + matches!(row.spec().role, EnrollmentRole::Follower { .. }) + && row.status() == EnrollmentStatus::Retired + }) + .cloned() + .collect::>(); + assert!(!committed.is_empty()); + let replay = FleetRecoveredFollowerRetirement::capture( + &journal, + &fixture.directory, + &roster, + &fixture.sealed, + session(1), + deadline(), + || Ok(CHECK + 10), + ) + .await + .unwrap(); + let repeated = replay + .publish(&journal, &fixture.directory, session(1), deadline(), || { + Ok(CHECK + 10) + }) + .await + .unwrap(); + let closure = repeated.confirmed().unwrap(); + for original in committed { + assert_eq!( + closure + .members() + .iter() + .find(|row| row.spec() == original.spec()) + .unwrap(), + &original + ); + } + assert_eq!(fixture.transport.retirements.load(Ordering::Acquire), 2); +} + +#[tokio::test] +async fn recovered_publication_final_barrier_rejects_a_delayed_new_original_epoch_request() { + let fixture = Fixture::new().await; + fixture.retire().await; + let capture = fixture.capture().await; + let (entered, resume) = fixture.journal.pause_next_enrollment_reply(true, false); + let mut work = Box::pin(capture.publish( + fixture.journal.as_ref(), + &fixture.directory, + session(1), + deadline(), + || Ok(CHECK), + )); + tokio::select! { + result = &mut work => panic!("publication completed before held reply: {}", result.is_ok()), + result = entered => result.unwrap(), + } + let mut delayed = fixture.originals[0].spec().clone(); + delayed.request = Digest::from_bytes([98; 32]); + fixture + .journal + .accept_enrollment(&delayed, CHECK) + .await + .unwrap(); + resume.send(()).unwrap(); + let result = work.await.unwrap(); + assert!( + result + .members() + .iter() + .all(|member| member.result().is_ok()) + ); + assert!(result.confirmed().is_err()); + assert!(matches!( + result.closure_error().as_deref(), + Some(Error::Control( + "recovered follower enrollment set is incomplete" + )) + )); + let current = fixture + .journal + .load_enrollment(scope(), delayed.key().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(current.status(), EnrollmentStatus::Pending); + assert_eq!(fixture.transport.retirements.load(Ordering::Acquire), 2); +} + +#[tokio::test] +async fn recovered_publication_settles_every_original_request_and_replays_after_native_collection() +{ + let fixture = Fixture::new().await; + fixture.retire().await; + let capture = fixture.capture().await; + assert_eq!(capture.leader_node(), node_id(0)); + assert_eq!(capture.members(), fixture.originals); + let published = capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + session(1), + deadline(), + || Ok(CHECK), + ) + .await + .unwrap(); + let closure = published.confirmed().unwrap(); + assert_eq!(closure.leader_node(), node_id(0)); + assert_eq!(closure.retired().session(), session(0)); + assert_eq!(closure.members().len(), 2); + let records = closure.members().to_vec(); + for (original, retired) in fixture.originals.iter().zip(&records) { + assert_eq!(retired.spec(), original.spec()); + assert_eq!(retired.accepted_at_ms(), original.accepted_at_ms()); + assert_eq!( + retired.established_evidence(), + original.established_evidence() + ); + assert_eq!(retired.status(), EnrollmentStatus::Retired); + assert_eq!(retired.updated_at_ms(), CHECK); + } + assert!( + !fixture + .directory + .log_epoch_referenced(session(0), 4) + .await + .unwrap() + ); + for store in &fixture.stores { + let lanes = store.retired_lanes(i64::MAX, 2).await.unwrap(); + assert_eq!(lanes.len(), 1); + assert!( + store + .remove_retired(lanes[0], lanes[0].retired_at_ms()) + .await + .unwrap() + ); + assert_eq!(store.retained_bytes(), 0); + } + let journal = fixture.reconstruct().await; + let snapshot = journal.load_snapshot(scope()).await.unwrap(); + let roster = FleetRoster::collect(&journal, &snapshot, deadline()) + .await + .unwrap(); + assert!( + roster + .required_boots() + .iter() + .any(|boot| boot.node == node_id(0) && boot.session == session(0)) + ); + let replay = FleetRecoveredFollowerRetirement::capture( + &journal, + &fixture.directory, + &roster, + &fixture.sealed, + session(1), + deadline(), + || Ok(CHECK + 10), + ) + .await + .unwrap(); + let repeated = replay + .publish(&journal, &fixture.directory, session(1), deadline(), || { + Ok(CHECK + 10) + }) + .await + .unwrap(); + let repeated = repeated.confirmed().unwrap(); + assert_eq!(repeated.members(), records); + assert_eq!(repeated.digest(), closure.digest()); + assert_eq!(repeated.snapshot(), closure.snapshot()); + assert_eq!(fixture.transport.retirements.load(Ordering::Acquire), 2); +} + +#[tokio::test] +async fn recovered_publication_joins_healthy_siblings_and_retains_lost_reply_across_reconstruction() +{ + let fixture = Fixture::new().await; + fixture.retire().await; + let capture = fixture.capture().await; + let (entered, resume) = fixture.journal.pause_next_enrollment_reply(true, true); + let mut work = Box::pin(capture.publish( + fixture.journal.as_ref(), + &fixture.directory, + session(1), + deadline(), + || Ok(CHECK), + )); + tokio::select! { + result = &mut work => panic!("publication completed before held reply: {}", result.is_ok()), + result = entered => result.unwrap(), + } + resume.send(()).unwrap(); + let result = work.await.unwrap(); + assert_eq!(result.members().len(), 2); + let failed = result + .members() + .iter() + .position(|member| member.result().is_err()) + .unwrap(); + assert_eq!( + result + .members() + .iter() + .filter(|member| member.result().is_err()) + .count(), + 1 + ); + let original_error = result.members()[failed].result().unwrap_err(); + let source = std::error::Error::source(original_error.as_ref()).unwrap(); + assert_eq!( + source.downcast_ref::().unwrap().kind(), + std::io::ErrorKind::Other + ); + assert!( + source + .to_string() + .contains("injected lost enrollment reply after durable commit") + ); + assert!(result.members()[1 - failed].result().is_ok()); + assert!(result.confirmed().is_err()); + assert!(Arc::ptr_eq( + &original_error, + &result.members()[failed].result().unwrap_err() + )); + assert!( + capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + session(1), + deadline(), + || Ok(CHECK + 1) + ) + .await + .is_err() + ); + let journal = fixture.reconstruct().await; + let snapshot = journal.load_snapshot(scope()).await.unwrap(); + let roster = FleetRoster::collect(&journal, &snapshot, deadline()) + .await + .unwrap(); + let replay = FleetRecoveredFollowerRetirement::capture( + &journal, + &fixture.directory, + &roster, + &fixture.sealed, + session(1), + deadline(), + || Ok(CHECK + 2), + ) + .await + .unwrap(); + let repeated = replay + .publish(&journal, &fixture.directory, session(1), deadline(), || { + Ok(CHECK + 2) + }) + .await + .unwrap(); + let closure = repeated.confirmed().unwrap(); + assert!( + closure + .members() + .iter() + .all(|row| row.updated_at_ms() == CHECK) + ); + assert!(Arc::ptr_eq( + &original_error, + &result.members()[failed].result().unwrap_err() + )); + assert_eq!(fixture.transport.retirements.load(Ordering::Acquire), 2); +} + +#[tokio::test] +async fn recovered_publication_refuses_unretired_authority_stale_rosters_and_expired_claimants() { + let fixture = Fixture::new().await; + let roster = fixture.roster().await; + assert!( + FleetRecoveredFollowerRetirement::capture( + fixture.journal.as_ref(), + &fixture.directory, + &roster, + &fixture.sealed, + session(1), + deadline(), + || Ok(CHECK) + ) + .await + .is_err() + ); + assert_eq!(fixture.transport.retirements.load(Ordering::Acquire), 0); + fixture.retire().await; + let capture = fixture.capture().await; + let before = fixture.journal.load_snapshot(scope()).await.unwrap(); + fixture + .journal + .register_initial_intent( + &NodeIntent::initial( + scope(), + NodeId::from_bytes([99; 16]), + SessionId::from_bytes([99; 16]), + ) + .unwrap(), + ) + .await + .unwrap(); + assert_ne!( + fixture + .journal + .load_snapshot(scope()) + .await + .unwrap() + .registry(), + before.registry() + ); + assert!( + capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + session(1), + deadline(), + || Ok(CHECK) + ) + .await + .is_err() + ); + let roster = fixture.roster().await; + for original in &fixture.originals { + assert_eq!( + roster + .enrollments() + .iter() + .find(|row| row.spec() == original.spec()) + .unwrap(), + original + ); + } + assert!( + FleetRecoveredFollowerRetirement::capture( + fixture.journal.as_ref(), + &fixture.directory, + &roster, + &fixture.sealed, + session(1), + deadline(), + || Ok(NOW + 19_000) + ) + .await + .is_err() + ); + assert_eq!(fixture.transport.retirements.load(Ordering::Acquire), 2); +} + +#[tokio::test] +async fn recovered_publication_refuses_duplicate_original_member_requests_before_effects() { + let fixture = Fixture::new().await; + fixture.retire().await; + let mut duplicate = fixture.originals[0].spec().clone(); + duplicate.request = Digest::from_bytes([99; 32]); + fixture + .journal + .accept_enrollment(&duplicate, CHECK) + .await + .unwrap(); + assert!( + FleetRecoveredFollowerRetirement::capture( + fixture.journal.as_ref(), + &fixture.directory, + &fixture.roster().await, + &fixture.sealed, + session(1), + deadline(), + || Ok(CHECK) + ) + .await + .is_err() + ); + let roster = fixture.roster().await; + assert_eq!( + roster + .enrollments() + .iter() + .filter(|row| matches!(row.spec().role, EnrollmentRole::Follower { .. })) + .count(), + 3 + ); + assert!( + !roster + .enrollments() + .iter() + .any(|row| row.status() == EnrollmentStatus::Retired) + ); +} + +#[tokio::test] +async fn recovered_publication_clock_regression_retains_committed_rows_without_closure() { + let fixture = Fixture::new().await; + fixture.retire().await; + let capture = fixture.capture().await; + let mut calls = 0; + let result = capture + .publish( + fixture.journal.as_ref(), + &fixture.directory, + session(1), + deadline(), + || { + calls += 1; + Ok(if calls == 1 { CHECK + 2 } else { CHECK + 1 }) + }, + ) + .await + .unwrap(); + assert!( + result + .members() + .iter() + .all(|member| member.result().is_ok()) + ); + assert!(result.confirmed().is_err()); + assert!(matches!( + result.closure_error().as_deref(), + Some(Error::Deadline) + )); + assert!( + fixture + .roster() + .await + .enrollments() + .iter() + .filter(|row| matches!(row.spec().role, EnrollmentRole::Follower { .. })) + .all( + |row| row.status() == EnrollmentStatus::Retired && row.updated_at_ms() == CHECK + 2 + ) + ); +} diff --git a/crates/cellule-host/minion/scenario/startup/mod.rs b/crates/cellule-host/minion/scenario/startup/mod.rs new file mode 100644 index 00000000..7776a5a8 --- /dev/null +++ b/crates/cellule-host/minion/scenario/startup/mod.rs @@ -0,0 +1,371 @@ +//! Application-owned boot producer: Pending precedes canonical advertisement. + +use super::*; +use cellule_host::fleet::FleetEnrollmentAcceptance; +use cellule_runtime::fleet::operations::{ + EnrollmentEndpoint, EnrollmentEvent, EnrollmentRecord, EnrollmentRole, EnrollmentSpec, + EnrollmentStatus, +}; +use cellule_runtime::node::{NodeAdvertisement, NodeCapacity, NodeDirectory, NodeFailureDomain}; +use ed25519_dalek::SigningKey; + +pub(super) fn spec(intent: &NodeIntent) -> JournalResult { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.example-fleet-boot-request.v1\0"); + hash.update(intent.scope().fleet.as_bytes()); + hash.update(intent.scope().application.as_bytes()); + hash.update(intent.node().as_bytes()); + hash.update(intent.session().as_bytes()); + Ok(EnrollmentSpec { + scope: intent.scope(), + request: Digest::from_bytes(*hash.finalize().as_bytes()), + role: EnrollmentRole::Node { + mode: intent.mode(), + }, + source: None, + target: EnrollmentEndpoint { + node: intent.node(), + session: intent.session(), + intent_revision: intent.revision(), + }, + }) +} + +pub(super) async fn advertisement( + index: usize, + node: &CellNode, + intent: &NodeIntent, +) -> JournalResult { + // Boot discovery publishes no receive capacity before readiness. The + // observer separately captures real signed capacity after startup. + let sample = tokio::time::timeout(Duration::from_secs(3), async { + loop { + if let Some(sample) = node.runtime().operational_sample()? { + return Ok::<_, cellule_runtime::Error>(sample); + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await??; + let now = clock()?; + let stats = node.stats(); + let key = SigningKey::from_bytes(&[index as u8 + 1; 32]); + Ok(NodeAdvertisement::sign( + intent.node(), + intent.session(), + owner(index).endpoint, + intent.scope().fleet, + Digest::from_bytes([30; 32]), + Digest::from_bytes([31; 32]), + node.application().registry().release_digest(), + &key, + 1, + now, + now + 30_000, + node.application().registry().module_digests(), + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + log_protocol: 1, + ..NodeCapacity::default() + }, + )? + .with_operational_placement( + cellule_runtime::node::NodePlacementCapacity { + memory_capacity_bytes: u64::try_from( + stats.resident_capacity_bytes() + stats.retained_capacity_bytes(), + )?, + disk_capacity_bytes: stats.local_disk_capacity_bytes(), + active_cells: stats.placement_active_cells(), + max_active_cells: stats.placement_active_cell_capacity(), + running_jobs: stats.placement_running_jobs(), + job_capacity: stats.placement_job_capacity(), + ..Default::default() + }, + sample, + &key, + )?) +} + +fn evidence(spec: &EnrollmentSpec, ad: &NodeAdvertisement) -> JournalResult { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.example-fleet-boot-evidence.v1\0"); + hash.update(&spec.to_bytes()?); + hash.update(ad.node().as_bytes()); + hash.update(ad.session().as_bytes()); + hash.update(ad.fleet().as_bytes()); + hash.update(ad.certificate().as_bytes()); + hash.update(ad.image().as_bytes()); + hash.update(ad.release().as_bytes()); + hash.update(&ad.verifying_key()?.to_bytes()); + hash.update(&(ad.endpoint().len() as u64).to_be_bytes()); + hash.update(ad.endpoint().as_bytes()); + hash.update(&ad.generation().to_be_bytes()); + hash.update(&ad.issued_at_ms().to_be_bytes()); + hash.update(&ad.expires_at_ms().to_be_bytes()); + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) +} + +pub(super) async fn enroll( + journal: &SqliteJournal, + directory: &NodeDirectory, + spec: &EnrollmentSpec, + ad: NodeAdvertisement, + now: i64, +) -> JournalResult { + if ad.node() != spec.target.node + || ad.session() != spec.target.session + || ad.fleet() != spec.scope.fleet + || !matches!(spec.role, EnrollmentRole::Node { .. }) + || spec.source.is_some() + { + return Err(invalid("example boot advertisement binding differs")); + } + let (record, observed) = match journal.accept_enrollment(spec, now).await? { + FleetEnrollmentAcceptance::New(record) => { + (record, directory.create(ad.clone(), now).await?) + } + FleetEnrollmentAcceptance::Existing(record) => { + if !matches!( + record.status(), + EnrollmentStatus::Pending | EnrollmentStatus::Established + ) { + return Err(invalid("example boot enrollment is already settled")); + } + // Existing Pending cannot authorize another unobserved side effect. + // Inspect the canonical advertisement; absence retains the blocker. + let observed = directory + .load(ad.session(), now) + .await? + .ok_or_else(|| invalid("example pending boot has no proven advertisement"))?; + (record, observed) + } + }; + if observed.advertisement() != &ad { + return Err(invalid("example original boot advertisement changed")); + } + journal + .publish_enrollment_result( + &record, + EnrollmentEvent::Established(evidence(spec, observed.advertisement())?), + now, + ) + .await +} + +#[cfg(test)] +mod tests; + +/// Application retains the boot scope through joined runtime shutdown and +/// canonical withdrawal. Pending or failed boots stay in the journal unless +/// that exact directory obligation can be closed. +#[derive(Clone)] +pub(super) struct BootOwner { + pub(super) node: Arc, + pub(super) directory: NodeDirectory, + pub(super) spec: EnrollmentSpec, + pub(super) advertisement: NodeAdvertisement, + pub(super) guard: Option, +} + +impl BootOwner { + /// Refresh only the retained canonical boot. No missing/expired record can + /// authorize recreation, and local lease credit advances only after CAS. + pub(super) async fn refresh_capacity( + &self, + index: usize, + journal: &dyn FleetEnrollmentJournal, + deadline: Instant, + ) -> JournalResult { + if !self.node.is_management_ready() { + return Err(invalid("example capacity node is not management ready")); + } + let guard = self + .guard + .as_ref() + .ok_or_else(|| invalid("example boot lease guard is unbound"))?; + guard.check()?; + // A live boot must consume retained maintenance intent before another + // membership/lease renewal. Lost Cordon RPCs cannot keep its gate open. + self.node + .refresh_fleet_intent(journal, deadline.into_std()) + .await?; + let now = clock()?; + let observed = self + .directory + .load(self.spec.target.session, now) + .await? + .ok_or_else(|| invalid("example capacity boot is missing or expired"))?; + self.validate_successor(observed.advertisement())?; + let previous = observed.advertisement().operational_sample(); + // A new capacity block cannot reuse an earlier classifier sequence. + // Wait for its real sample; neither sequence nor time is fabricated. + let sample = tokio::time::timeout_at(deadline, async { + loop { + if let Some(sample) = self.node.runtime().operational_sample()? + && previous.is_none_or(|old| sample.sequence > old.sequence) + { + return Ok::<_, cellule_runtime::Error>(sample); + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await??; + let stats = self.node.stats(); + let memory = + u64::try_from(stats.resident_capacity_bytes() + stats.retained_capacity_bytes())?; + let used = u64::try_from(stats.resident_bytes() + stats.retained_bytes())?; + let follower = self + .node + .try_owned_component::( + cellule_host::FOLLOWER_STORE_COMPONENT, + )?; + let original = &self.advertisement; + let key = SigningKey::from_bytes(&[index as u8 + 1; 32]); + let now = clock()?; + let next = NodeAdvertisement::sign( + original.node(), + original.session(), + original.endpoint().to_owned(), + original.fleet(), + original.certificate(), + original.image(), + original.release(), + &key, + 1, + now, + now.checked_add(30_000) + .ok_or_else(|| invalid("example heartbeat deadline overflow"))?, + original.module_digests().to_vec(), + original.peer_versions().to_vec(), + original.failure_domain().clone(), + NodeCapacity { + follower_free_bytes: follower + .as_ref() + .map_or(0, |store| store.available_bytes()) + .min( + stats + .local_disk_capacity_bytes() + .saturating_sub(stats.local_disk_reserved_bytes()), + ), + follower_retained_bytes: follower + .as_ref() + .map_or(0, |store| store.retained_bytes()), + free_memory_bytes: memory.saturating_sub(used), + free_disk_bytes: stats + .local_disk_capacity_bytes() + .saturating_sub(stats.local_disk_reserved_bytes()), + job_credits: stats + .placement_job_capacity() + .saturating_sub(stats.placement_running_jobs()), + log_protocol: 1, + }, + )? + .with_operational_placement( + cellule_runtime::node::NodePlacementCapacity { + memory_capacity_bytes: memory, + disk_capacity_bytes: stats.local_disk_capacity_bytes(), + active_cells: stats.placement_active_cells(), + max_active_cells: stats.placement_active_cell_capacity(), + running_jobs: stats.placement_running_jobs(), + job_capacity: stats.placement_job_capacity(), + ..Default::default() + }, + sample, + &key, + )?; + guard.check()?; + let refreshed = self.directory.refresh(&observed, next, now).await?; + self.validate_successor(refreshed.advertisement())?; + guard.renew(clock()?, refreshed.advertisement().expires_at_ms())?; + Ok(refreshed.advertisement().clone()) + } + + fn validate_successor(&self, ad: &NodeAdvertisement) -> JournalResult<()> { + let original = &self.advertisement; + // Pin the original enrolled signing key and executable identity; a + // self-valid signature on an unrelated advertisement is insufficient. + if ad.node() != original.node() + || ad.session() != original.session() + || ad.endpoint() != original.endpoint() + || ad.fleet() != original.fleet() + || ad.certificate() != original.certificate() + || ad.image() != original.image() + || ad.release() != original.release() + || ad.verifying_key()? != original.verifying_key()? + || ad.module_digests() != original.module_digests() + || ad.peer_versions() != original.peer_versions() + || ad.failure_domain() != original.failure_domain() + || ad.generation() < original.generation() + || ad.issued_at_ms() < original.issued_at_ms() + || ad.expires_at_ms() < original.expires_at_ms() + { + return Err(invalid("example retained boot identity differs")); + } + Ok(()) + } + + pub(super) async fn withdraw(&self, journal: &SqliteJournal) -> JournalResult<()> { + if self.node.state() != NodeState::Stopped { + return Err(invalid("example boot runtime has not joined shutdown")); + } + if let Some(guard) = &self.guard { + guard.fence(); + } + let Some(record) = journal + .load_enrollment(self.spec.scope, self.spec.key()?) + .await? + else { + return Ok(()); + }; + record.validate_replay(&self.spec)?; + if matches!( + record.status(), + EnrollmentStatus::Retired | EnrollmentStatus::Refused + ) { + return Ok(()); + } + let original_evidence = evidence(&self.spec, &self.advertisement)?; + if record + .established_evidence() + .is_some_and(|evidence| evidence != original_evidence) + { + return Err(invalid("example boot withdrawal original evidence differs")); + } + // A previous withdrawal can commit before its reply or journal result + // is observed. The permanent exact-session tombstone closes that boot; + // an absent live advertisement by itself never supplies closure. + if !self + .directory + .is_withdrawn(self.spec.target.session) + .await? + { + let now = clock()?; + let observed = self + .directory + .load(self.spec.target.session, now) + .await? + .ok_or_else(|| invalid("example boot withdrawal remains unresolved"))?; + self.validate_successor(observed.advertisement())?; + self.directory.withdraw_after_drain(&observed, now).await?; + } + if !self + .directory + .is_withdrawn(self.spec.target.session) + .await? + { + return Err(invalid("example boot withdrawal lacks its tombstone")); + } + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.example-fleet-boot-retirement.v1\0"); + hash.update(original_evidence.as_bytes()); + journal + .publish_enrollment_result( + &record, + EnrollmentEvent::Retired(Digest::from_bytes(*hash.finalize().as_bytes())), + clock()?, + ) + .await?; + Ok(()) + } +} diff --git a/crates/cellule-host/minion/scenario/startup/tests/mod.rs b/crates/cellule-host/minion/scenario/startup/tests/mod.rs new file mode 100644 index 00000000..924173ad --- /dev/null +++ b/crates/cellule-host/minion/scenario/startup/tests/mod.rs @@ -0,0 +1,1172 @@ +use super::*; +use cellule_host::fleet::FleetJournal; +use cellule_runtime::Error; +use cellule_runtime::fleet::operations::*; +use cellule_runtime::node::NodeMode; + +struct Fixture { + root: tempfile::TempDir, + journal: Arc, + node: Arc, + directory: NodeDirectory, + layout: CellStorageLayout, + intent: NodeIntent, + ad: NodeAdvertisement, +} + +async fn fixture() -> Fixture { + let root = tempfile::tempdir().unwrap(); + let now = clock().unwrap(); + let journal = Arc::new( + SqliteJournal::open( + root.path().join("startup.sqlite"), + scope(), + FleetProfile::default(), + now, + ) + .await + .unwrap(), + ); + let intent = journal + .register_initial_intent(&NodeIntent::initial(scope(), node_id(0), session(0)).unwrap()) + .await + .unwrap(); + let node = build(&intent, session(0)); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + ObjectPath::from("boot"), + [3; 16], + ); + let directory = NodeDirectory::new( + layout.clone(), + scope().fleet, + Digest::from_bytes([31; 32]), + node.application().registry().release_digest(), + ); + let ad = advertisement(0, &node, &intent).await.unwrap(); + Fixture { + root, + journal, + node, + directory, + layout, + intent, + ad, + } +} + +fn build(intent: &NodeIntent, boot: SessionId) -> Arc { + let node = Arc::new( + CellNodeBuilder::new(super::super::application::compile().unwrap()) + .with_runtime(SqlWorkerPool::new(1, 10).unwrap(), 16 << 20) + .with_replica_host(Host::default()) + .with_session(boot) + .with_fleet_startup_intent(intent.clone()) + .build() + .unwrap(), + ); + node.install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + let now = clock().unwrap(); + node.install_node_lease_for_startup(NodeLeaseGuard::new(now, now + 60_000).unwrap()) + .unwrap(); + node +} + +async fn close(fixture: Fixture) { + fixture.node.shutdown().await.unwrap(); + assert_eq!(fixture.node.state(), NodeState::Stopped); + assert_eq!(fixture.node.stats().retained_bytes(), 0); + assert_eq!(fixture.node.stats().active_cells(), 0); + fixture.journal.close().await.unwrap(); +} + +async fn maintenance(fixture: &Fixture) -> cellule_host::fleet::FleetJournalSnapshot { + let now = clock().unwrap(); + let old = fixture.journal.load_snapshot(scope()).await.unwrap(); + let controller = fixture + .journal + .claim_controller( + scope(), + old.head().revision(), + SessionId::from_bytes([206; 16]), + now, + ) + .await + .unwrap(); + fixture + .journal + .compare_exchange( + &controller, + controller.head().controller().unwrap().epoch, + now, + &JournalTransition::BeginMaintenance( + MaintenanceOperation::new( + OperationId::from_bytes([80; 16]).unwrap(), + Digest::from_bytes([81; 32]), + node_id(0), + session(0), + 2, + now, + now + 60_000, + ) + .unwrap(), + ), + ) + .await + .unwrap() +} + +#[tokio::test] +async fn fleet_boot_stays_closed_for_missing_pending_or_foreign_role_records() { + let fixture = fixture().await; + let spec = spec(&fixture.intent).unwrap(); + assert!(matches!( + fixture.node.runtime().node_admission().check_new_role(), + Err(Error::CellDraining) + )); + assert!(fixture.node.start().is_err()); + assert!(!fixture.node.is_ready() && !fixture.node.is_management_ready()); + assert!( + fixture + .node + .confirm_fleet_startup(fixture.journal.as_ref(), spec.key().unwrap()) + .await + .is_err() + ); + fixture + .journal + .accept_enrollment(&spec, clock().unwrap()) + .await + .unwrap(); + assert!( + fixture + .node + .confirm_fleet_startup(fixture.journal.as_ref(), spec.key().unwrap()) + .await + .is_err() + ); + assert!(fixture.node.start().is_err()); + let source = fixture + .journal + .register_initial_intent(&NodeIntent::initial(scope(), node_id(1), session(1)).unwrap()) + .await + .unwrap(); + let role = EnrollmentSpec { + request: Digest::from_bytes([90; 32]), + role: EnrollmentRole::Follower { log_epoch: 1 }, + source: Some(EnrollmentEndpoint { + node: source.node(), + session: source.session(), + intent_revision: source.revision(), + }), + ..spec.clone() + }; + let FleetEnrollmentAcceptance::New(record) = fixture + .journal + .accept_enrollment(&role, clock().unwrap()) + .await + .unwrap() + else { + panic!("role duplicate") + }; + fixture + .journal + .publish_enrollment_result( + &record, + EnrollmentEvent::Established(Digest::from_bytes([91; 32])), + clock().unwrap(), + ) + .await + .unwrap(); + assert!( + fixture + .node + .confirm_fleet_startup(fixture.journal.as_ref(), role.key().unwrap()) + .await + .is_err() + ); + assert!(fixture.node.start().is_err()); + assert_eq!(fixture.node.state(), NodeState::Starting); + assert!( + fixture + .node + .runtime() + .node_admission() + .check_new_role() + .is_err() + ); + close(fixture).await; +} + +#[tokio::test] +async fn canonical_boot_enrollment_and_lost_result_reply_open_only_the_original_boot() { + let fixture = fixture().await; + let now = clock().unwrap(); + let spec = spec(&fixture.intent).unwrap(); + let FleetEnrollmentAcceptance::New(record) = + fixture.journal.accept_enrollment(&spec, now).await.unwrap() + else { + panic!("boot duplicate") + }; + let observed = fixture + .directory + .create(fixture.ad.clone(), now) + .await + .unwrap(); + fixture.journal.lose_next_commit_reply(); + assert!( + fixture + .journal + .publish_enrollment_result( + &record, + EnrollmentEvent::Established(evidence(&spec, observed.advertisement()).unwrap()), + now + ) + .await + .is_err() + ); + let independent = SqliteJournal::open( + fixture.root.path().join("startup.sqlite"), + scope(), + FleetProfile::default(), + clock().unwrap(), + ) + .await + .unwrap(); + let checked = independent + .load_boot(scope(), node_id(0), spec.key().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(checked.enrollment().accepted_at_ms(), now); + let contradictory = NodeIntent::maintenance( + scope(), + &MaintenanceOperation::new( + OperationId::from_bytes([80; 16]).unwrap(), + Digest::from_bytes([81; 32]), + node_id(0), + session(0), + checked.intent().revision(), + now, + now + 60_000, + ) + .unwrap(), + ) + .unwrap(); + assert!(matches!( + cellule_host::fleet::FleetBootObservation::new(contradictory, checked.enrollment().clone()), + Err(OperationError::Conflict) + )); + let replay = enroll( + &independent, + &fixture.directory, + &spec, + fixture.ad.clone(), + clock().unwrap(), + ) + .await + .unwrap(); + assert_eq!(&replay, checked.enrollment()); + fixture + .node + .confirm_fleet_startup(&independent, spec.key().unwrap()) + .await + .unwrap(); + assert!(matches!( + fixture.node.runtime().node_admission().check_new_role(), + Err(Error::CellDraining) + )); + fixture + .node + .require_owned_components(["startup-probe"]) + .unwrap(); + assert!(fixture.node.start().is_err()); + assert_eq!(fixture.node.state(), NodeState::Starting); + assert!( + fixture + .node + .runtime() + .node_admission() + .check_new_role() + .is_err() + ); + fixture + .node + .install_owned_component("startup-probe", Arc::new(())) + .unwrap(); + fixture.node.start().unwrap(); + assert_eq!(fixture.node.state(), NodeState::Ready); + assert!(fixture.node.is_ready() && fixture.node.is_management_ready()); + assert!( + fixture + .node + .runtime() + .node_admission() + .check_new_role() + .is_ok() + ); + independent.close().await.unwrap(); + close(fixture).await; +} + +#[tokio::test] +async fn lost_acceptance_reply_retains_pending_boot_without_repeating_advertisement() { + let fixture = fixture().await; + let spec = spec(&fixture.intent).unwrap(); + fixture.journal.lose_next_commit_reply(); + assert!( + enroll( + &fixture.journal, + &fixture.directory, + &spec, + fixture.ad.clone(), + clock().unwrap() + ) + .await + .is_err() + ); + let original = fixture + .journal + .load_enrollment(scope(), spec.key().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(original.status(), EnrollmentStatus::Pending); + assert!( + fixture + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .is_none() + ); + assert!( + enroll( + &fixture.journal, + &fixture.directory, + &spec, + fixture.ad.clone(), + clock().unwrap() + ) + .await + .is_err() + ); + assert_eq!( + fixture + .journal + .load_enrollment(scope(), spec.key().unwrap()) + .await + .unwrap() + .unwrap(), + original + ); + assert!( + fixture + .node + .confirm_fleet_startup(fixture.journal.as_ref(), spec.key().unwrap()) + .await + .is_err() + ); + assert!(fixture.node.start().is_err()); + close(fixture).await; +} + +#[tokio::test] +async fn cordon_racing_accepted_boot_and_draining_reboot_keep_management_without_serving() { + let fixture = fixture().await; + let now = clock().unwrap(); + let boot_spec = spec(&fixture.intent).unwrap(); + let FleetEnrollmentAcceptance::New(record) = fixture + .journal + .accept_enrollment(&boot_spec, now) + .await + .unwrap() + else { + panic!("boot duplicate") + }; + let observed = fixture + .directory + .create(fixture.ad.clone(), now) + .await + .unwrap(); + let requested = maintenance(&fixture).await; + fixture + .journal + .publish_enrollment_result( + &record, + EnrollmentEvent::Established(evidence(&boot_spec, observed.advertisement()).unwrap()), + clock().unwrap(), + ) + .await + .unwrap(); + fixture + .node + .confirm_fleet_startup(fixture.journal.as_ref(), boot_spec.key().unwrap()) + .await + .unwrap(); + fixture.node.start().unwrap(); + assert_eq!(fixture.node.state(), NodeState::Maintenance); + assert!(!fixture.node.is_ready() && fixture.node.is_management_ready()); + assert_eq!( + fixture.node.runtime().node_admission().mode().unwrap(), + NodeMode::Draining + ); + fixture.node.start().unwrap(); + assert!(!fixture.node.is_ready()); + let predecessor = fixture + .journal + .load_boot(scope(), node_id(0), boot_spec.key().unwrap()) + .await + .unwrap() + .unwrap() + .intent() + .clone(); + fixture.node.shutdown().await.unwrap(); + fixture + .directory + .withdraw_after_drain(&observed, clock().unwrap()) + .await + .unwrap(); + assert!(fixture.directory.is_retired(session(0)).await.unwrap()); + fixture + .journal + .publish_enrollment_result( + &record, + EnrollmentEvent::Retired(Digest::from_bytes([92; 32])), + clock().unwrap(), + ) + .await + .unwrap(); + let new_session = SessionId::from_bytes([99; 16]); + let reboot = build(&predecessor, new_session); + let now = clock().unwrap(); + assert!( + fixture + .journal + .compare_exchange( + &requested, + requested.head().controller().unwrap().epoch, + now, + &JournalTransition::Maintenance(MaintenanceEvent::SessionReplaced(new_session)) + ) + .await + .is_err() + ); + let current = fixture.journal.load_snapshot(scope()).await.unwrap(); + let replaced = fixture + .journal + .compare_exchange( + ¤t, + current.head().controller().unwrap().epoch, + now, + &JournalTransition::Maintenance(MaintenanceEvent::SessionReplaced(new_session)), + ) + .await + .unwrap(); + let intents = fixture + .journal + .intents_page(replaced.registry(), None, 128) + .await + .unwrap(); + let intent = &intents.entries()[0]; + let ad = advertisement(0, &reboot, intent).await.unwrap(); + let record = enroll( + &fixture.journal, + &fixture.directory, + &spec(intent).unwrap(), + ad, + now, + ) + .await + .unwrap(); + reboot + .install_fleet_actions( + scope(), + node_id(0), + fixture.journal.clone(), + Arc::new(super::super::adapters::Cells { + records: Arc::new(HashMap::new()), + local: 0, + root: fixture.root.path().into(), + }), + ) + .unwrap(); + reboot + .confirm_fleet_startup(fixture.journal.as_ref(), record.spec().key().unwrap()) + .await + .unwrap(); + reboot.start().unwrap(); + assert_eq!(reboot.state(), NodeState::Maintenance); + assert!(!reboot.is_ready() && reboot.is_management_ready()); + assert!(matches!( + reboot.runtime().node_admission().check_new_role(), + Err(Error::CellDraining) + )); + let current = fixture.journal.load_snapshot(scope()).await.unwrap(); + let action = current + .head() + .maintenance_action(MaintenanceAction::Cordon, clock().unwrap()) + .unwrap(); + let result = reboot + .apply_fleet_action(action, clock().unwrap()) + .await + .unwrap(); + assert!(result.committed && result.execution_error.is_none()); + assert_eq!(result.outcome.outcome, FleetOutcome::Cordoned); + reboot.shutdown().await.unwrap(); + assert_eq!(reboot.stats().retained_bytes(), 0); + close(fixture).await; +} + +#[tokio::test] +async fn boot_cleanup_adopts_canonical_withdrawal_and_lost_retirement_reply() { + for lost_retirement in [false, true] { + let fixture = fixture().await; + let boot_spec = spec(&fixture.intent).unwrap(); + let record = enroll( + &fixture.journal, + &fixture.directory, + &boot_spec, + fixture.ad.clone(), + clock().unwrap(), + ) + .await + .unwrap(); + let boot = BootOwner { + node: fixture.node.clone(), + directory: fixture.directory.clone(), + spec: boot_spec.clone(), + advertisement: fixture.ad.clone(), + guard: None, + }; + assert!(boot.withdraw(&fixture.journal).await.is_err()); + assert_eq!( + fixture + .journal + .load_enrollment(scope(), boot_spec.key().unwrap()) + .await + .unwrap(), + Some(record.clone()) + ); + fixture.node.shutdown().await.unwrap(); + let observed = fixture + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .unwrap(); + fixture + .directory + .withdraw_after_drain(&observed, clock().unwrap()) + .await + .unwrap(); + assert!(fixture.directory.is_retired(session(0)).await.unwrap()); + // Simulate the caller losing the canonical withdrawal reply. The boot + // owner must observe the tombstone instead of requiring a live ad again. + if lost_retirement { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.example-fleet-boot-retirement.v1\0"); + hash.update(evidence(&boot_spec, &fixture.ad).unwrap().as_bytes()); + fixture.journal.lose_next_commit_reply(); + assert!( + fixture + .journal + .publish_enrollment_result( + &record, + EnrollmentEvent::Retired(Digest::from_bytes(*hash.finalize().as_bytes())), + clock().unwrap(), + ) + .await + .is_err() + ); + } + boot.withdraw(&fixture.journal).await.unwrap(); + let retired = fixture + .journal + .load_enrollment(scope(), boot_spec.key().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(retired.status(), EnrollmentStatus::Retired); + let different = BootOwner { + spec: EnrollmentSpec { + role: EnrollmentRole::Node { + mode: NodeMode::Draining, + }, + ..boot_spec.clone() + }, + node: fixture.node.clone(), + directory: fixture.directory.clone(), + advertisement: fixture.ad.clone(), + guard: None, + }; + let error = different.withdraw(&fixture.journal).await.unwrap_err(); + assert!(matches!( + error.downcast_ref::(), + Some(OperationError::Conflict) + )); + boot.withdraw(&fixture.journal).await.unwrap(); + assert_eq!( + fixture + .journal + .load_enrollment(scope(), boot_spec.key().unwrap()) + .await + .unwrap(), + Some(retired) + ); + close(fixture).await; + } +} + +#[tokio::test] +async fn delayed_boot_confirmation_cannot_replace_newer_intent_or_reopen_shutdown() { + let fixture = fixture().await; + let boot_spec = spec(&fixture.intent).unwrap(); + let key = boot_spec.key().unwrap(); + enroll( + &fixture.journal, + &fixture.directory, + &boot_spec, + fixture.ad.clone(), + clock().unwrap(), + ) + .await + .unwrap(); + let (captured, resume) = fixture.journal.pause_next_boot_reply(); + let node = fixture.node.clone(); + let journal = fixture.journal.clone(); + let old = tokio::spawn(async move { node.confirm_fleet_startup(journal.as_ref(), key).await }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + maintenance(&fixture).await; + fixture + .node + .confirm_fleet_startup(fixture.journal.as_ref(), key) + .await + .unwrap(); + resume.send(()).unwrap(); + assert!(matches!(old.await.unwrap(), Err(Error::Fenced))); + assert_eq!(fixture.node.state(), NodeState::Starting); + assert_eq!( + fixture.node.runtime().node_admission().mode().unwrap(), + NodeMode::Draining + ); + assert!( + fixture + .node + .runtime() + .node_admission() + .check_new_role() + .is_err() + ); + + let (captured, resume) = fixture.journal.pause_next_boot_reply(); + let node = fixture.node.clone(); + let journal = fixture.journal.clone(); + let delayed = + tokio::spawn(async move { node.confirm_fleet_startup(journal.as_ref(), key).await }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + fixture.node.shutdown().await.unwrap(); + resume.send(()).unwrap(); + assert!(matches!(delayed.await.unwrap(), Err(Error::CellDraining))); + assert_eq!(fixture.node.state(), NodeState::Stopped); + assert!(!fixture.node.is_ready() && !fixture.node.is_management_ready()); + assert!( + fixture + .node + .runtime() + .node_admission() + .check_new_role() + .is_err() + ); + close(fixture).await; +} + +#[tokio::test] +async fn foreign_active_boot_and_unavailable_startup_journal_fail_closed() { + let intent = NodeIntent::initial(scope(), node_id(0), session(1)).unwrap(); + assert!(matches!( + CellNodeBuilder::new(super::super::application::compile().unwrap()) + .with_runtime(SqlWorkerPool::new(1, 10).unwrap(), 16 << 20) + .with_replica_host(Host::default()) + .with_session(session(0)) + .with_fleet_startup_intent(intent) + .build(), + Err(Error::Fenced) + )); + let fixture = fixture().await; + fixture.journal.close().await.unwrap(); + let error = fixture + .node + .confirm_fleet_startup( + fixture.journal.as_ref(), + spec(&fixture.intent).unwrap().key().unwrap(), + ) + .await + .unwrap_err(); + let Error::Facility { name, source } = error else { + panic!("startup backend error lost") + }; + assert_eq!(name, "fleet-enrollment-journal"); + assert!(matches!( + source.downcast_ref::(), + Some(Error::RuntimeClosed) + )); + assert!(fixture.node.start().is_err()); + assert!( + fixture + .node + .runtime() + .node_admission() + .check_new_role() + .is_err() + ); + close(fixture).await; +} + +#[tokio::test] +async fn fleet_reader_manager_requires_its_journal_binding_before_start() { + let fixture = fixture().await; + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + ObjectPath::from("startup-readers"), + [3; 16], + ); + let manager = fixture + .node + .install_read_replicas( + layout, + fixture.directory.clone(), + fixture.root.path().join("readers"), + Limits::default(), + ) + .unwrap(); + let boot_spec = spec(&fixture.intent).unwrap(); + enroll( + &fixture.journal, + &fixture.directory, + &boot_spec, + fixture.ad.clone(), + clock().unwrap(), + ) + .await + .unwrap(); + fixture + .node + .confirm_fleet_startup(fixture.journal.as_ref(), boot_spec.key().unwrap()) + .await + .unwrap(); + assert!(matches!( + fixture.node.start(), + Err(Error::Control("CellNode required component is missing")) + )); + assert_eq!(fixture.node.state(), NodeState::Starting); + assert!(matches!( + fixture.node.runtime().node_admission().check_new_role(), + Err(Error::CellDraining) + )); + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + scope().application, + super::super::application::NAMESPACE, + &[1], + ) + .unwrap(); + assert!(matches!( + manager.activate(target, session(1)).await, + Err(Error::Control("fleet reader enrollment is not installed")) + )); + fixture + .node + .install_fleet_reader_enrollment(scope(), node_id(0), fixture.journal.clone()) + .unwrap(); + fixture.node.start().unwrap(); + assert!(fixture.node.is_ready()); + close(fixture).await; +} + +#[tokio::test] +async fn missing_reader_enrollment_does_not_prevent_joined_startup_shutdown() { + let fixture = fixture().await; + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + ObjectPath::from("startup-readers"), + [3; 16], + ); + fixture + .node + .install_read_replicas( + layout, + fixture.directory.clone(), + fixture.root.path().join("readers"), + Limits::default(), + ) + .unwrap(); + close(fixture).await; +} + +async fn start_enrolled_boot(fixture: &Fixture) -> EnrollmentRecord { + let original = enroll( + &fixture.journal, + &fixture.directory, + &spec(&fixture.intent).unwrap(), + fixture.ad.clone(), + clock().unwrap(), + ) + .await + .unwrap(); + fixture + .node + .confirm_fleet_startup(fixture.journal.as_ref(), original.spec().key().unwrap()) + .await + .unwrap(); + fixture.node.start().unwrap(); + original +} + +#[tokio::test] +async fn live_intent_refresh_closes_new_roles_without_cordon_rpc_or_boot_restamping() { + let fixture = fixture().await; + let original = start_enrolled_boot(&fixture).await; + assert!( + fixture + .node + .runtime() + .node_admission() + .check_new_role() + .is_ok() + ); + let snapshot = maintenance(&fixture).await; + let current = fixture + .node + .refresh_fleet_intent( + fixture.journal.as_ref(), + (Instant::now() + Duration::from_secs(3)).into_std(), + ) + .await + .unwrap(); + assert_eq!(current, snapshot.head().node_intent().unwrap().unwrap()); + assert_eq!(current.mode(), NodeMode::Draining); + assert!( + fixture + .node + .runtime() + .node_admission() + .check_new_role() + .is_err() + ); + // Cordon preserves routing to existing owners and management. Terminal + // shutdown has not begun and no role settlement is claimed by this read. + assert_eq!(fixture.node.state(), NodeState::Ready); + assert!(fixture.node.is_ready() && fixture.node.is_management_ready()); + let observed = fixture + .journal + .load_boot(scope(), node_id(0), original.spec().key().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(observed.enrollment(), &original); + let before = fixture.journal.load_snapshot(scope()).await.unwrap(); + assert_eq!( + fixture + .node + .refresh_fleet_intent( + fixture.journal.as_ref(), + (Instant::now() + Duration::from_secs(3)).into_std(), + ) + .await + .unwrap(), + current + ); + assert_eq!( + fixture.journal.load_snapshot(scope()).await.unwrap(), + before + ); + close(fixture).await; +} + +#[tokio::test] +async fn delayed_live_intent_reply_cannot_regress_a_newer_checked_intent() { + let fixture = fixture().await; + start_enrolled_boot(&fixture).await; + let (captured, resume) = fixture.journal.pause_next_boot_reply(); + let node = fixture.node.clone(); + let journal = fixture.journal.clone(); + let old = tokio::spawn(async move { + node.refresh_fleet_intent( + journal.as_ref(), + (Instant::now() + Duration::from_secs(3)).into_std(), + ) + .await + }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + maintenance(&fixture).await; + let current = fixture + .node + .refresh_fleet_intent( + fixture.journal.as_ref(), + (Instant::now() + Duration::from_secs(3)).into_std(), + ) + .await + .unwrap(); + resume.send(()).unwrap(); + assert!(matches!(old.await.unwrap(), Err(Error::Fenced))); + assert_eq!( + fixture.node.runtime().node_admission().mode().unwrap(), + NodeMode::Draining + ); + assert_eq!( + fixture + .node + .refresh_fleet_intent( + fixture.journal.as_ref(), + (Instant::now() + Duration::from_secs(3)).into_std(), + ) + .await + .unwrap(), + current + ); + close(fixture).await; +} + +#[tokio::test] +async fn delayed_live_intent_reply_cannot_reopen_joined_shutdown() { + let fixture = fixture().await; + start_enrolled_boot(&fixture).await; + let (captured, resume) = fixture.journal.pause_next_boot_reply(); + let node = fixture.node.clone(); + let journal = fixture.journal.clone(); + let old = tokio::spawn(async move { + node.refresh_fleet_intent( + journal.as_ref(), + (Instant::now() + Duration::from_secs(3)).into_std(), + ) + .await + }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + fixture.node.shutdown().await.unwrap(); + resume.send(()).unwrap(); + assert!(matches!(old.await.unwrap(), Err(Error::CellDraining))); + assert_eq!(fixture.node.state(), NodeState::Stopped); + assert!(!fixture.node.is_ready() && !fixture.node.is_management_ready()); + assert!( + fixture + .node + .runtime() + .node_admission() + .check_new_role() + .is_err() + ); + close(fixture).await; +} + +#[tokio::test] +async fn live_intent_deadline_and_journal_error_preserve_original_state_and_source() { + let fixture = fixture().await; + let original = start_enrolled_boot(&fixture).await; + let before = fixture.journal.load_snapshot(scope()).await.unwrap(); + assert!(matches!( + fixture + .node + .refresh_fleet_intent(fixture.journal.as_ref(), Instant::now().into_std(),) + .await, + Err(Error::Deadline) + )); + assert_eq!( + fixture.journal.load_snapshot(scope()).await.unwrap(), + before + ); + let (captured, resume) = fixture.journal.pause_next_boot_reply(); + let node = fixture.node.clone(); + let journal = fixture.journal.clone(); + let expired = tokio::spawn(async move { + node.refresh_fleet_intent( + journal.as_ref(), + (Instant::now() + Duration::from_secs(1)).into_std(), + ) + .await + }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + maintenance(&fixture).await; + assert!(matches!(expired.await.unwrap(), Err(Error::Deadline))); + assert!(resume.send(()).is_err()); + assert_eq!( + fixture.node.runtime().node_admission().mode().unwrap(), + NodeMode::Active + ); + assert_eq!( + fixture + .node + .refresh_fleet_intent( + fixture.journal.as_ref(), + (Instant::now() + Duration::from_secs(3)).into_std(), + ) + .await + .unwrap() + .mode(), + NodeMode::Draining + ); + assert_eq!( + fixture + .journal + .load_enrollment(scope(), original.spec().key().unwrap()) + .await + .unwrap(), + Some(original) + ); + fixture.journal.close().await.unwrap(); + let error = fixture + .node + .refresh_fleet_intent( + fixture.journal.as_ref(), + (Instant::now() + Duration::from_secs(3)).into_std(), + ) + .await + .unwrap_err(); + let Error::Facility { name, source } = error else { + panic!("live intent backend source lost") + }; + assert_eq!(name, "fleet-enrollment-journal"); + assert!(matches!( + source.downcast_ref::(), + Some(Error::RuntimeClosed) + )); + assert_eq!( + fixture.node.runtime().node_admission().mode().unwrap(), + NodeMode::Draining + ); + close(fixture).await; +} + +#[tokio::test] +async fn cancelled_live_intent_read_and_active_reply_cannot_clear_local_cordon() { + let fixture = fixture().await; + start_enrolled_boot(&fixture).await; + let (captured, resume) = fixture.journal.pause_next_boot_reply(); + let node = fixture.node.clone(); + let journal = fixture.journal.clone(); + let cancelled = tokio::spawn(async move { + node.refresh_fleet_intent( + journal.as_ref(), + (Instant::now() + Duration::from_secs(3)).into_std(), + ) + .await + }); + tokio::time::timeout(Duration::from_secs(3), captured) + .await + .unwrap() + .unwrap(); + cancelled.abort(); + assert!(cancelled.await.unwrap_err().is_cancelled()); + assert!(resume.send(()).is_err()); + fixture.node.runtime().node_admission().cordon().unwrap(); + let observed = fixture + .node + .refresh_fleet_intent( + fixture.journal.as_ref(), + (Instant::now() + Duration::from_secs(3)).into_std(), + ) + .await + .unwrap(); + assert_eq!(observed.mode(), NodeMode::Active); + assert_eq!( + fixture.node.runtime().node_admission().mode().unwrap(), + NodeMode::Cordoned + ); + assert!( + fixture + .node + .runtime() + .node_admission() + .check_new_role() + .is_err() + ); + close(fixture).await; +} + +#[tokio::test] +async fn confirmed_boot_cannot_be_replaced_by_another_established_request() { + let fixture = fixture().await; + let original_spec = spec(&fixture.intent).unwrap(); + let original = enroll( + &fixture.journal, + &fixture.directory, + &original_spec, + fixture.ad.clone(), + clock().unwrap(), + ) + .await + .unwrap(); + fixture + .node + .confirm_fleet_startup(fixture.journal.as_ref(), original_spec.key().unwrap()) + .await + .unwrap(); + let other_spec = EnrollmentSpec { + request: Digest::from_bytes([89; 32]), + ..original_spec.clone() + }; + let other = enroll( + &fixture.journal, + &fixture.directory, + &other_spec, + fixture.ad.clone(), + clock().unwrap(), + ) + .await + .unwrap(); + assert_ne!(other, original); + assert!(matches!( + fixture + .node + .confirm_fleet_startup(fixture.journal.as_ref(), other_spec.key().unwrap(),) + .await, + Err(Error::Fenced) + )); + assert!( + fixture + .node + .runtime() + .node_admission() + .check_new_role() + .is_err() + ); + fixture.node.start().unwrap(); + assert_eq!( + fixture + .node + .refresh_fleet_intent( + fixture.journal.as_ref(), + (Instant::now() + Duration::from_secs(3)).into_std(), + ) + .await + .unwrap(), + fixture.intent + ); + assert_eq!( + fixture + .journal + .load_enrollment(scope(), original_spec.key().unwrap()) + .await + .unwrap(), + Some(original) + ); + close(fixture).await; +} + +mod withdrawal; diff --git a/crates/cellule-host/minion/scenario/startup/tests/withdrawal.rs b/crates/cellule-host/minion/scenario/startup/tests/withdrawal.rs new file mode 100644 index 00000000..b89ecf93 --- /dev/null +++ b/crates/cellule-host/minion/scenario/startup/tests/withdrawal.rs @@ -0,0 +1,356 @@ +//! Native closing owns canonical withdrawal and the original durable boot row. +use super::*; + +async fn prepare(fixture: &Fixture) -> EnrollmentRecord { + let original = enroll( + &fixture.journal, + &fixture.directory, + &spec(&fixture.intent).unwrap(), + fixture.ad.clone(), + clock().unwrap(), + ) + .await + .unwrap(); + fixture + .node + .confirm_fleet_startup(fixture.journal.as_ref(), original.spec().key().unwrap()) + .await + .unwrap(); + let observed = fixture + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .unwrap(); + fixture + .node + .install_fleet_boot_withdrawal( + fixture.directory.clone(), + observed, + original.clone(), + fixture.journal.clone(), + ) + .unwrap(); + original +} + +async fn retired(fixture: &Fixture, original: &EnrollmentRecord) -> EnrollmentRecord { + let record = fixture + .journal + .load_enrollment(scope(), original.spec().key().unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(record.spec(), original.spec()); + assert_eq!(record.accepted_at_ms(), original.accepted_at_ms()); + assert_eq!( + record.established_evidence(), + original.established_evidence() + ); + assert_eq!(record.status(), EnrollmentStatus::Retired); + assert!(record.settlement_evidence().is_some()); + assert!(fixture.directory.is_retired(session(0)).await.unwrap()); + assert!(fixture.directory.is_withdrawn(session(0)).await.unwrap()); + record +} + +#[tokio::test] +async fn native_shutdown_withdraws_and_retires_original_boot_before_stopped() { + let fixture = fixture().await; + let original = prepare(&fixture).await; + fixture.node.start().unwrap(); + fixture.node.shutdown().await.unwrap(); + assert_eq!(fixture.node.state(), NodeState::Stopped); + let terminal = retired(&fixture, &original).await; + fixture.node.shutdown().await.unwrap(); + assert_eq!(retired(&fixture, &original).await, terminal); + close(fixture).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn sole_waiter_cancellation_keeps_withdrawal_and_durable_retirement_owned() { + let fixture = fixture().await; + let original = prepare(&fixture).await; + let entered = Arc::new(tokio::sync::Notify::new()); + let gate = Arc::new(tokio::sync::Semaphore::new(0)); + let entered_drain = entered.clone(); + let gate_drain = gate.clone(); + fixture + .node + .install_facility( + cellule_host::CellNodeFacility::new("withdrawal-order", move || { + let entered = entered_drain.clone(); + let gate = gate_drain.clone(); + async move { + entered.notify_one(); + gate.acquire().await.unwrap().forget(); + Ok(()) + } + }) + .unwrap(), + ) + .unwrap(); + fixture.node.start().unwrap(); + let node = fixture.node.clone(); + let caller = tokio::spawn(async move { node.shutdown().await }); + entered.notified().await; + assert_eq!(fixture.node.state(), NodeState::Draining); + assert!(!fixture.directory.is_retired(session(0)).await.unwrap()); + assert_eq!( + fixture + .journal + .load_enrollment(scope(), original.spec().key().unwrap()) + .await + .unwrap() + .unwrap(), + original + ); + caller.abort(); + assert!(caller.await.unwrap_err().is_cancelled()); + gate.add_permits(2); + tokio::time::timeout(Duration::from_secs(5), async { + while fixture.node.state() != NodeState::Stopped { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + retired(&fixture, &original).await; + close(fixture).await; +} + +#[tokio::test] +async fn ambiguous_retirement_reply_keeps_draining_and_replays_original_evidence() { + let fixture = fixture().await; + let original = prepare(&fixture).await; + fixture.node.start().unwrap(); + let (committed, resume) = fixture.journal.pause_next_enrollment_reply(true, true); + let node = fixture.node.clone(); + let caller = tokio::spawn(async move { node.shutdown().await }); + committed.await.unwrap(); + let terminal = retired(&fixture, &original).await; + assert_eq!(fixture.node.state(), NodeState::Draining); + assert_eq!(fixture.node.stats().active_cells(), 0); + resume.send(()).unwrap(); + let error = caller.await.unwrap().unwrap_err(); + assert!(std::error::Error::source(&error).is_some()); + assert_eq!(fixture.node.state(), NodeState::Draining); + fixture.node.shutdown().await.unwrap(); + assert_eq!(fixture.node.state(), NodeState::Stopped); + assert_eq!(retired(&fixture, &original).await, terminal); + close(fixture).await; +} + +#[tokio::test] +async fn retirement_reply_deadline_retains_original_commit_and_resumes_closing() { + let fixture = fixture().await; + let original = prepare(&fixture).await; + fixture.node.start().unwrap(); + let (committed, resume) = fixture.journal.pause_next_enrollment_reply(true, false); + let node = fixture.node.clone(); + let caller = tokio::spawn(async move { + node.drain_until(Some((Instant::now() + Duration::from_secs(1)).into_std())) + .await + }); + committed.await.unwrap(); + let terminal = retired(&fixture, &original).await; + assert_eq!(fixture.node.state(), NodeState::Draining); + let error = caller.await.unwrap().unwrap_err(); + assert!(matches!( + error, + Error::Facility { + name: "fleet-boot-withdrawal-deadline", + .. + } + )); + assert_eq!(fixture.node.state(), NodeState::Draining); + assert_eq!(fixture.node.stats().active_cells(), 0); + assert!( + resume.send(()).is_err(), + "expired publication waiter still runs" + ); + fixture.node.shutdown().await.unwrap(); + assert_eq!(retired(&fixture, &original).await, terminal); + close(fixture).await; +} + +#[tokio::test] +async fn native_withdrawal_reconciles_the_original_token_after_a_late_heartbeat() { + let fixture = fixture().await; + let original = prepare(&fixture).await; + let previous = fixture + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .unwrap(); + // Republishing the original signed lease is an idempotent refresh. Issue + // an actual later heartbeat so the original withdrawal token is stale. + while clock().unwrap() <= previous.advertisement().issued_at_ms() { + tokio::time::sleep(Duration::from_millis(1)).await; + } + let next = advertisement(0, &fixture.node, &fixture.intent) + .await + .unwrap(); + let newer = fixture + .directory + .refresh(&previous, next, clock().unwrap()) + .await + .unwrap(); + assert!(newer.advertisement().generation() > previous.advertisement().generation()); + fixture.node.start().unwrap(); + fixture.node.shutdown().await.unwrap(); + assert_eq!(fixture.node.state(), NodeState::Stopped); + retired(&fixture, &original).await; + close(fixture).await; +} + +#[tokio::test] +async fn missing_canonical_boot_cannot_be_retired_or_report_stopped() { + let fixture = fixture().await; + let original = prepare(&fixture).await; + fixture.node.start().unwrap(); + fixture + .layout + .store() + .delete(&fixture.layout.node_path(session(0).as_bytes())) + .await + .unwrap(); + for _ in 0..2 { + assert!(matches!( + fixture.node.shutdown().await, + Err(Error::Control("fleet boot withdrawal lacks its tombstone")) + )); + assert_eq!(fixture.node.state(), NodeState::Draining); + assert_eq!(fixture.node.stats().active_cells(), 0); + assert_eq!(fixture.node.stats().retained_bytes(), 0); + assert!(!fixture.directory.is_retired(session(0)).await.unwrap()); + assert_eq!( + fixture + .journal + .load_enrollment(scope(), original.spec().key().unwrap()) + .await + .unwrap() + .unwrap(), + original + ); + } + // The failed attempt still joined native resources. The original journal + // obligation stays Established; deleting storage cannot invent retirement. + assert_eq!( + fixture.node.drain_observation().unwrap().unwrap().phase, + cellule_host::NodeDrainPhase::Joined + ); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn failed_facility_keeps_original_boot_live_after_runtime_join() { + let fixture = fixture().await; + let original = prepare(&fixture).await; + fixture + .node + .install_facility( + cellule_host::CellNodeFacility::new("required-role-owner", || async { + Err( + Box::new(std::io::Error::other("original role closure failed")) + as Box, + ) + }) + .unwrap(), + ) + .unwrap(); + fixture.node.start().unwrap(); + for _ in 0..2 { + let error = fixture.node.shutdown().await.unwrap_err(); + assert!(matches!( + error, + Error::Facility { + name: "required-role-owner", + .. + } + )); + assert!(std::error::Error::source(&error).is_some()); + assert_eq!(fixture.node.state(), NodeState::Draining); + assert_eq!(fixture.node.stats().active_cells(), 0); + assert_eq!(fixture.node.stats().retained_bytes(), 0); + assert!(!fixture.directory.is_retired(session(0)).await.unwrap()); + assert!( + fixture + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .is_some() + ); + assert_eq!( + fixture + .journal + .load_enrollment(scope(), original.spec().key().unwrap()) + .await + .unwrap() + .unwrap(), + original + ); + } + assert_eq!( + fixture.node.drain_observation().unwrap().unwrap().phase, + cellule_host::NodeDrainPhase::Joined + ); + fixture.journal.close().await.unwrap(); +} + +#[tokio::test] +async fn withdrawal_binding_is_exact_and_cannot_be_replaced_after_startup() { + let fixture = fixture().await; + let original = prepare(&fixture).await; + let observed = fixture + .directory + .load(session(0), clock().unwrap()) + .await + .unwrap() + .unwrap(); + assert!(matches!( + fixture.node.install_fleet_boot_withdrawal( + fixture.directory.clone(), + observed.clone(), + original.clone(), + fixture.journal.clone() + ), + Err(Error::Control("CellNode boot withdrawal already installed")) + )); + let changed = EnrollmentRecord::pending( + EnrollmentSpec { + request: Digest::from_bytes([77; 32]), + ..original.spec().clone() + }, + None, + &fixture.intent, + clock().unwrap(), + ) + .unwrap() + .establish(original.established_evidence().unwrap(), clock().unwrap()) + .unwrap(); + assert!(matches!( + fixture.node.install_fleet_boot_withdrawal( + fixture.directory.clone(), + observed.clone(), + changed, + fixture.journal.clone() + ), + Err(Error::Fenced) + )); + fixture.node.start().unwrap(); + assert!(matches!( + fixture.node.install_fleet_boot_withdrawal( + fixture.directory.clone(), + observed, + original.clone(), + fixture.journal.clone() + ), + Err(Error::CellDraining) + )); + fixture.node.shutdown().await.unwrap(); + retired(&fixture, &original).await; + close(fixture).await; +} diff --git a/crates/cellule-host/minion/scenario/successor_tests.rs b/crates/cellule-host/minion/scenario/successor_tests.rs new file mode 100644 index 00000000..c5c34fe2 --- /dev/null +++ b/crates/cellule-host/minion/scenario/successor_tests.rs @@ -0,0 +1,214 @@ +use super::*; +use cellule_host::fleet::FleetActionJournal; +use cellule_runtime::fleet::operations::*; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn driver_adopts_ordinary_winner_and_joins_original_receiver_before_retirement() { + let root = tempfile::tempdir().unwrap(); + let profile = FleetProfile::default(); + let journal = Arc::new( + SqliteJournal::open( + root.path().join("successor-journal.sqlite"), + scope(), + profile, + clock().unwrap(), + ) + .await + .unwrap(), + ); + let mut nodes = Vec::new(); + let mut boots = Vec::new(); + let (records, acknowledged) = initialize(&root, &journal, &mut nodes, &mut boots, 60_000) + .await + .unwrap(); + let fleet = Arc::new(adapters::LocalFleet { + nodes: nodes.clone(), + journal: journal.clone(), + boots: boots.clone(), + records: records.clone(), + capture_sequence: std::sync::atomic::AtomicU64::new(0), + lose_release_replies: false, + lost_release_replies: std::sync::atomic::AtomicUsize::new(0), + expired_receiver_cleanups: std::sync::atomic::AtomicUsize::new(0), + }); + let driver = FleetReconciler::new( + scope(), + SessionId::from_bytes([206; 16]), + profile, + journal.clone(), + fleet.clone(), + fleet, + ) + .unwrap(); + let first = driver + .reconcile_once(clock, Instant::now() + Duration::from_secs(5)) + .await + .unwrap(); + assert!(first.snapshot.head().attempts().is_empty()); + let page = nodes[0] + .runtime() + .fleet_cells_page(None, 128) + .await + .unwrap(); + let CellInventoryEntry::Owned(row) = &page.entries()[0] else { + panic!("fixture writer missing") + }; + let spec = MoveAttemptSpec { + id: AttemptId { + operation: OperationId::from_bytes([210; 16]).unwrap(), + sequence: 1, + }, + target: row.target.clone(), + incarnation: row.incarnation, + source_node: node_id(0), + source: session(0), + generation: row.generation, + source_epoch: row.position.as_ref().unwrap().epoch, + destination_node: node_id(1), + destination: session(1), + cost: row.cost.unwrap(), + snapshot_digest: Digest::from_bytes([211; 32]), + deadline_ms: clock().unwrap() + 60_000, + }; + journal + .compare_exchange( + &first.snapshot, + first.snapshot.head().controller().unwrap().epoch, + clock().unwrap(), + &JournalTransition::Allocate(spec.clone()), + ) + .await + .unwrap(); + let version = journal.load_snapshot(scope()).await.unwrap().registry(); + journal.set_scheduling(version, false).await.unwrap(); + drop(page); + let preparing = driver + .reconcile_once(clock, Instant::now() + Duration::from_secs(5)) + .await + .unwrap(); + assert_eq!( + preparing.snapshot.head().attempts()[0].phase(), + AttemptPhase::Reserved + ); + let release = driver + .reconcile_once(clock, Instant::now() + Duration::from_secs(5)) + .await + .unwrap(); + assert_eq!(release.released, 1); + let released = release.snapshot.head().attempts()[0] + .released() + .unwrap() + .clone(); + let record = &records[&spec.target.cell_id()]; + let idle = record + .authority + .load(spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + let successor = nodes[2] + .runtime() + .acquire_idle_restored( + record.catalog.clone(), + record.replica.clone(), + record.authority.clone(), + idle, + root.path().join("ordinary-successor.sqlite"), + owner(2), + ) + .await + .unwrap(); + let charged = nodes[1].stats().local_disk_reserved_bytes(); + assert_eq!(charged, spec.cost.disk_bytes); + let refused = driver + .reconcile_once(clock, Instant::now() + Duration::from_secs(5)) + .await + .unwrap(); + assert_eq!( + refused.snapshot.head().attempts()[0].phase(), + AttemptPhase::Activating + ); + assert!(!refused.snapshot.head().attempts()[0].receiver_resources_settled()); + let adopted = driver + .reconcile_once(clock, Instant::now() + Duration::from_secs(5)) + .await + .unwrap(); + assert_eq!(adopted.activated, 1); + let attempt = &adopted.snapshot.head().attempts()[0]; + let evidence = attempt.activated().unwrap(); + assert_eq!((evidence.node, evidence.session), (node_id(2), session(2))); + assert_eq!(evidence.position.root, released.root); + assert_eq!(evidence.position.epoch, released.epoch + 1); + assert!(!attempt.receiver_resources_settled()); + assert!(adopted.snapshot.head().retirement_page(&[spec.id]).is_err()); + assert_eq!(nodes[1].stats().local_disk_reserved_bytes(), charged); + let cleaned = driver + .reconcile_once(clock, Instant::now() + Duration::from_secs(5)) + .await + .unwrap(); + assert!(cleaned.snapshot.head().attempts()[0].receiver_resources_settled()); + assert_eq!(nodes[1].stats().local_disk_reserved_bytes(), 0); + let retired = driver + .reconcile_once(clock, Instant::now() + Duration::from_secs(5)) + .await + .unwrap(); + assert_eq!(retired.retired, 1); + assert!(retired.snapshot.head().attempts().is_empty()); + assert_eq!(retired.snapshot.head().reserved_restore_bytes(), 0); + let original = &acknowledged[&spec.target.cell_id()]; + assert_eq!( + successor + .resolve(original.identity, original.digest, clock().unwrap(), 64) + .await + .unwrap(), + Resolution::Committed(original.outcome.clone()) + ); + let bytes = successor + .query(64, 64, |connection| { + let value: i64 = + connection.query_row("SELECT value FROM counter", [], |row| row.get(0))?; + Ok(value.to_be_bytes().to_vec()) + }) + .await + .unwrap(); + assert_eq!(bytes, original.value.to_be_bytes()); + assert!( + journal + .load_movement_action( + scope(), + spec.id, + MovementAction::Activate, + node_id(2), + session(2) + ) + .await + .unwrap() + .is_none() + ); + assert_eq!( + record + .authority + .load(spec.target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .epoch, + released.epoch + 1 + ); + for node in &nodes { + node.shutdown().await.unwrap(); + assert_eq!(node.state(), NodeState::Stopped); + let stats = node.stats(); + assert_eq!(stats.active_cells(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.retained_bytes(), 0); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.file_descriptors(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); + } + for boot in &boots { + boot.withdraw(&journal).await.unwrap(); + } + journal.close().await.unwrap(); +} diff --git a/crates/cellule-host/minion/scenario/tests.rs b/crates/cellule-host/minion/scenario/tests.rs new file mode 100644 index 00000000..74865e84 --- /dev/null +++ b/crates/cellule-host/minion/scenario/tests.rs @@ -0,0 +1,30 @@ +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn measured_overload_moves_real_cells_after_durable_controller_reconstruction() { + let summary = super::overload().await.unwrap(); + assert_eq!(summary.released, 2); + assert_eq!(summary.activated, 2); + assert_eq!(summary.retired, 2); + assert_eq!(summary.receipt_checks, 2); + assert_eq!(summary.max_inflight, 2); + assert!(summary.max_restore_bytes > 0 && summary.max_restore_bytes <= 8 << 30); + assert_eq!(summary.joined_nodes, 3); + assert_eq!(summary.boot_retirements, 3); + assert_eq!(summary.receiver_nodes, 2); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn new_controller_adopts_lost_releases_after_real_expiry_and_joins_receiver_credit() { + let summary = super::controller_restart().await.unwrap(); + assert_eq!(summary.lost_release_replies, 2); + assert_eq!(summary.controller_epoch, 2); + assert_eq!(summary.expired_receiver_cleanups, 2); + assert_eq!(summary.released, 2); + assert_eq!(summary.activated, 2); + assert_eq!(summary.retired, 2); + assert_eq!(summary.receipt_checks, 2); + assert_eq!(summary.max_inflight, 2); + assert!(summary.max_restore_bytes > 0 && summary.max_restore_bytes <= 8 << 30); + assert_eq!(summary.joined_nodes, 3); + assert_eq!(summary.boot_retirements, 3); + assert_eq!(summary.receiver_nodes, 2); +} diff --git a/crates/cellule-host/src/builder.rs b/crates/cellule-host/src/builder.rs index 1b0f074d..2e797050 100644 --- a/crates/cellule-host/src/builder.rs +++ b/crates/cellule-host/src/builder.rs @@ -11,6 +11,7 @@ pub struct CellNodeBuilder { pub(super) node_retained_bytes: Option, pub(super) required_components: Vec<&'static str>, pub(super) follower_store: Option<(PathBuf, ReplicaLimits, DiskBudget)>, + pub(super) fleet_startup: Option, } pub(crate) struct CellNodeParts { @@ -21,6 +22,7 @@ pub(crate) struct CellNodeParts { pub(super) node_retained_bytes: usize, pub(super) required_components: Vec<&'static str>, pub(super) follower_store: Option<(PathBuf, ReplicaLimits, DiskBudget)>, + pub(super) fleet_startup: Option, } impl CellNodeBuilder { @@ -35,6 +37,7 @@ impl CellNodeBuilder { node_retained_bytes: None, required_components: Vec::new(), follower_store: None, + fleet_startup: None, } } @@ -60,6 +63,19 @@ impl CellNodeBuilder { self } + /// Holds writer/reader/follower admission until this boot's established + /// enrollment and current intent are read from the shared fleet journal. + /// Supply the retained physical-node row; missing or ambiguous rows must + /// fail before building. A draining predecessor cannot reopen this boot. + #[must_use] + pub fn with_fleet_startup_intent( + mut self, + intent: cellule_runtime::fleet::operations::NodeIntent, + ) -> Self { + self.fleet_startup = Some(intent); + self + } + /// Declares the owned production components required before readiness. pub fn with_required_owned_components( mut self, @@ -91,6 +107,7 @@ impl CellNodeBuilder { node_retained_bytes, required_components, follower_store, + fleet_startup, } = self.required_parts()?; let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( pool, @@ -99,17 +116,22 @@ impl CellNodeBuilder { replica_host, )?; install_application_limits(&runtime, &application)?; - let node = CellNode { + if let Some(intent) = &fleet_startup { + let admission = runtime.node_admission(); + admission.hold_startup()?; + match intent.mode() { + cellule_runtime::node::NodeMode::Active => {} + cellule_runtime::node::NodeMode::Cordoned => admission.cordon()?, + cellule_runtime::node::NodeMode::Draining => admission.begin_drain()?, + } + } + let node = CellNode::from_runtime( application, runtime, session, - state: Arc::new(Mutex::new(NodeState::Starting)), - lease_installed: AtomicBool::new(false), - shutdown_lock: Arc::new(tokio::sync::Mutex::new(())), - facilities: Arc::new(Mutex::new(Vec::new())), - required_components: Arc::new(Mutex::new(required_components)), - task_group: Arc::new(Mutex::new(None)), - }; + required_components, + fleet_startup, + ); node.install_follower_store(follower_store)?; Ok(node) } @@ -119,6 +141,9 @@ impl CellNodeBuilder { /// This path intentionally uses object-only runtime admission: the caller /// must keep the host private and may not expose serving readiness. pub fn build_unleased_for_maintenance(self) -> cellule_runtime::Result { + if self.fleet_startup.is_some() { + return Err(Error::Control("fleet startup requires a leased host")); + } let CellNodeParts { application, pool, @@ -127,21 +152,12 @@ impl CellNodeBuilder { node_retained_bytes, required_components, follower_store, + fleet_startup: _, } = self.required_parts()?; let runtime = CellRuntime::new_with_replica_host(pool, node_retained_bytes, session, replica_host)?; install_application_limits(&runtime, &application)?; - let node = CellNode { - application, - runtime, - session, - state: Arc::new(Mutex::new(NodeState::Starting)), - lease_installed: AtomicBool::new(false), - shutdown_lock: Arc::new(tokio::sync::Mutex::new(())), - facilities: Arc::new(Mutex::new(Vec::new())), - required_components: Arc::new(Mutex::new(required_components)), - task_group: Arc::new(Mutex::new(None)), - }; + let node = CellNode::from_runtime(application, runtime, session, required_components, None); node.install_follower_store(follower_store)?; Ok(node) } @@ -159,6 +175,14 @@ impl CellNodeBuilder { if session.as_bytes().iter().all(|byte| *byte == 0) { return Err(Error::Control("CellNode node session is zero")); } + if let Some(intent) = &self.fleet_startup { + intent.to_bytes().map_err(crate::fleet::operation)?; + if intent.mode() == cellule_runtime::node::NodeMode::Active + && intent.session() != session + { + return Err(Error::Fenced); + } + } let node_retained_bytes = self .node_retained_bytes .filter(|bytes| *bytes != 0) @@ -171,6 +195,7 @@ impl CellNodeBuilder { node_retained_bytes, required_components: self.required_components, follower_store: self.follower_store, + fleet_startup: self.fleet_startup, }) } } diff --git a/crates/cellule-host/src/durability.rs b/crates/cellule-host/src/durability.rs deleted file mode 100644 index 0885687a..00000000 --- a/crates/cellule-host/src/durability.rs +++ /dev/null @@ -1,242 +0,0 @@ -//! Durability internals for the Cell node host. - -use super::*; - -/// Error returned by a provider-owned node facility during drain. -pub type FacilityResult = std::result::Result>; - -/// Node-log rotation events emitted by the host supervisor. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum NodeDurabilityRotation { - /// The supervisor is retiring the current node-log generation. - Started, - /// Shutdown is waiting for pending publications to settle. - Pending, - /// A rotation preflight or step failed. - Failed, - /// The replacement generation is installed and serving. - Completed, -} - -/// Provider-owned enrollment adapter used by the host durability supervisor. -/// -/// The provider is responsible for authority and transport enrollment. The -/// host consumes the resulting provider-neutral configuration and is the only -/// owner that constructs and installs [`cellule_runtime::node::durability::NodeDurability`]. -pub trait NodeDurabilityProvider: Send + Sync + 'static { - /// Recruits one enrollment round for the replacement generation. - /// - /// `Ok(None)` means the provider is not ready and the supervisor should - /// ask again after its recruit interval. - fn recruit( - self: Arc, - limits: ReplicaLimits, - required_follower_bytes: u64, - live_node_limit: usize, - ) -> Pin>> + Send>>; - - /// Reports whether provider-owned enrollment state requires epoch rotation. - /// - /// An error defers the decision to a later supervisor tick; it does not - /// stop the current object-durable publication path. - fn rotation_required( - self: Arc, - live_node_limit: usize, - ) -> Pin> + Send>>; - - /// Reports one rotation event to the provider; the default ignores it. - fn rotation_event(&self, _event: NodeDurabilityRotation) {} -} - -/// Fixed host-owned bounds and identity for the node-log supervisor. -#[derive(Clone, Copy, Debug)] -pub struct NodeDurabilitySupervisorConfig { - pub(super) application: ApplicationId, - pub(super) limits: ReplicaLimits, - pub(super) required_follower_bytes: u64, - pub(super) live_node_limit: usize, - pub(super) recruit_interval: std::time::Duration, - pub(super) rotation_interval: std::time::Duration, - pub(super) max_issued_frames: u64, -} - -impl NodeDurabilitySupervisorConfig { - /// Creates a bounded supervisor configuration. - pub fn new( - application: ApplicationId, - limits: ReplicaLimits, - required_follower_bytes: u64, - live_node_limit: usize, - recruit_interval: std::time::Duration, - rotation_interval: std::time::Duration, - max_issued_frames: u64, - ) -> cellule_runtime::Result { - if application.as_bytes().iter().all(|byte| *byte == 0) - || required_follower_bytes == 0 - || live_node_limit == 0 - || recruit_interval.is_zero() - || rotation_interval.is_zero() - || max_issued_frames == 0 - { - return Err(Error::Control( - "invalid CellNode durability supervisor configuration", - )); - } - Ok(Self { - application, - limits, - required_follower_bytes, - live_node_limit, - recruit_interval, - rotation_interval, - max_issued_frames, - }) - } -} - -pub(crate) async fn run_node_durability_supervisor

( - provider: Arc

, - runtime: CellRuntime, - configuration: NodeDurabilitySupervisorConfig, - cancellation: CancellationToken, -) -> FacilityResult -where - P: NodeDurabilityProvider, -{ - let mut recruit = tokio::time::interval(configuration.recruit_interval); - let mut rotation = tokio::time::interval(configuration.rotation_interval); - loop { - tokio::select! { - () = cancellation.cancelled() => return Ok(()), - _ = recruit.tick(), if runtime.node_durability().is_none() => { - match provider.clone().recruit( - configuration.limits, - configuration.required_follower_bytes, - configuration.live_node_limit, - ).await { - Ok(Some(config)) => { - match config.build() { - Ok(durability) => { - if let Err(error) = runtime.install_node_durability( - configuration.application, - durability, - ) { - provider.rotation_event(NodeDurabilityRotation::Failed); - return Err(Box::new(error)); - } - } - Err(_error) => { - provider.rotation_event(NodeDurabilityRotation::Failed); - } - } - } - Ok(None) => {} - Err(_) => provider.rotation_event(NodeDurabilityRotation::Failed), - } - } - _ = rotation.tick(), if runtime.node_durability().is_some() => { - rotate_node_durability( - Arc::clone(&provider), - runtime.clone(), - configuration, - cancellation.clone(), - ).await?; - } - } - } -} - -pub(crate) async fn rotate_node_durability

( - provider: Arc

, - runtime: CellRuntime, - configuration: NodeDurabilitySupervisorConfig, - cancellation: CancellationToken, -) -> FacilityResult -where - P: NodeDurabilityProvider, -{ - let Some((application, durability)) = runtime.node_durability() else { - return Ok(()); - }; - if application != configuration.application { - return Err(Box::new(Error::Control( - "CellNode node durability application changed during rotation", - ))); - } - if !durability.needs_rotation(configuration.max_issued_frames) { - match provider - .clone() - .rotation_required(configuration.live_node_limit) - .await - { - Ok(true) => {} - Ok(false) => return Ok(()), - Err(_) => { - provider.rotation_event(NodeDurabilityRotation::Failed); - return Ok(()); - } - } - } - provider.rotation_event(NodeDurabilityRotation::Started); - loop { - match durability.shutdown().await { - Ok(()) => break, - Err(Error::PendingPublication) => { - provider.rotation_event(NodeDurabilityRotation::Pending); - tokio::select! { - () = cancellation.cancelled() => return Ok(()), - () = tokio::time::sleep(configuration.recruit_interval) => {} - } - } - Err(error) => { - provider.rotation_event(NodeDurabilityRotation::Failed); - return Err(Box::new(error)); - } - } - } - if cancellation.is_cancelled() { - return Ok(()); - } - let replacement = loop { - match provider - .clone() - .recruit( - configuration.limits, - configuration.required_follower_bytes, - configuration.live_node_limit, - ) - .await - { - Ok(Some(config)) => match config.build() { - Ok(durability) => break durability, - Err(_) => provider.rotation_event(NodeDurabilityRotation::Failed), - }, - Ok(None) => {} - Err(_) => provider.rotation_event(NodeDurabilityRotation::Failed), - } - tokio::select! { - () = cancellation.cancelled() => return Ok(()), - () = tokio::time::sleep(configuration.recruit_interval) => {} - } - }; - if cancellation.is_cancelled() { - replacement - .shutdown() - .await - .map_err(|error| Box::new(error) as Box)?; - return Ok(()); - } - match runtime.replace_node_durability(configuration.application, Arc::clone(&replacement)) { - Ok(_) => { - provider.rotation_event(NodeDurabilityRotation::Completed); - Ok(()) - } - Err(error) => { - provider.rotation_event(NodeDurabilityRotation::Failed); - replacement.shutdown().await.map_err(|shutdown_error| { - Box::new(shutdown_error) as Box - })?; - Err(Box::new(error)) - } - } -} diff --git a/crates/cellule-host/src/durability/enrollment/authority.rs b/crates/cellule-host/src/durability/enrollment/authority.rs new file mode 100644 index 00000000..5d9dd8f2 --- /dev/null +++ b/crates/cellule-host/src/durability/enrollment/authority.rs @@ -0,0 +1,167 @@ +//! Confirmed native closure precedes every member's durable retirement event. +use super::*; +use futures_util::future::BoxFuture; + +pub(super) struct EnrollmentAuthority { + // The producer bank owns live obligations. A weak callback prevents a + // cycle through an undelivered epoch's retained runtime cleanup object. + pub(super) record: std::sync::Weak, + pub(super) journal: Arc, + pub(super) interval: Duration, +} +impl EnrollmentAuthority { + fn record(&self) -> cellule_runtime::Result> { + self.record + .upgrade() + .ok_or(Error::Node("follower enrollment owner missing")) + } +} + +impl NodeLogAuthority for EnrollmentAuthority { + fn requires_confirmed_retirement(&self) -> bool { + true + } + fn observe_shutdown_failure(&self, error: Arc) -> cellule_runtime::Result<()> { + let record = self.record()?; + let mut progress = record.progress()?; + if progress.execution_error.is_none() { + progress.execution_error = Some(error); + } + Ok(()) + } + fn observe_retirement( + &self, + observation: Arc, + ) -> cellule_runtime::Result<()> { + let record = self.record()?; + let prepared = record.inputs.attempt.prepared(); + if observation.barrier().leader_session() != prepared.source().session() + || observation.barrier().log_epoch() != prepared.log().epoch() + || observation.barrier().members() != prepared.log().members() + { + return Err(Error::Fenced); + } + if let Err(error) = observation.confirmed() { + record.remember(error, false)?; + } + let mut progress = record.progress()?; + if !progress.native_closed { + progress.retirement = Some(observation); + } + Ok(()) + } + + fn activate<'a>(&'a self, epoch: u64) -> BoxFuture<'a, cellule_runtime::Result<()>> { + Box::pin(async move { + let record = self.record()?; + if record.inputs.attempt.prepared().log().epoch() != epoch { + return Err(Error::Fenced); + } + record + .inputs + .authority + .activate(epoch) + .await + .map_err(|error| { + record + .remember(error, false) + .map(retained) + .unwrap_or_else(|error| error) + }) + }) + } + fn advance_coverage<'a>( + &'a self, + epoch: u64, + through: u64, + ) -> BoxFuture<'a, cellule_runtime::Result<()>> { + Box::pin(async move { + let record = self.record()?; + if record.inputs.attempt.prepared().log().epoch() != epoch { + return Err(Error::Fenced); + } + record + .inputs + .authority + .advance_coverage(epoch, through) + .await + .map_err(|error| { + record + .remember(error, false) + .map(retained) + .unwrap_or_else(|error| error) + }) + }) + } + fn close<'a>( + &'a self, + retirement: &'a NodeLogRetirementObservation, + ) -> BoxFuture<'a, cellule_runtime::Result<()>> { + Box::pin(async move { + retirement.confirmed()?; + let record = self.record()?; + let _closing = record.closing.lock().await; + let prepared = record.inputs.attempt.prepared(); + let barrier = retirement.barrier(); + if barrier.leader_session() != prepared.source().session() + || barrier.log_epoch() != prepared.log().epoch() + || barrier.members() != prepared.log().members() + { + return Err(Error::Fenced); + } + { + let mut progress = record.progress()?; + if let Some(original) = &progress.retirement { + if original.barrier() != barrier { + return Err(Error::Fenced); + } + } else { + // Keep the original full observation before either CAS or + // journal await. Member RPCs cannot be repeated after close. + progress.retirement = Some(Arc::new(retirement.clone())); + } + } + loop { + if !record.progress()?.native_closed { + match record.inputs.authority.close(retirement).await { + Ok(()) => { + // Record the successful canonical close before fallible + // evidence construction. A later error must not replay CAS. + record.progress()?.native_closed = true; + } + Err(error) => { + record.remember(error, false)?; + tokio::time::sleep(self.interval).await; + continue; + } + } + } + let members = record.progress()?.members.clone(); + for (index, member) in members.iter().enumerate() { + if !matches!(member.event, Some(EnrollmentEvent::Retired(_))) { + let event = EnrollmentEvent::Retired(evidence( + &record, + member, + b"confirmed-closed", + )?); + let mut progress = record.progress()?; + progress.members[index].event = Some(event); + progress.members[index].published = false; + } + } + match protocol::publish(&self.journal, &record).await { + Ok(()) => { + record.finished()?; + return Ok(()); + } + Err(error) => { + record.remember(error, true)?; + // This accepted closure is owned by the retained runtime + // or supervisor join. A deadline cancels only its waiter. + tokio::time::sleep(self.interval).await; + } + } + } + }) + } +} diff --git a/crates/cellule-host/src/durability/enrollment/inventory/mod.rs b/crates/cellule-host/src/durability/enrollment/inventory/mod.rs new file mode 100644 index 00000000..ce2a669a --- /dev/null +++ b/crates/cellule-host/src/durability/enrollment/inventory/mod.rs @@ -0,0 +1,366 @@ +//! Fixed-size producer metadata; opaque signed ensembles are never deep-copied. +use super::*; +use cellule_runtime::node::NodeMode; + +const PAGE_BYTES: usize = 1 << 20; +const MAX_EPOCHS: usize = MAX_FOLLOWER_ENROLLMENT_EPOCHS; +const MAX_MEMBERS: usize = 16; + +#[cfg(test)] +mod tests; + +/// Process-local continuation tied to every retained epoch and producer state. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct FollowerEnrollmentInventoryCursor { + topology: Digest, + after: u64, +} +impl FollowerEnrollmentInventoryCursor { + /// Encodes the fixed-width progress digest and last included epoch. + #[must_use] + pub fn to_bytes(self) -> [u8; 40] { + let mut bytes = [0; 40]; + bytes[..32].copy_from_slice(self.topology.as_bytes()); + bytes[32..].copy_from_slice(&self.after.to_le_bytes()); + bytes + } + /// Decodes a continuation; capture rechecks all original producer progress. + pub fn from_bytes(bytes: &[u8]) -> cellule_runtime::Result { + let bytes: &[u8; 40] = bytes + .try_into() + .map_err(|_| Error::Node("invalid follower enrollment cursor width"))?; + let mut topology = [0; 32]; + topology.copy_from_slice(&bytes[..32]); + let mut epoch = [0; 8]; + epoch.copy_from_slice(&bytes[32..]); + let after = u64::from_le_bytes(epoch); + if topology == [0; 32] || after == 0 { + return Err(Error::Node("zero follower enrollment cursor identity")); + } + Ok(Self { + topology: Digest::from_bytes(topology), + after, + }) + } +} + +/// Original requests and advisory progress for one supervisor-owned epoch. +/// Digests identify retained canonical inputs/proofs; they do not authenticate +/// them or establish current authority, native closure or replacement policy. +pub struct FollowerEnrollmentProgress { + /// Exact original leader epoch, in ascending order within a page. + pub epoch: u64, + /// Digest of the original signed selection and conditional-write token. + pub attempt: Digest, + /// All original selected member requests, including unknown acceptance. + pub members: Vec, + /// Whether the original owner's enrollment CAS started. + pub native_started: bool, + /// Whether the original owner chose the nonexecution/refusal cleanup path. + /// This flag alone supplies no durable exclusion fence. + pub no_effect: bool, + /// Whether configuration was delivered to the one native supervisor. + pub delivered: bool, + /// Digest of the original checked enrollment, when available. + pub enrollment: Option, + /// Digest of the original-token conditional refusal, when available. + pub refusal: Option, + /// Original joined member replies, including failures before canonical close. + pub retirement: Option>, + /// Whether the original canonical authority callback confirmed closure. + pub native_closed: bool, + /// Original native error, independently of journal failures. + pub execution_error: Option>, + /// Original journal error, retained across successful replay. + pub journal_error: Option>, +} + +/// Bounded advisory capture retaining one MiB from the node byte ledger. +/// Separate record locks make this an interval scan, not an atomic node barrier. +/// Zero epochs and an idle protocol do not prove a joined supervisor or no roles. +pub struct FollowerEnrollmentInventoryPage { + scope: FleetScope, + node: NodeId, + session: SessionId, + mode: NodeMode, + observed_at_ms: i64, + topology: Digest, + total_epochs: usize, + protocol_busy: bool, + draining: bool, + pending_epoch: Option, + entries: Vec, + next: Option, + _memory: NodeByteReservation, +} +impl FollowerEnrollmentInventoryPage { + /// Returns the installed journal namespace. + #[must_use] + pub const fn scope(&self) -> FleetScope { + self.scope + } + /// Returns the exact original physical leader. + #[must_use] + pub const fn node(&self) -> NodeId { + self.node + } + /// Returns the installed leader boot. + #[must_use] + pub const fn session(&self) -> SessionId { + self.session + } + /// Returns shared admission mode independently of local role counts. + #[must_use] + pub const fn mode(&self) -> NodeMode { + self.mode + } + /// Returns the caller's original capture time, without refreshing proof age. + #[must_use] + pub const fn observed_at_ms(&self) -> i64 { + self.observed_at_ms + } + /// Returns the fingerprint of this local producer capture. + #[must_use] + pub const fn topology(&self) -> Digest { + self.topology + } + /// Counts all retained original epochs, including those outside this page. + #[must_use] + pub const fn total_epochs(&self) -> usize { + self.total_epochs + } + /// Reports a held protocol lane, including preparation before a row exists. + /// It cannot distinguish preparation, acceptance, native dispatch or drain. + #[must_use] + pub const fn protocol_busy(&self) -> bool { + self.protocol_busy + } + /// Reports permanent closure of this producer's recruitment admission. + #[must_use] + pub const fn draining(&self) -> bool { + self.draining + } + /// Returns the original undelivered attempt, when registered locally. + #[must_use] + pub const fn pending_epoch(&self) -> Option { + self.pending_epoch + } + /// Returns original requests and checked progress in ascending epoch order. + #[must_use] + pub fn entries(&self) -> &[FollowerEnrollmentProgress] { + &self.entries + } + /// Continues only if every captured epoch and producer state still matches. + #[must_use] + pub const fn next(&self) -> Option { + self.next + } +} + +impl FleetFollowerEnrollment { + pub(crate) fn page( + &self, + cursor: Option, + limit: usize, + now_ms: i64, + ) -> cellule_runtime::Result { + if !(1..=MAX_EPOCHS).contains(&limit) || now_ms < 0 { + return Err(Error::Node("invalid follower enrollment inventory bounds")); + } + // Reserve before cloning any retained index/progress. Member roles are + // fixed-size Follower variants; signed inputs/proofs are hashed in place. + let memory = self.runtime.try_reserve_node_bytes(PAGE_BYTES)?; + let (records, draining, pending_epoch, last_epoch) = { + let bank = self + .bank + .lock() + .map_err(|_| Error::Node("follower enrollment bank lock poisoned"))?; + if bank.epochs.len() > MAX_EPOCHS { + return Err(Error::Capacity("follower enrollment inventory bound")); + } + ( + bank.epochs + .iter() + .map(|(epoch, row)| (*epoch, row.clone())) + .collect::>(), + bank.draining, + bank.pending + .as_ref() + .map(|row| row.inputs.attempt.prepared().log().epoch()), + bank.last_epoch, + ) + }; + let protocol_busy = self.protocol.try_lock().is_err(); + let mode = self.runtime.node_admission().mode()?; + let total_epochs = records.len(); + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.follower-enrollment-inventory.v1\0"); + hash.update(self.scope.fleet.as_bytes()); + hash.update(self.scope.application.as_bytes()); + hash.update(self.node.as_bytes()); + hash.update(self.session.as_bytes()); + hash.update(&[mode as u8, u8::from(draining), u8::from(protocol_busy)]); + hash.update(&last_epoch.to_le_bytes()); + hash.update(&pending_epoch.unwrap_or(0).to_le_bytes()); + hash.update(&(total_epochs as u64).to_le_bytes()); + let mut entries = Vec::with_capacity(limit.min(total_epochs)); + let mut after_found = cursor.is_none(); + let mut has_more = false; + for (epoch, record) in records { + if cursor.is_some_and(|cursor| cursor.after == epoch) { + after_found = true; + } + let progress = record.progress()?; + validate_members(&record, &progress, epoch)?; + let attempt = record.inputs.attempt.evidence_digest()?; + let enrollment = progress + .enrollment + .as_ref() + .map(NodeLogEnrollmentProof::evidence_digest) + .transpose()?; + let refusal = progress + .refusal + .as_ref() + .map(NodeLogEnrollmentRefusalProof::evidence_digest) + .transpose()?; + hash_progress(&mut hash, epoch, attempt, enrollment, refusal, &progress)?; + if cursor.is_some_and(|cursor| epoch <= cursor.after) { + continue; + } + if entries.len() == limit { + has_more = true; + continue; + } + entries.push(FollowerEnrollmentProgress { + epoch, + attempt, + members: progress.members.clone(), + native_started: progress.native_started, + no_effect: progress.no_effect, + delivered: progress.delivered, + enrollment, + refusal, + retirement: progress.retirement.clone(), + native_closed: progress.native_closed, + execution_error: progress.execution_error.clone(), + journal_error: progress.journal_error.clone(), + }); + } + let topology = Digest::from_bytes(*hash.finalize().as_bytes()); + if !after_found || cursor.is_some_and(|cursor| cursor.topology != topology) { + return Err(Error::Node( + "follower enrollment inventory changed; restart pagination", + )); + } + let next = if has_more { + entries + .last() + .map(|entry| FollowerEnrollmentInventoryCursor { + topology, + after: entry.epoch, + }) + } else { + None + }; + Ok(FollowerEnrollmentInventoryPage { + scope: self.scope, + node: self.node, + session: self.session, + mode, + observed_at_ms: now_ms, + topology, + total_epochs, + protocol_busy, + draining, + pending_epoch, + entries, + next, + _memory: memory, + }) + } +} + +fn validate_members( + record: &Responsibility, + progress: &Progress, + epoch: u64, +) -> cellule_runtime::Result<()> { + let prepared = record.inputs.attempt.prepared(); + let invalid_member = progress.members.iter().any(|member| { + !matches!(member.spec.role, EnrollmentRole::Follower { log_epoch } if log_epoch == epoch) + || member + .accepted + .as_ref() + .is_some_and(|accepted| accepted.spec() != &member.spec) + }); + if epoch != prepared.log().epoch() + || progress.members.len() != prepared.followers().len() + || progress.members.len() > MAX_MEMBERS + || invalid_member + { + return Err(Error::Fenced); + } + Ok(()) +} + +fn hash_progress( + hash: &mut blake3::Hasher, + epoch: u64, + attempt: Digest, + enrollment: Option, + refusal: Option, + progress: &Progress, +) -> cellule_runtime::Result<()> { + hash.update(&epoch.to_le_bytes()); + hash.update(attempt.as_bytes()); + hash.update(&[ + u8::from(progress.native_started), + u8::from(progress.no_effect), + u8::from(progress.delivered), + u8::from(progress.native_closed), + u8::from(progress.execution_error.is_some()), + u8::from(progress.journal_error.is_some()), + ]); + for digest in [enrollment, refusal] { + hash.update(&[u8::from(digest.is_some())]); + if let Some(digest) = digest { + hash.update(digest.as_bytes()); + } + } + hash.update(&(progress.members.len() as u64).to_le_bytes()); + for member in &progress.members { + hash.update(&member.spec.to_bytes().map_err(operation)?); + hash.update(&[ + u8::from(member.accepted.is_some()), + u8::from(member.published), + ]); + if let Some(accepted) = &member.accepted { + hash.update(&accepted.to_bytes().map_err(operation)?); + } + match member.event { + None => { + hash.update(&[0]); + } + Some(event) => { + let (tag, digest) = match event { + EnrollmentEvent::Established(digest) => (1, digest), + EnrollmentEvent::Refused(digest) => (2, digest), + EnrollmentEvent::Retired(digest) => (3, digest), + }; + hash.update(&[tag]); + hash.update(digest.as_bytes()); + } + } + } + hash.update(&[u8::from(progress.retirement.is_some())]); + if let Some(retirement) = &progress.retirement { + // Each original joined observation is immutable. Bind its process-local + // identity as well as the barrier; retry replacement must invalidate a + // continuation even when only an original member error changes. + hash.update(&(Arc::as_ptr(retirement) as usize).to_le_bytes()); + hash.update(retirement.barrier().leader_session().as_bytes()); + hash.update(&retirement.barrier().log_epoch().to_le_bytes()); + hash.update(&retirement.barrier().covered_through().to_le_bytes()); + } + Ok(()) +} diff --git a/crates/cellule-host/src/durability/enrollment/inventory/tests.rs b/crates/cellule-host/src/durability/enrollment/inventory/tests.rs new file mode 100644 index 00000000..7aeb3320 --- /dev/null +++ b/crates/cellule-host/src/durability/enrollment/inventory/tests.rs @@ -0,0 +1,83 @@ +use super::*; + +#[test] +fn cursor_checks_width_nonzero_and_round_trip() { + let cursor = FollowerEnrollmentInventoryCursor { + topology: Digest::from_bytes([7; 32]), + after: 19, + }; + assert_eq!( + FollowerEnrollmentInventoryCursor::from_bytes(&cursor.to_bytes()).unwrap(), + cursor + ); + for width in [0, 8, 32, 39, 41, 64] { + assert!(FollowerEnrollmentInventoryCursor::from_bytes(&vec![1; width]).is_err()); + } + assert!(FollowerEnrollmentInventoryCursor::from_bytes(&[0; 40]).is_err()); + let mut bytes = cursor.to_bytes(); + bytes[32..].fill(0); + assert!(FollowerEnrollmentInventoryCursor::from_bytes(&bytes).is_err()); + bytes = cursor.to_bytes(); + bytes[..32].fill(0); + assert!(FollowerEnrollmentInventoryCursor::from_bytes(&bytes).is_err()); +} + +#[test] +fn fixed_metadata_leaves_room_for_bounded_signing_scratch() { + // No signed ensemble/proof Vec or provider token is deep-copied into a page. + // Each validated member is the fixed Follower role, including its acceptance. + let rows = MAX_EPOCHS + * (std::mem::size_of::() + + MAX_MEMBERS * std::mem::size_of::()); + assert!(rows + 4 * (MAX_RECORD_BYTES as usize) < PAGE_BYTES); +} + +fn progress() -> Progress { + Progress { + members: Vec::new(), + native_started: false, + no_effect: false, + delivered: false, + enrollment: None, + refusal: None, + retirement: None, + native_closed: false, + execution_error: None, + journal_error: None, + reservation: None, + cleanup: None, + } +} +fn digest(progress: &Progress) -> Digest { + let mut hash = blake3::Hasher::new(); + hash_progress( + &mut hash, + 1, + Digest::from_bytes([2; 32]), + None, + None, + progress, + ) + .unwrap(); + Digest::from_bytes(*hash.finalize().as_bytes()) +} + +#[test] +fn continuation_changes_for_native_delivery_and_separate_original_errors() { + let mut progress = progress(); + let original = digest(&progress); + assert_eq!(original, digest(&progress)); + progress.native_started = true; + let started = digest(&progress); + assert_ne!(original, started); + progress.delivered = true; + let delivered = digest(&progress); + assert_ne!(started, delivered); + progress.execution_error = Some(Arc::new(Error::Node("original native error"))); + let failed = digest(&progress); + assert_ne!(delivered, failed); + progress.journal_error = Some(Arc::new(Error::Node("original journal error"))); + assert_ne!(failed, digest(&progress)); + progress.native_closed = true; + assert_ne!(original, digest(&progress)); +} diff --git a/crates/cellule-host/src/durability/enrollment/maintenance.rs b/crates/cellule-host/src/durability/enrollment/maintenance.rs new file mode 100644 index 00000000..f1a3dfa3 --- /dev/null +++ b/crates/cellule-host/src/durability/enrollment/maintenance.rs @@ -0,0 +1,523 @@ +//! Current managed replacement evidence after the original supervisor rotation. +use super::*; +use crate::fleet::{FleetJournalSnapshot, FleetRoster}; +use cellule_runtime::fleet::operations::{MaintenanceOperation, MaintenancePhase}; +use cellule_runtime::node::log_state::NodeLogPhase; +use cellule_runtime::node::{MAX_NODE_BYTES, NodeAdvertisement, NodeMode}; + +/// Checked evacuation of one live owner's original follower responsibility. +/// +/// This is interval evidence, not permission to stop a physical node. Persist +/// and revalidate this evidence together with complete native/foreign inventory, +/// Pending producers, failed-owner recovery and all other role obligations. +/// The original rotation retains its errors independently of a successful retry. +pub struct FollowerEvacuation { + original: EnrollmentRecord, + retired: EnrollmentRecord, + retired_members: Vec, + snapshot: FleetJournalSnapshot, + rotation: Arc, + replacement: NodeLogEnrollmentProof, + replacements: Vec, + authority: NodeAdvertisement, + minimum_members: usize, + started_at_ms: i64, + finished_at_ms: i64, + _memory: NodeByteReservation, +} + +impl FollowerEvacuation { + /// Complete original Retired ensemble, preserving every member's history. + #[must_use] + pub fn retired_members(&self) -> &[EnrollmentRecord] { + &self.retired_members + } + + /// Builds immutable durable metadata under the current application policy. + /// Copied buffers remain the embedding application's accounting obligation. + pub fn durable_record( + &self, + policy: cellule_runtime::fleet::operations::FollowerReplacementPolicy, + ) -> cellule_runtime::Result { + use cellule_runtime::fleet::operations::{ + FollowerEvacuationRecord, FollowerReplacementWitness, + }; + if usize::from(policy.minimum_members()) != self.minimum_members { + return Err(Error::Fenced); + } + FollowerEvacuationRecord::new( + self.snapshot + .head() + .maintenance() + .ok_or(Error::Fenced)? + .clone(), + ( + Digest::from_bytes( + *blake3::hash(&self.snapshot.head().to_bytes().map_err(super::operation)?) + .as_bytes(), + ), + self.snapshot.registry(), + ), + policy, + ( + self.original.spec().key().map_err(super::operation)?, + Digest::from_bytes( + *blake3::hash(&self.original.to_bytes().map_err(super::operation)?).as_bytes(), + ), + ), + self.retired_members.clone(), + self.rotation.retirement().barrier().covered_through(), + boot_identity(&self.authority)?, + ( + self.rotation.replacement_epoch(), + self.replacement.evidence_digest()?, + ), + self.replacements + .iter() + .zip(self.replacement.prepared().followers()) + .map(|(enrollment, boot)| { + Ok(FollowerReplacementWitness { + enrollment: enrollment.clone(), + boot_identity: boot_identity(boot)?, + }) + }) + .collect::>>()?, + self.interval(), + ) + .map_err(super::operation) + } + /// Immutable Established request named by the caller, without restamping. + #[must_use] + pub fn original(&self) -> &EnrollmentRecord { + &self.original + } + /// Committed retirement of that exact request after confirmed native closure. + #[must_use] + pub fn retired(&self) -> &EnrollmentRecord { + &self.retired + } + /// Complete journal head and registry version rechecked after collection. + #[must_use] + pub fn snapshot(&self) -> &FleetJournalSnapshot { + &self.snapshot + } + /// Original confirmed retirement and installed replacement, retaining its Arc. + #[must_use] + pub fn rotation(&self) -> &Arc { + &self.rotation + } + /// Checked canonical replacement with its original signed source/member boots. + /// Retain these identities when persisting and revalidating the evidence. + #[must_use] + pub fn replacement(&self) -> &NodeLogEnrollmentProof { + &self.replacement + } + /// Every Established member of the complete replacement ensemble. + #[must_use] + pub fn replacements(&self) -> &[EnrollmentRecord] { + &self.replacements + } + /// Current signed leader authority at the end of the checked interval. + #[must_use] + pub fn authority(&self) -> &NodeAdvertisement { + &self.authority + } + /// Explicit application redundancy requirement checked by this capture. + #[must_use] + pub const fn minimum_members(&self) -> usize { + self.minimum_members + } + /// Original capture interval; durable records keep their original timestamps. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + (self.started_at_ms, self.finished_at_ms) + } +} + +impl CellNode { + /// Checks one foreign follower evacuation after an accepted epoch rotation. + /// + /// Authenticate the caller and request rotation through + /// [`Self::request_node_log_rotation`] on the live owner. This method starts + /// no effects and does not wait for recruitment: incomplete rotation returns + /// a blocker, leaving the existing supervisor and its original errors owned. + /// Supply the original Established follower row and current Evacuating + /// operation. A replacement must exclude the donor, meet the explicit + /// nonzero member minimum and retain every original managed boot/request. + /// Dead-owner recovery and complete fleet settlement remain separate paths. + pub async fn follower_evacuation( + &self, + original: &EnrollmentRecord, + operation: &MaintenanceOperation, + minimum_members: usize, + deadline: tokio::time::Instant, + ) -> cellule_runtime::Result { + let captured = tokio::time::Instant::now(); + let remaining = operation + .deadline_ms() + .checked_sub(now_ms()?) + .filter(|remaining| *remaining > 0) + .ok_or(Error::Deadline)?; + if captured >= deadline { + return Err(Error::Deadline); + } + let limit = captured + .checked_add(Duration::from_millis(remaining.min(30_000) as u64)) + .ok_or(Error::Deadline)?; + let deadline = deadline.min(limit); + let producer = self + .try_owned_component::(NODE_DURABILITY_PROVIDER_COMPONENT)? + .ok_or(Error::Control( + "follower evacuation requires managed enrollment", + ))?; + tokio::time::timeout_at( + deadline, + producer.evacuated(self, original, operation, minimum_members, deadline), + ) + .await + .map_err(|source| Error::Facility { + name: "follower-evacuation-deadline", + source: Box::new(source), + })? + } +} + +impl FleetFollowerEnrollment { + async fn evacuated( + &self, + node: &CellNode, + original: &EnrollmentRecord, + operation: &MaintenanceOperation, + minimum_members: usize, + deadline: tokio::time::Instant, + ) -> cellule_runtime::Result { + // Reserve before encoding/cloning records or opaque prepared advertisements. + // A canonical epoch has at most two members; copied records and transient + // directory bodies remain bounded even when the full roster is much larger. + let memory = self.runtime.try_reserve_node_metadata_bytes( + 10 * MAX_RECORD_BYTES as usize + 8 * MAX_NODE_BYTES as usize + 8192, + )?; + original.to_bytes().map_err(super::operation)?; + let EnrollmentRole::Follower { log_epoch } = original.spec().role else { + return Err(Error::Fenced); + }; + if minimum_members == 0 + || minimum_members > 2 + || original.status() != EnrollmentStatus::Established + || original.established_evidence().is_none() + || original.spec().scope != self.scope + || original + .spec() + .source + .is_none_or(|source| source.node != self.node || source.session != self.session) + || original.spec().target.node != operation.node() + || original.spec().target.session != operation.session() + || operation.node() == self.node + || !matches!( + operation.phase(), + MaintenancePhase::Evacuating | MaintenancePhase::Closing + ) + { + return Err(Error::Fenced); + } + let request = node + .node_log_rotation_request(log_epoch)? + .ok_or(Error::Control( + "follower evacuation lacks original rotation", + ))?; + let observation = request.observe()?; + let rotation = observation.completion().cloned().ok_or(Error::Control( + "follower evacuation replacement is incomplete", + ))?; + let barrier = rotation.retirement().barrier(); + if barrier.leader_session() != self.session + || barrier.log_epoch() != log_epoch + || !barrier.members().contains(&operation.node()) + || rotation.replacement_epoch() <= log_epoch + { + return Err(Error::Fenced); + } + let replacement = self + .completion(rotation.replacement_epoch())? + .ok_or(Error::Control( + "follower evacuation lacks retained replacement producer", + ))?; + let prepared = replacement.attempt.prepared(); + if !replacement.native_started + || replacement.native_closed + || replacement.refusal.is_some() + || replacement.enrollment.is_none() + || prepared.log().epoch() != rotation.replacement_epoch() + || prepared.log().members().len() < minimum_members + || prepared.log().members().contains(&operation.node()) + || prepared.source().node() != self.node + || prepared.source().session() != self.session + { + return Err(Error::Capacity( + "follower maintenance replacements unavailable", + )); + } + let (directory, lease) = { + let bank = self + .bank + .lock() + .map_err(|_| Error::Node("follower enrollment bank lock poisoned"))?; + let inputs = &bank + .epochs + .get(&rotation.replacement_epoch()) + .ok_or(Error::Fenced)? + .inputs; + (inputs.directory.clone(), inputs.lease.clone()) + }; + lease.check()?; + let started_at_ms = now_ms()?; + let snapshot = self + .journal + .load_snapshot(self.scope) + .await + .map_err(journal_error)?; + if snapshot.head().maintenance() != Some(operation) + || snapshot.registry().bootstrap_revision().is_none() + { + return Err(Error::Fenced); + } + let roster = FleetRoster::collect_admitted( + self.journal.as_ref(), + &snapshot, + deadline, + &self.runtime, + ) + .await?; + let donor = roster.boot(operation.node(), operation.session())?; + if donor.intent().mode() != NodeMode::Draining + || donor.intent().revision() != operation.intent_revision() + { + return Err(Error::Fenced); + } + let retired = roster + .enrollments() + .iter() + .find(|row| row.spec() == original.spec()) + .filter(|row| { + row.status() == EnrollmentStatus::Retired + && row.accepted_at_ms() == original.accepted_at_ms() + && row.established_evidence() == original.established_evidence() + }) + .cloned() + .ok_or(Error::Control( + "follower evacuation retirement is unconfirmed", + ))?; + // Every old member is settled, not only the donor's row. A changed or + // delayed original producer must not be hidden by a newer ensemble. + let mut retired_members = Vec::with_capacity(barrier.members().len()); + for member in barrier.members() { + let mut rows = roster.enrollments().iter().filter(|row| { + row.spec().source.is_some_and(|source| source.node == self.node && source.session == self.session) + && row.spec().target.node == *member + && matches!(row.spec().role, EnrollmentRole::Follower { log_epoch: epoch } if epoch == log_epoch) + }); + let row = rows.next().ok_or(Error::Fenced)?; + if row.status() != EnrollmentStatus::Retired || rows.next().is_some() { + return Err(Error::Fenced); + } + retired_members.push(row.clone()); + } + let mut replacements = Vec::new(); + if roster + .enrollments() + .iter() + .filter(|row| { + row.spec().source.is_some_and(|source| { + source.node == self.node && source.session == self.session + }) && row.spec().role == (EnrollmentRole::Follower { log_epoch }) + }) + .count() + != retired_members.len() + { + return Err(Error::Fenced); + } + if replacement.members.len() != prepared.followers().len() { + return Err(Error::Fenced); + } + for (member, signed) in replacement.members.iter().zip(prepared.followers()) { + let row = roster + .enrollments() + .iter() + .find(|row| row.spec() == &member.spec) + .ok_or(Error::Fenced)?; + if !member.published + || member.accepted.as_ref().is_none_or(|accepted| { + accepted.accepted_at_ms() != row.accepted_at_ms() + || accepted.spec() != row.spec() + }) + || !matches!(member.event, Some(EnrollmentEvent::Established(evidence)) if Some(evidence) == row.established_evidence()) + || row.status() != EnrollmentStatus::Established + || member.spec.target.node != signed.node() + || member.spec.target.session != signed.session() + || !matches!(member.spec.role, EnrollmentRole::Follower { log_epoch: epoch } if epoch == rotation.replacement_epoch()) + { + return Err(Error::Fenced); + } + let boot = roster.boot(signed.node(), signed.session())?; + if boot.intent().mode() != NodeMode::Active + || boot.intent().revision() != member.spec.target.intent_revision + { + return Err(Error::Fenced); + } + replacements.push(row.clone()); + } + let enrollment = directory + .inspect_log_enrollment(&replacement.attempt, now_ms()?) + .await? + .ok_or(Error::Fenced)?; + let authority = enrollment.enrollment().advertisement().clone(); + let source = roster.boot(self.node, self.session)?; + if source.intent().mode() != NodeMode::Active + || replacement.members.iter().any(|member| { + member + .spec + .source + .is_none_or(|endpoint| endpoint.intent_revision != source.intent().revision()) + }) + { + return Err(Error::Fenced); + } + // Probe exact signed boots twice around the full journal confirmation. + // This also checks the donor's actual signed admission gate is closed. + for _ in 0..2 { + for signed in prepared.followers() { + let current = directory + .load_if_live(signed.session(), now_ms()?) + .await? + .ok_or(Error::Fenced)?; + if !same_boot(signed, current.advertisement())? + || !current.advertisement().accepts_new_roles(now_ms()?) + { + return Err(Error::Fenced); + } + } + let donor = directory + .load_if_live(operation.session(), now_ms()?) + .await? + .ok_or(Error::Fenced)?; + if donor.advertisement().node() != operation.node() + || donor + .advertisement() + .operational_sample() + .is_none_or(|sample| sample.mode != NodeMode::Draining) + { + return Err(Error::Fenced); + } + let current = directory + .inspect_log_enrollment(&replacement.attempt, now_ms()?) + .await? + .ok_or(Error::Fenced)?; + if current.enrollment().advertisement() != &authority + || authority + .log() + .is_none_or(|log| log.phase() != NodeLogPhase::Open) + || authority + .operational_sample() + .is_none_or(|sample| sample.mode != NodeMode::Active) + { + return Err(Error::Fenced); + } + let (application, durability) = self.runtime.node_durability().ok_or(Error::Fenced)?; + if application != self.scope.application + || !node.is_ready() + || self.runtime.is_shutting_down() + || self.cancellation.is_cancelled() + || self.runtime.node_admission().mode()? != NodeMode::Active + || durability.identity()? != (self.session, self.node, rotation.replacement_epoch()) + || !request + .observe()? + .completion() + .is_some_and(|current| Arc::ptr_eq(current, &rotation)) + { + return Err(Error::Fenced); + } + lease.check()?; + roster.confirm(self.journal.as_ref(), deadline).await?; + } + let finished_at_ms = now_ms()?; + if finished_at_ms < started_at_ms + || finished_at_ms - started_at_ms > 30_000 + || finished_at_ms >= operation.deadline_ms() + { + return Err(Error::Deadline); + } + Ok(FollowerEvacuation { + original: original.clone(), + retired, + retired_members, + snapshot, + rotation, + replacement: enrollment, + replacements, + authority, + minimum_members, + started_at_ms, + finished_at_ms, + _memory: memory, + }) + } +} + +fn same_boot( + original: &NodeAdvertisement, + current: &NodeAdvertisement, +) -> cellule_runtime::Result { + Ok(original.node() == current.node() + && original.session() == current.session() + && original.fleet() == current.fleet() + && original.endpoint() == current.endpoint() + && original.certificate() == current.certificate() + && original.image() == current.image() + && original.release() == current.release() + && original.verifying_key()? == current.verifying_key()? + && original.module_digests() == current.module_digests() + && original.peer_versions() == current.peer_versions() + && original.failure_domain() == current.failure_domain() + && current.generation() >= original.generation() + && current.issued_at_ms() >= original.issued_at_ms()) +} + +/// Immutable identity shared by native capture and durable revalidation. +pub(crate) fn boot_identity(node: &NodeAdvertisement) -> cellule_runtime::Result { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.follower-maintenance-boot.v1\0"); + for value in [ + node.node().as_bytes().as_slice(), + node.session().as_bytes().as_slice(), + node.fleet().as_bytes().as_slice(), + node.certificate().as_bytes().as_slice(), + node.image().as_bytes().as_slice(), + node.release().as_bytes().as_slice(), + node.endpoint().as_bytes(), + ] { + hash.update(&(value.len() as u64).to_le_bytes()); + hash.update(value); + } + hash.update(&node.verifying_key()?.to_bytes()); + hash.update(&(node.module_digests().len() as u64).to_le_bytes()); + for digest in node.module_digests() { + hash.update(digest.as_bytes()); + } + hash.update(&(node.peer_versions().len() as u64).to_le_bytes()); + for version in node.peer_versions() { + hash.update(&version.to_le_bytes()); + } + for label in [node.failure_domain().zone(), node.failure_domain().host()] { + match label { + Some(label) => { + hash.update(&[1]); + hash.update(&(label.len() as u64).to_le_bytes()); + hash.update(label.as_bytes()); + } + None => { + hash.update(&[0]); + } + } + } + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) +} diff --git a/crates/cellule-host/src/durability/enrollment/mod.rs b/crates/cellule-host/src/durability/enrollment/mod.rs new file mode 100644 index 00000000..1825c0b8 --- /dev/null +++ b/crates/cellule-host/src/durability/enrollment/mod.rs @@ -0,0 +1,382 @@ +//! Pending-before-CAS ownership in the existing durability supervisor and drain lane. +use super::*; +use crate::fleet::{FleetEnrollmentAcceptance, FleetJournal}; +use cellule_runtime::cell::actor::NodeByteReservation; +use cellule_runtime::fleet::operations::{ + EnrollmentEndpoint, EnrollmentEvent, EnrollmentRecord, EnrollmentRole, EnrollmentSpec, + EnrollmentStatus, FleetScope, MAX_RECORD_BYTES, +}; +use cellule_runtime::identity::{Digest, NodeId}; +use cellule_runtime::node::durability::NodeLogAuthority; +use cellule_runtime::node::lease::NodeLeaseGuard; +use cellule_runtime::node::log::NodeLogRetirementObservation; +use cellule_runtime::node::log_transport::NodeLogTransport; +use cellule_runtime::node::{ + NodeDirectory, NodeLogEnrollmentAttempt, NodeLogEnrollmentProof, NodeLogEnrollmentRefusalProof, +}; +use std::collections::BTreeMap; +use std::sync::Mutex as StdMutex; +use tokio::sync::Mutex as AsyncMutex; + +pub(crate) const MAX_FOLLOWER_ENROLLMENT_EPOCHS: usize = 32; + +mod authority; +mod inventory; +pub(crate) mod maintenance; +mod protocol; +pub use inventory::{ + FollowerEnrollmentInventoryCursor, FollowerEnrollmentInventoryPage, FollowerEnrollmentProgress, +}; +pub use maintenance::FollowerEvacuation; + +/// Read-only input provider for journal-bound follower recruitment. +/// Applications own signed directory/transport/authority construction and unique +/// advancing epochs. Preparation must perform no enrollment CAS or frame append. +pub trait FleetNodeDurabilityProvider: Send + Sync + 'static { + /// Selects one exact ensemble and fresh immutable CAS attempt. None leaves + /// the existing object proof path available until the next supervisor tick. + fn prepare( + self: Arc, + limits: ReplicaLimits, + required_follower_bytes: u64, + live_node_limit: usize, + ) -> Pin>> + Send>>; + + /// Reports whether provider-owned enrollment state requires epoch rotation. + /// + /// An error defers the decision to a later supervisor tick; it does not + /// stop the current object-durable publication path. + fn rotation_required( + self: Arc, + live_node_limit: usize, + ) -> Pin> + Send>>; + + /// Observes the existing supervisor's rotation events. + fn rotation_event(&self, _event: NodeDurabilityRotation) {} +} + +/// Provider-owned transport and authority for one opaque prepared enrollment. +/// Construction does not start a shipper or mutate directory authority. +pub struct FleetNodeLogRecruitment { + directory: NodeDirectory, + attempt: NodeLogEnrollmentAttempt, + transport: Arc, + authority: Arc, + lease: NodeLeaseGuard, + telemetry: cellule_runtime::fleet::telemetry::CellTelemetryHandle, +} +impl FleetNodeLogRecruitment { + /// Binds one prepared attempt to application-owned authenticated transports. + /// The transport must address the attempt's exact original follower boots; + /// resolving a replacement physical boot does not authorize a new enrollment. + /// The authority must fresh-load and reconcile this exact source boot/epoch + /// after host enrollment; a pre-enrollment ETag cannot authorize later writes. + /// An ambiguous close requires its original checked close receipt, never an + /// arbitrary absent or newer log record. + pub fn new( + directory: NodeDirectory, + attempt: NodeLogEnrollmentAttempt, + transport: Arc, + authority: Arc, + lease: NodeLeaseGuard, + telemetry: cellule_runtime::fleet::telemetry::CellTelemetryHandle, + ) -> cellule_runtime::Result { + if directory.fleet() != attempt.prepared().source().fleet() { + return Err(Error::Fenced); + } + Ok(Self { + directory, + attempt, + transport, + authority, + lease, + telemetry, + }) + } + fn config( + &self, + limits: ReplicaLimits, + authority: Arc, + ) -> cellule_runtime::Result { + let prepared = self.attempt.prepared(); + NodeDurabilityConfig::new( + prepared.source().session(), + prepared.source().node(), + prepared.log().epoch(), + prepared.log().members().to_vec(), + self.transport.clone(), + authority, + self.lease.clone(), + limits, + self.telemetry.clone(), + ) + } +} + +/// Original durable request and retained result for one selected follower boot. +#[derive(Clone)] +pub struct FollowerEnrollmentMember { + /// Immutable request, including both intent revisions and the original epoch. + pub spec: EnrollmentSpec, + /// Original accepted/tombstoned request, or None while its reply is unknown. + pub accepted: Option, + /// Original checked native or joined-nonexecution event being published. + pub event: Option, + /// Whether the journal confirmed this exact event and immutable request. + pub published: bool, +} +/// Read-only inventory of an owned follower enrollment, including unknown work. +/// Completed retirement removes local inventory; absence proves no fleet fact. +#[derive(Clone)] +pub struct FollowerEnrollmentCompletion { + /// Immutable original source version and complete signed selected ensemble. + pub attempt: NodeLogEnrollmentAttempt, + /// Every original selected member, including missing acceptance replies. + pub members: Vec, + /// Whether this owner's one enrollment CAS started. + pub native_started: bool, + /// Checked canonical enrollment for the original epoch/member set. + pub enrollment: Option, + /// Checked original-token refusal fence, if it won against an ambiguous CAS. + pub refusal: Option, + /// Every original confirmed native retirement response, before authority close. + pub retirement: Option>, + /// Whether the original authority callback confirmed canonical closure. + pub native_closed: bool, + /// Original native failure, independent of publication/acceptance failures. + pub execution_error: Option>, + /// Original journal failure, preserved even after successful replay. + pub journal_error: Option>, +} + +#[derive(Default)] +struct Bank { + draining: bool, + last_epoch: u64, + pending: Option>, + epochs: BTreeMap>, +} +pub(crate) struct FleetFollowerEnrollment { + scope: FleetScope, + node: NodeId, + session: SessionId, + provider: Arc, + journal: Arc, + runtime: CellRuntime, + interval: Duration, + cancellation: CancellationToken, + bank: Arc>, + protocol: AsyncMutex<()>, +} +struct Responsibility { + inputs: FleetNodeLogRecruitment, + limits: ReplicaLimits, + data: StdMutex, + closing: AsyncMutex<()>, + bank: std::sync::Weak>, +} +struct Progress { + members: Vec, + native_started: bool, + no_effect: bool, + delivered: bool, + enrollment: Option, + refusal: Option, + retirement: Option>, + native_closed: bool, + execution_error: Option>, + journal_error: Option>, + reservation: Option, + cleanup: Option>, +} +impl Responsibility { + fn progress(&self) -> cellule_runtime::Result> { + self.data + .lock() + .map_err(|_| Error::Node("follower enrollment progress lock poisoned")) + } + fn remember(&self, error: Error, journal: bool) -> cellule_runtime::Result> { + let error = Arc::new(error); + let mut progress = self.progress()?; + let slot = if journal { + &mut progress.journal_error + } else { + &mut progress.execution_error + }; + if slot.is_none() { + *slot = Some(error.clone()); + } + Ok(error) + } + fn remember_unrecorded(&self, error: Error) -> cellule_runtime::Result> { + let error = Arc::new(error); + let mut progress = self.progress()?; + // Journal and native calls already retain their original source. Only + // record otherwise unreported validation/clock failures here. + if progress.execution_error.is_none() && progress.journal_error.is_none() { + progress.execution_error = Some(error.clone()); + } + Ok(error) + } + fn finished(&self) -> cellule_runtime::Result<()> { + self.progress()?.reservation.take(); + if let Some(bank) = self.bank.upgrade() { + bank.lock() + .map_err(|_| Error::Node("follower enrollment bank lock poisoned"))? + .epochs + .remove(&self.inputs.attempt.prepared().log().epoch()); + } + Ok(()) + } +} + +fn journal_error(source: Box) -> Error { + Error::Facility { + name: "fleet-enrollment-journal", + source, + } +} +fn operation(error: cellule_runtime::fleet::operations::OperationError) -> Error { + Error::FleetOperation(Box::new(error)) +} +#[derive(Debug)] +struct RetainedError(Arc); +impl std::fmt::Display for RetainedError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + self.0.fmt(f) + } +} +impl std::error::Error for RetainedError { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + Some(self.0.as_ref()) + } +} +fn retained(error: Arc) -> Error { + Error::Facility { + name: "fleet-follower-enrollment", + source: Box::new(RetainedError(error)), + } +} +fn now_ms() -> cellule_runtime::Result { + let now = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map_err(|_| Error::Node("follower enrollment clock precedes epoch"))?; + i64::try_from(now.as_millis()).map_err(|_| Error::Node("follower enrollment clock overflow")) +} + +impl FleetFollowerEnrollment { + pub(crate) fn new( + scope: FleetScope, + node: NodeId, + session: SessionId, + provider: Arc, + journal: Arc, + runtime: CellRuntime, + interval: Duration, + cancellation: CancellationToken, + ) -> cellule_runtime::Result { + cellule_runtime::fleet::operations::FleetHead::new(scope, 0).map_err(operation)?; + if node.as_bytes().iter().all(|byte| *byte == 0) || interval.is_zero() { + return Err(Error::Fenced); + } + Ok(Self { + scope, + node, + session, + provider, + journal, + runtime, + interval, + cancellation, + bank: Arc::new(StdMutex::new(Bank::default())), + protocol: AsyncMutex::new(()), + }) + } + pub(crate) fn completion( + &self, + epoch: u64, + ) -> cellule_runtime::Result> { + let record = self + .bank + .lock() + .map_err(|_| Error::Node("follower enrollment bank lock poisoned"))? + .epochs + .get(&epoch) + .cloned(); + let Some(record) = record else { + return Ok(None); + }; + let progress = record.progress()?; + Ok(Some(FollowerEnrollmentCompletion { + attempt: record.inputs.attempt.clone(), + members: progress.members.clone(), + native_started: progress.native_started, + enrollment: progress.enrollment.clone(), + refusal: progress.refusal.clone(), + retirement: progress.retirement.clone(), + native_closed: progress.native_closed, + execution_error: progress.execution_error.clone(), + journal_error: progress.journal_error.clone(), + })) + } +} + +impl NodeDurabilityProvider for FleetFollowerEnrollment { + fn recruit( + self: Arc, + limits: ReplicaLimits, + required_follower_bytes: u64, + live_node_limit: usize, + ) -> Pin>> + Send>> { + Box::pin(async move { + self.recruit_owned(limits, required_follower_bytes, live_node_limit) + .await + .map_err(|error| Box::new(error) as Box) + }) + } + fn drain(self: Arc) -> Pin + Send>> { + Box::pin(async move { + self.drain_owned() + .await + .map_err(|error| Box::new(error) as Box) + }) + } + fn rotation_required( + self: Arc, + live_node_limit: usize, + ) -> Pin> + Send>> { + Arc::clone(&self.provider).rotation_required(live_node_limit) + } + fn rotation_event(&self, event: NodeDurabilityRotation) { + self.provider.rotation_event(event); + } +} + +fn evidence( + record: &Responsibility, + member: &FollowerEnrollmentMember, + tag: &[u8], +) -> cellule_runtime::Result { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet.follower-enrollment.v1\0"); + hash.update(tag); + hash.update(&member.spec.to_bytes().map_err(operation)?); + hash.update(record.inputs.attempt.evidence_digest()?.as_bytes()); + let progress = record.progress()?; + if let Some(proof) = &progress.enrollment { + hash.update(proof.evidence_digest()?.as_bytes()); + } + if let Some(proof) = &progress.refusal { + hash.update(proof.evidence_digest()?.as_bytes()); + } + if let Some(retirement) = &progress.retirement { + hash.update(&retirement.barrier().covered_through().to_le_bytes()); + for member in retirement.members() { + let receipt = member.result().map_err(retained)?; + hash.update(member.member().as_bytes()); + hash.update(&receipt.base_sequence.to_le_bytes()); + hash.update(&receipt.durable_through.to_le_bytes()); + } + } + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) +} diff --git a/crates/cellule-host/src/durability/enrollment/protocol.rs b/crates/cellule-host/src/durability/enrollment/protocol.rs new file mode 100644 index 00000000..14ce5522 --- /dev/null +++ b/crates/cellule-host/src/durability/enrollment/protocol.rs @@ -0,0 +1,461 @@ +//! One retained enrollment attempt, with no native replay after ambiguous dispatch. +use super::*; + +impl FleetFollowerEnrollment { + async fn specs( + &self, + inputs: &FleetNodeLogRecruitment, + ) -> cellule_runtime::Result> { + let prepared = inputs.attempt.prepared(); + if prepared.source().fleet() != self.scope.fleet + || prepared.source().node() != self.node + || prepared.source().session() != self.session + { + return Err(Error::Fenced); + } + let version = self + .journal + .load_snapshot(self.scope) + .await + .map_err(journal_error)? + .registry(); + if version.scope() != self.scope { + return Err(Error::Fenced); + } + let mut endpoints = std::collections::HashMap::new(); + let mut after = None; + let mut count = 0; + loop { + let page = self + .journal + .intents_page(version, after, 128) + .await + .map_err(journal_error)?; + if page.version() != version || page.after() != after { + return Err(Error::Fenced); + } + count += page.entries().len(); + if count > 10_000 { + return Err(Error::Capacity("follower intent inventory bound")); + } + for intent in page.entries() { + if (intent.node() == self.node || prepared.log().members().contains(&intent.node())) + && endpoints + .insert( + intent.node(), + EnrollmentEndpoint { + node: intent.node(), + session: intent.session(), + intent_revision: intent.revision(), + }, + ) + .is_some() + { + return Err(Error::Fenced); + } + } + if endpoints.len() == prepared.followers().len() + 1 || page.next().is_none() { + break; + } + if page.next().map(|node| *node.as_bytes()) <= after.map(|node| *node.as_bytes()) { + return Err(Error::Fenced); + } + after = page.next(); + } + let source = endpoints.get(&self.node).copied().ok_or(Error::Fenced)?; + if source.session != self.session { + return Err(Error::Fenced); + } + prepared + .followers() + .iter() + .map(|member| { + let target = endpoints + .get(&member.node()) + .copied() + .ok_or(Error::Fenced)?; + if target.session != member.session() { + return Err(Error::Fenced); + } + Ok(EnrollmentSpec { + scope: self.scope, + source: Some(source), + target, + role: EnrollmentRole::Follower { + log_epoch: prepared.log().epoch(), + }, + request: Digest::from_bytes( + *blake3::hash(uuid::Uuid::now_v7().as_bytes()).as_bytes(), + ), + }) + }) + .collect() + } + + pub(super) async fn recruit_owned( + &self, + limits: ReplicaLimits, + required_follower_bytes: u64, + live_node_limit: usize, + ) -> cellule_runtime::Result> { + let _protocol = self.protocol.lock().await; + let pending = self + .bank + .lock() + .map_err(|_| Error::Node("follower enrollment bank lock poisoned"))? + .pending + .clone(); + let record = match pending { + Some(record) => record, + None => { + if self.cancellation.is_cancelled() + || self.runtime.node_admission().startup_held()? + { + return Ok(None); + } + let Some(inputs) = self + .provider + .clone() + .prepare(limits, required_follower_bytes, live_node_limit) + .await + .map_err(|source| Error::Facility { + name: "fleet-follower-inputs", + source, + })? + else { + return Ok(None); + }; + inputs + .config(limits, inputs.authority.clone())? + .validate()?; + let specs = self.specs(&inputs).await?; + let reservation = self + .runtime + .try_reserve_node_bytes(16 * MAX_RECORD_BYTES as usize)?; + let epoch = inputs.attempt.prepared().log().epoch(); + let mut bank = self + .bank + .lock() + .map_err(|_| Error::Node("follower enrollment bank lock poisoned"))?; + if bank.draining || self.cancellation.is_cancelled() { + return Ok(None); + } + if epoch <= bank.last_epoch || bank.epochs.len() >= MAX_FOLLOWER_ENROLLMENT_EPOCHS { + return Err(Error::Fenced); + } + let record = Arc::new(Responsibility { + inputs, + limits, + closing: AsyncMutex::new(()), + bank: Arc::downgrade(&self.bank), + data: StdMutex::new(Progress { + members: specs + .into_iter() + .map(|spec| FollowerEnrollmentMember { + spec, + accepted: None, + event: None, + published: false, + }) + .collect(), + native_started: false, + no_effect: false, + delivered: false, + enrollment: None, + refusal: None, + retirement: None, + native_closed: false, + execution_error: None, + journal_error: None, + reservation: Some(reservation), + cleanup: None, + }), + }); + bank.last_epoch = epoch; + bank.epochs.insert(epoch, record.clone()); + bank.pending = Some(record.clone()); + record + } + }; + if let Err(error) = self.step(&record).await { + return Err(retained(record.remember_unrecorded(error)?)); + } + if record.progress()?.no_effect { + self.clear_pending(&record)?; + return Ok(None); + } + let config = self.configuration(&record, limits)?; + if self.cancellation.is_cancelled() { + drop(config); + self.close_unused(&record).await?; + self.clear_pending(&record)?; + return Ok(None); + } + record.progress()?.delivered = true; + self.clear_pending(&record)?; + Ok(Some(config)) + } + + fn configuration( + &self, + record: &Arc, + limits: ReplicaLimits, + ) -> cellule_runtime::Result { + record.inputs.config( + limits, + Arc::new(authority::EnrollmentAuthority { + record: Arc::downgrade(record), + journal: self.journal.clone(), + interval: self.interval, + }), + ) + } + + async fn close_unused(&self, record: &Arc) -> cellule_runtime::Result<()> { + if record.progress()?.cleanup.is_none() { + let cleanup = self.configuration(record, record.limits)?.build()?; + record.progress()?.cleanup = Some(cleanup); + } + let cleanup = record + .progress()? + .cleanup + .clone() + .ok_or(Error::Node("follower cleanup owner missing"))?; + if let Err(error) = cleanup.shutdown_for_maintenance().await { + return Err(retained(record.remember(error, false)?)); + } + record.progress()?.cleanup.take(); + Ok(()) + } + + fn clear_pending(&self, record: &Arc) -> cellule_runtime::Result<()> { + let mut bank = self + .bank + .lock() + .map_err(|_| Error::Node("follower enrollment bank lock poisoned"))?; + if !bank + .pending + .as_ref() + .is_some_and(|pending| Arc::ptr_eq(pending, record)) + { + return Err(Error::Fenced); + } + bank.pending = None; + Ok(()) + } + + async fn step(&self, record: &Arc) -> cellule_runtime::Result<()> { + if !record.progress()?.native_started && self.cancellation.is_cancelled() { + record.progress()?.no_effect = true; + } + if record.progress()?.no_effect { + return self.refuse_unexecuted(record).await; + } + if !record.progress()?.native_started { + let members = record.progress()?.members.clone(); + for (index, member) in members.iter().enumerate() { + let result = self + .journal + .accept_enrollment(&member.spec, now_ms()?) + .await; + // Until every first-acceptance reply validates, this owner has + // no native effect. Keep that fact even on malformed replies. + record.progress()?.no_effect = true; + match result { + Ok(FleetEnrollmentAcceptance::New(original)) => { + original.validate_replay(&member.spec).map_err(operation)?; + original.to_bytes().map_err(operation)?; + if original.status() != EnrollmentStatus::Pending { + return Err(Error::Fenced); + } + let mut progress = record.progress()?; + progress.members[index].accepted = Some(original); + progress.no_effect = false; + } + Ok(FleetEnrollmentAcceptance::Existing(original)) => { + original.validate_replay(&member.spec).map_err(operation)?; + record.progress()?.members[index].accepted = Some(original); + record.progress()?.no_effect = true; + return self.refuse_unexecuted(record).await; + } + Err(source) => { + record.remember(journal_error(source), true)?; + // The finite acceptance call has joined. No native CAS + // began; atomically fence any delayed acceptance itself. + record.progress()?.no_effect = true; + return self.refuse_unexecuted(record).await; + } + } + } + if self.cancellation.is_cancelled() { + record.progress()?.no_effect = true; + return self.refuse_unexecuted(record).await; + } + record.progress()?.native_started = true; + match record + .inputs + .directory + .commit_log_enrollment(&record.inputs.attempt, now_ms()?) + .await + { + Ok(proof) => record.progress()?.enrollment = Some(proof), + Err(error) => { + record.remember(error, false)?; + } + } + } + if record.progress()?.enrollment.is_none() { + match record + .inputs + .directory + .inspect_log_enrollment(&record.inputs.attempt, now_ms()?) + .await + { + Ok(Some(proof)) => record.progress()?.enrollment = Some(proof), + Ok(None) => { + match record + .inputs + .directory + .fence_log_enrollment(&record.inputs.attempt, now_ms()?) + .await + { + Ok(refusal) => { + let mut progress = record.progress()?; + progress.refusal = Some(refusal); + progress.no_effect = true; + } + Err(error) => return Err(retained(record.remember(error, false)?)), + } + return self.refuse_unexecuted(record).await; + } + Err(error) => return Err(retained(record.remember(error, false)?)), + } + } + let members = record.progress()?.members.clone(); + for (index, member) in members.iter().enumerate() { + if member.event.is_none() { + let event = EnrollmentEvent::Established(evidence(record, member, b"enrolled")?); + record.progress()?.members[index].event = Some(event); + } + } + publish(&self.journal, record).await + } + + async fn refuse_unexecuted(&self, record: &Arc) -> cellule_runtime::Result<()> { + let members = record.progress()?.members.clone(); + for (index, member) in members.iter().enumerate() { + if member.published { + continue; + } + let digest = match member.event { + Some(EnrollmentEvent::Refused(digest)) => digest, + None => evidence(record, member, b"joined-unexecuted")?, + _ => return Err(Error::Fenced), + }; + record.progress()?.members[index].event = Some(EnrollmentEvent::Refused(digest)); + let result = match self + .journal + .refuse_unexecuted_enrollment(&member.spec, digest, now_ms()?) + .await + { + Ok(result) => result, + Err(source) => return Err(retained(record.remember(journal_error(source), true)?)), + }; + result.validate_replay(&member.spec).map_err(operation)?; + if result.status() != EnrollmentStatus::Refused + || result.settlement_evidence() != Some(digest) + { + return Err(Error::Fenced); + } + let mut progress = record.progress()?; + progress.members[index].accepted = Some(result); + progress.members[index].published = true; + } + record.finished() + } + + pub(super) async fn drain_owned(&self) -> cellule_runtime::Result<()> { + let _protocol = self.protocol.lock().await; + self.bank + .lock() + .map_err(|_| Error::Node("follower enrollment bank lock poisoned"))? + .draining = true; + loop { + let record = self + .bank + .lock() + .map_err(|_| Error::Node("follower enrollment bank lock poisoned"))? + .pending + .clone(); + let Some(record) = record else { + return Ok(()); + }; + // The supervisor is joined. Only its original undelivered attempt + // can remain; delivered epochs close later through runtime drain. + if !record.progress()?.native_started { + record.progress()?.no_effect = true; + } + let result = async { + self.step(&record).await?; + if !record.progress()?.no_effect { + // No Cell ever used this undelivered configuration. Its + // canonical gate starts at zero; retire every cold member. + self.close_unused(&record).await?; + } + self.clear_pending(&record) + } + .await; + match result { + Ok(()) => return Ok(()), + Err(error) => { + record.remember_unrecorded(error)?; + } + } + tokio::time::sleep(self.interval).await; + } + } +} + +pub(super) async fn publish( + journal: &Arc, + record: &Responsibility, +) -> cellule_runtime::Result<()> { + let members = record.progress()?.members.clone(); + for (index, member) in members.iter().enumerate() { + if member.published { + continue; + } + let original = member + .accepted + .as_ref() + .ok_or(Error::Node("follower acceptance remains unknown"))?; + let event = member + .event + .ok_or(Error::Node("follower event remains unknown"))?; + let result = match journal + .publish_enrollment_result(original, event, now_ms()?) + .await + { + Ok(result) => result, + Err(source) => return Err(retained(record.remember(journal_error(source), true)?)), + }; + result.validate_replay(&member.spec).map_err(operation)?; + result.to_bytes().map_err(operation)?; + let agrees = match event { + EnrollmentEvent::Established(digest) => { + result.status() == EnrollmentStatus::Established + && result.established_evidence() == Some(digest) + } + EnrollmentEvent::Retired(digest) => { + result.status() == EnrollmentStatus::Retired + && result.settlement_evidence() == Some(digest) + } + EnrollmentEvent::Refused(_) => false, + }; + if result.accepted_at_ms() != original.accepted_at_ms() || !agrees { + return Err(Error::Fenced); + } + record.progress()?.members[index].published = true; + } + Ok(()) +} diff --git a/crates/cellule-host/src/durability/mod.rs b/crates/cellule-host/src/durability/mod.rs new file mode 100644 index 00000000..e861c890 --- /dev/null +++ b/crates/cellule-host/src/durability/mod.rs @@ -0,0 +1,125 @@ +//! Durability internals for the Cell node host. + +use super::*; + +/// Error returned by a provider-owned node facility during drain. +pub type FacilityResult = std::result::Result>; + +/// Node-log rotation events emitted by the host supervisor. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum NodeDurabilityRotation { + /// The supervisor is retiring the current node-log generation. + Started, + /// Shutdown is waiting for pending publications to settle. + Pending, + /// A rotation step failed; requested rotations retain failure and may retry. + Failed, + /// The replacement generation is installed and serving. + Completed, +} + +/// Provider-owned enrollment adapter used by the host durability supervisor. +/// +/// The provider is responsible for authority and transport enrollment. The +/// host consumes the resulting provider-neutral configuration and is the only +/// owner that constructs and installs [`cellule_runtime::node::durability::NodeDurability`]. +pub trait NodeDurabilityProvider: Send + Sync + 'static { + /// Joins and settles producer-owned enrollment that was not delivered to the + /// runtime. Called after the retained supervisor joins, before runtime drain. + /// Ordinary providers without producer-owned work retain a no-op drain. + fn drain(self: Arc) -> Pin + Send>> { + Box::pin(async { Ok(()) }) + } + /// Recruits one enrollment round for the replacement generation. + /// + /// `Ok(None)` means the provider is not ready and the supervisor should + /// ask again after its recruit interval. + /// Return the same physical leader and boot with a strictly newer epoch on + /// replacement. The provider owns Pending-before-CAS enrollment, original + /// reply reconciliation, maintenance-node exclusion and redundancy policy. + fn recruit( + self: Arc, + limits: ReplicaLimits, + required_follower_bytes: u64, + live_node_limit: usize, + ) -> Pin>> + Send>>; + + /// Reports whether provider-owned enrollment state requires epoch rotation. + /// + /// An error defers the decision to a later supervisor tick; it does not + /// stop the current object-durable publication path. + fn rotation_required( + self: Arc, + live_node_limit: usize, + ) -> Pin> + Send>>; + + /// Reports one rotation event to the provider; the default ignores it. + fn rotation_event(&self, _event: NodeDurabilityRotation) {} +} + +/// Fixed host-owned bounds and identity for the node-log supervisor. +#[derive(Clone, Copy, Debug)] +pub struct NodeDurabilitySupervisorConfig { + pub(super) application: ApplicationId, + pub(super) limits: ReplicaLimits, + pub(super) required_follower_bytes: u64, + pub(super) live_node_limit: usize, + pub(super) recruit_interval: std::time::Duration, + pub(super) rotation_interval: std::time::Duration, + pub(super) max_issued_frames: u64, +} + +impl NodeDurabilitySupervisorConfig { + /// Creates a bounded supervisor configuration. + pub fn new( + application: ApplicationId, + limits: ReplicaLimits, + required_follower_bytes: u64, + live_node_limit: usize, + recruit_interval: std::time::Duration, + rotation_interval: std::time::Duration, + max_issued_frames: u64, + ) -> cellule_runtime::Result { + if application.as_bytes().iter().all(|byte| *byte == 0) + || required_follower_bytes == 0 + || live_node_limit == 0 + || recruit_interval.is_zero() + || rotation_interval.is_zero() + || max_issued_frames == 0 + { + return Err(Error::Control( + "invalid CellNode durability supervisor configuration", + )); + } + Ok(Self { + application, + limits, + required_follower_bytes, + live_node_limit, + recruit_interval, + rotation_interval, + max_issued_frames, + }) + } +} + +pub(crate) mod enrollment; +mod observation; +mod owner; +pub use enrollment::{ + FleetNodeDurabilityProvider, FleetNodeLogRecruitment, FollowerEnrollmentCompletion, + FollowerEnrollmentInventoryCursor, FollowerEnrollmentInventoryPage, FollowerEnrollmentMember, + FollowerEnrollmentProgress, FollowerEvacuation, +}; +pub use observation::{ + NodeDurabilitySupervisorObservation, NodeDurabilitySupervisorState, NodeLogRotationEntry, + NodeLogRotationInventory, +}; +mod requests; +mod supervisor; +pub(crate) use owner::DurabilitySupervisor; +pub use requests::{ + NodeLogRotationCompletion, NodeLogRotationObservation, NodeLogRotationPhase, + NodeLogRotationRequest, +}; +use supervisor::run_node_durability_supervisor; diff --git a/crates/cellule-host/src/durability/observation/mod.rs b/crates/cellule-host/src/durability/observation/mod.rs new file mode 100644 index 00000000..bd1967a9 --- /dev/null +++ b/crates/cellule-host/src/durability/observation/mod.rs @@ -0,0 +1,184 @@ +//! Fixed-size observation of the original supervisor and bounded rotation bank. +use super::*; +use tokio::task::AbortHandle; + +pub(super) type SharedFailure = Arc; + +/// Lifecycle of the one retained durability supervisor. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum NodeDurabilitySupervisorState { + /// The retained future has not been started or joined. + NotStarted, + /// The original task is running, including accepted provider/native work. + Running, + /// The task finished without a returned result; its original join is required. + FinishedUnobserved, + /// The supervisor future returned; its task and request stop are not yet joined. + Returned, + /// The task was joined but stopping its request bank failed. + JoinedUnsettled, + /// The original task was joined and request-stop bookkeeping succeeded. + Joined, +} + +/// Original epoch and its local progress, without a new rotation request. +#[derive(Clone, Debug)] +pub struct NodeLogRotationEntry { + /// Exact epoch named at original acceptance. + pub epoch: u64, + /// Original errors, retirement and replacement references at capture. + pub progress: NodeLogRotationObservation, +} + +/// The supervisor's existing bounded bank, including automatic rotation. +#[derive(Clone, Debug)] +pub struct NodeLogRotationInventory { + /// New requests are permanently closed after supervisor join. + pub stopped: bool, + /// Claimed epoch, including automatic work without a requested receipt. + pub running_epoch: Option, + /// At most one original pending request; Interrupted does not prove absence. + pub pending: Option, + /// At most one retained local completion; eviction does not prove settlement. + pub completed: Option, +} + +/// Local interval observation; it cannot certify role absence or fleet completion. +/// +/// Capture only clones fixed metadata and original Arcs. It never awaits the +/// supervisor's join lane, provider I/O or native retirement. The installed +/// owner retains four KiB of metadata admission until that owner is released. +/// Returned is distinct from Joined, and a joined error remains visible. +#[derive(Clone, Debug)] +pub struct NodeDurabilitySupervisorObservation { + /// Compiled application bound at installation. + pub application: ApplicationId, + /// Exact original host boot session. + pub session: SessionId, + /// Supplied local observation time, without restamping remote evidence. + pub observed_at_ms: i64, + /// Work cancellation was requested; accepted calls may still be running. + pub cancellation_requested: bool, + /// Original supervisor lifecycle, independent of producer row counts. + pub state: NodeDurabilitySupervisorState, + /// Original returned or task-join failure, preserved across repeated captures. + pub supervisor_error: Option>, + /// Original request-stop failure, independently of the supervisor's result. + pub requests_error: Option>, + /// Every original bank entry, or an explicit capture error supplying no coverage. + pub rotations: Result>, +} + +struct Status { + state: NodeDurabilitySupervisorState, + task: Option, + supervisor_error: Option, + requests_error: Option>, +} + +pub(super) struct SupervisorProgress { + status: Mutex, + _bytes: cellule_runtime::cell::actor::NodeByteReservation, +} + +impl SupervisorProgress { + pub(super) fn new(runtime: &CellRuntime) -> cellule_runtime::Result { + Ok(Self { + status: Mutex::new(Status { + state: NodeDurabilitySupervisorState::NotStarted, + task: None, + supervisor_error: None, + requests_error: None, + }), + _bytes: runtime.try_reserve_node_metadata_bytes(4 * 1024)?, + }) + } + + fn lock(&self) -> cellule_runtime::Result> { + self.status + .lock() + .map_err(|_| Error::Control("node-log supervisor observation lock poisoned")) + } + + pub(super) fn start( + &self, + future: Pin> + Send>>, + ) -> cellule_runtime::Result>> { + let mut status = self.lock()?; + if status.state != NodeDurabilitySupervisorState::NotStarted { + return Err(Error::Control("node-log supervisor already started")); + } + // Hold only this short metadata lock through spawn. A task that returns + // immediately cannot publish Returned before Running overwrites it. + let task = tokio::spawn(future); + status.task = Some(task.abort_handle()); + status.state = NodeDurabilitySupervisorState::Running; + Ok(task) + } + + pub(super) fn returned( + &self, + result: &Result<(), SharedFailure>, + ) -> cellule_runtime::Result<()> { + let mut status = self.lock()?; + status.state = NodeDurabilitySupervisorState::Returned; + status.supervisor_error = result.as_ref().err().cloned(); + Ok(()) + } + + pub(super) fn joined( + &self, + result: &Result<(), SharedFailure>, + requests_error: Option>, + ) -> cellule_runtime::Result>> { + let mut status = self.lock()?; + status.supervisor_error = result.as_ref().err().cloned(); + let failed = requests_error.is_some(); + if let Some(error) = requests_error { + status.requests_error.get_or_insert(error); + status.state = NodeDurabilitySupervisorState::JoinedUnsettled; + } else { + status.state = NodeDurabilitySupervisorState::Joined; + } + Ok(failed.then(|| status.requests_error.clone()).flatten()) + } + + pub(super) fn capture( + &self, + application: ApplicationId, + session: SessionId, + now_ms: i64, + cancellation_requested: bool, + requests: &requests::RotationRequests, + ) -> cellule_runtime::Result { + let status = self.lock()?; + // Status precedes the bank; join releases the bank before publishing + // Joined. A Joined capture therefore cannot carry pre-stop bank state. + let rotations = requests.inventory().map_err(|error| { + status + .requests_error + .clone() + .unwrap_or_else(|| Arc::new(error)) + }); + let state = if status.state == NodeDurabilitySupervisorState::Running + && status.task.as_ref().is_some_and(AbortHandle::is_finished) + { + NodeDurabilitySupervisorState::FinishedUnobserved + } else { + status.state + }; + Ok(NodeDurabilitySupervisorObservation { + application, + session, + observed_at_ms: now_ms, + cancellation_requested, + state, + supervisor_error: status.supervisor_error.clone(), + requests_error: status.requests_error.clone(), + rotations, + }) + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-host/src/durability/observation/tests.rs b/crates/cellule-host/src/durability/observation/tests.rs new file mode 100644 index 00000000..aa488678 --- /dev/null +++ b/crates/cellule-host/src/durability/observation/tests.rs @@ -0,0 +1,194 @@ +use super::*; + +fn runtime() -> CellRuntime { + CellRuntime::new( + SqlWorkerPool::new(1, 2).unwrap(), + 1 << 20, + SessionId::from_bytes([7; 16]), + ) + .unwrap() +} + +fn capture( + progress: &SupervisorProgress, + requests: &requests::RotationRequests, +) -> NodeDurabilitySupervisorObservation { + progress + .capture( + ApplicationId::from_bytes([3; 16]), + SessionId::from_bytes([7; 16]), + 17, + false, + requests, + ) + .unwrap() +} + +#[tokio::test] +async fn returned_error_is_original_and_requires_join_and_request_stop() { + let runtime = runtime(); + let before = runtime.stats().retained_bytes(); + let progress = Arc::new(SupervisorProgress::new(&runtime).unwrap()); + assert_eq!(runtime.stats().retained_bytes(), before + 4 * 1024); + assert!(std::mem::size_of::() < 4 * 1024); + let requests = requests::RotationRequests::new(ApplicationId::from_bytes([3; 16])); + assert_eq!( + capture(&progress, &requests).state, + NodeDurabilitySupervisorState::NotStarted + ); + let source: SharedFailure = Arc::new(std::io::Error::other("original supervisor failure")); + let returned = progress.clone(); + let original = source.clone(); + let task = progress + .start(Box::pin(async move { + let result = Err(original); + returned.returned(&result).unwrap(); + result + })) + .unwrap(); + let result = task.await.unwrap(); + let observed = capture(&progress, &requests); + assert_eq!(observed.state, NodeDurabilitySupervisorState::Returned); + assert!(!observed.rotations.unwrap().stopped); + assert!(Arc::ptr_eq( + observed.supervisor_error.as_ref().unwrap(), + &source + )); + requests.stop(Some(source.clone())).unwrap(); + assert!(progress.joined(&result, None).unwrap().is_none()); + for _ in 0..2 { + let observed = capture(&progress, &requests); + assert_eq!(observed.state, NodeDurabilitySupervisorState::Joined); + assert!(observed.rotations.unwrap().stopped); + assert!(Arc::ptr_eq( + observed.supervisor_error.as_ref().unwrap(), + &source + )); + } + drop(progress); + assert_eq!(runtime.stats().retained_bytes(), before); + runtime.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn panicked_task_is_unobserved_until_its_original_join() { + let runtime = runtime(); + let progress = SupervisorProgress::new(&runtime).unwrap(); + let requests = requests::RotationRequests::new(ApplicationId::from_bytes([3; 16])); + let task = progress + .start(Box::pin(async { panic!("original supervisor panic") })) + .unwrap(); + tokio::time::timeout(Duration::from_secs(3), async { + while !task.is_finished() { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + let observed = capture(&progress, &requests); + assert_eq!( + observed.state, + NodeDurabilitySupervisorState::FinishedUnobserved + ); + assert!(observed.supervisor_error.is_none()); + let source: SharedFailure = Arc::new(task.await.unwrap_err()); + requests.stop(Some(source.clone())).unwrap(); + progress.joined(&Err(source.clone()), None).unwrap(); + let observed = capture(&progress, &requests); + assert_eq!(observed.state, NodeDurabilitySupervisorState::Joined); + assert!(Arc::ptr_eq( + observed.supervisor_error.as_ref().unwrap(), + &source + )); + assert!( + observed + .supervisor_error + .unwrap() + .downcast_ref::() + .unwrap() + .is_panic() + ); + drop(progress); + runtime.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn bookkeeping_failure_stays_separate_and_preserves_first_error() { + let runtime = runtime(); + let progress = SupervisorProgress::new(&runtime).unwrap(); + let requests = requests::RotationRequests::new(ApplicationId::from_bytes([3; 16])); + let source: SharedFailure = Arc::new(std::io::Error::other("native error")); + let original = Arc::new(Error::Control("original stop error")); + let first = progress + .joined(&Err(source.clone()), Some(original.clone())) + .unwrap() + .unwrap(); + let retry = progress + .joined( + &Err(source.clone()), + Some(Arc::new(Error::Control("retry stop error"))), + ) + .unwrap() + .unwrap(); + assert!(Arc::ptr_eq(&first, &original)); + assert!(Arc::ptr_eq(&retry, &original)); + let observed = capture(&progress, &requests); + assert_eq!( + observed.state, + NodeDurabilitySupervisorState::JoinedUnsettled + ); + assert!(Arc::ptr_eq( + observed.supervisor_error.as_ref().unwrap(), + &source + )); + assert!(Arc::ptr_eq( + observed.requests_error.as_ref().unwrap(), + &original + )); + requests.stop(Some(source.clone())).unwrap(); + progress.joined(&Err(source), None).unwrap(); + let observed = capture(&progress, &requests); + assert_eq!(observed.state, NodeDurabilitySupervisorState::Joined); + assert!(observed.rotations.unwrap().stopped); + assert!(Arc::ptr_eq( + observed.requests_error.as_ref().unwrap(), + &original + )); + drop(progress); + runtime.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn metadata_admission_and_poison_fail_without_fake_absence() { + let runtime = runtime(); + let retained = runtime + .try_reserve_node_bytes(runtime.stats().retained_capacity_bytes()) + .unwrap(); + assert!(matches!( + SupervisorProgress::new(&runtime), + Err(Error::Capacity(_)) + )); + drop(retained); + let progress = SupervisorProgress::new(&runtime).unwrap(); + let requests = requests::RotationRequests::new(ApplicationId::from_bytes([3; 16])); + let poisoned = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + let _guard = progress.status.lock().unwrap(); + panic!("poison supervisor metadata"); + })); + assert!(poisoned.is_err()); + assert!(matches!( + progress.capture( + ApplicationId::from_bytes([3; 16]), + SessionId::from_bytes([7; 16]), + 17, + false, + &requests + ), + Err(Error::Control( + "node-log supervisor observation lock poisoned" + )) + )); + drop(progress); + assert_eq!(runtime.stats().retained_bytes(), 0); + runtime.shutdown().await.unwrap(); +} diff --git a/crates/cellule-host/src/durability/owner.rs b/crates/cellule-host/src/durability/owner.rs new file mode 100644 index 00000000..5d76774d --- /dev/null +++ b/crates/cellule-host/src/durability/owner.rs @@ -0,0 +1,140 @@ +//! Retain the one supervisor join across cancelled waiters and host deadlines. +use super::*; +use observation::{SharedFailure, SupervisorProgress}; +use requests::RotationRequests; + +type SupervisorFuture = Pin> + Send>>; +enum SupervisorJoin { + Unstarted(Option), + Running(JoinHandle>), + Finished(std::result::Result<(), SharedFailure>), +} + +pub(crate) struct DurabilitySupervisor { + pub(crate) requests: Arc, + pub(crate) cancellation: CancellationToken, + join: tokio::sync::Mutex, + application: ApplicationId, + session: SessionId, + progress: Arc, +} + +impl DurabilitySupervisor { + pub(crate) fn new( + provider: Arc

, + runtime: CellRuntime, + configuration: NodeDurabilitySupervisorConfig, + session: SessionId, + cancellation: CancellationToken, + ) -> cellule_runtime::Result { + let requests = Arc::new(RotationRequests::new(configuration.application)); + let progress = Arc::new(SupervisorProgress::new(&runtime)?); + let returned = Arc::clone(&progress); + let task_requests = Arc::clone(&requests); + let token = cancellation.clone(); + let future = Box::pin(async move { + let result = run_node_durability_supervisor( + provider, + runtime, + configuration, + session, + token, + task_requests, + ) + .await + .map_err(SharedFailure::from); + let captured = returned.returned(&result); + match (result, captured) { + (Err(source), _) => Err(source), + (Ok(()), Err(error)) => Err(Arc::new(error) as SharedFailure), + (Ok(()), Ok(())) => Ok(()), + } + }); + Ok(Self { + requests, + cancellation, + join: tokio::sync::Mutex::new(SupervisorJoin::Unstarted(Some(future))), + application: configuration.application, + session, + progress, + }) + } + pub(crate) fn observe( + &self, + now_ms: i64, + ) -> cellule_runtime::Result { + self.progress.capture( + self.application, + self.session, + now_ms, + self.cancellation.is_cancelled(), + &self.requests, + ) + } + pub(crate) async fn join(&self) -> FacilityResult { + let mut joining = self.join.lock().await; + if let SupervisorJoin::Unstarted(future) = &mut *joining { + let task = future.take().ok_or_else(|| { + Box::new(Error::Control("node-log supervisor future missing")) + as Box + })?; + *joining = if self.cancellation.is_cancelled() { + drop(task); + SupervisorJoin::Finished(Ok(())) + } else { + SupervisorJoin::Running(self.progress.start(task)?) + }; + } + if let SupervisorJoin::Running(task) = &mut *joining { + let result = match task.await { + Ok(result) => result, + Err(source) => Err(Arc::new(source) as SharedFailure), + }; + // No await after a terminal join: a cancelled join waiter leaves + // the same handle in Running; every later caller joins it in place. + *joining = SupervisorJoin::Finished(result); + } + match &*joining { + SupervisorJoin::Finished(result) => { + // Commit the consumed join before bookkeeping. A stop failure + // cannot leave a Ready JoinHandle to be polled a second time. + let stopped = self.requests.stop(result.as_ref().err().cloned()); + let stop_error = stopped.err().map(Arc::new); + let captured = self.progress.joined(result, stop_error); + if let Some(source) = result.as_ref().err() { + return Err(Box::new(SharedSupervisorError(Arc::clone(source)))); + } + match captured? { + Some(source) => Err(Box::new(SharedSupervisorError(source))), + None => Ok(()), + } + } + _ => Err(Box::new(Error::Control( + "node-log supervisor join incomplete", + ))), + } + } + pub(crate) async fn drain(&self) -> FacilityResult { + self.cancellation.cancel(); + self.join().await + } +} +impl Drop for DurabilitySupervisor { + fn drop(&mut self) { + if let SupervisorJoin::Running(task) = self.join.get_mut() { + task.abort(); + } + } +} +#[derive(Debug)] +struct SharedSupervisorError(SharedFailure); +impl std::fmt::Display for SharedSupervisorError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + self.0.fmt(f) + } +} +impl std::error::Error for SharedSupervisorError { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + Some(self.0.as_ref()) + } +} diff --git a/crates/cellule-host/src/durability/requests/mod.rs b/crates/cellule-host/src/durability/requests/mod.rs new file mode 100644 index 00000000..6aeb449d --- /dev/null +++ b/crates/cellule-host/src/durability/requests/mod.rs @@ -0,0 +1,353 @@ +//! Bounded epoch requests, linearized with automatic rotation. +use super::*; +use cellule_runtime::cell::actor::NodeByteReservation; +use cellule_runtime::node::durability::NodeDurability; +use cellule_runtime::node::log::NodeLogRetirementProof; +use std::sync::Weak; +use tokio::sync::Notify; + +const REQUEST_BYTES: usize = 4 * 1024; + +/// Local progress of an accepted request; no phase alone settles a fleet role. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum NodeLogRotationPhase { + /// Accepted before retirement starts. + Queued, + /// Waiting for object coverage or every original member's append fence. + Retiring, + /// The old epoch is closed; replacement recruitment remains outstanding. + Recruiting, + /// Confirmed retirement and a replacement binding were installed. + Completed, + /// Host drain or supervisor failure interrupted the request; inspect authority. + Interrupted, +} + +/// Opaque local completion of confirmed old-epoch retirement and replacement. +/// It does not establish durable journal settlement or replacement policy. +#[derive(Clone, Debug)] +pub struct NodeLogRotationCompletion { + retirement: Arc, + replacement_epoch: u64, +} +impl NodeLogRotationCompletion { + /// Returns confirmation of every original member's exact append fence. + #[must_use] + pub fn retirement(&self) -> &Arc { + &self.retirement + } + /// Returns the newer epoch installed through canonical runtime replacement. + #[must_use] + pub const fn replacement_epoch(&self) -> u64 { + self.replacement_epoch + } +} + +/// One coherent request observation with the original source failure retained. +#[derive(Clone, Debug)] +pub struct NodeLogRotationObservation { + phase: NodeLogRotationPhase, + first_failure: Option>, + latest_failure: Option>, + retirement: Option>, + completion: Option>, +} +impl NodeLogRotationObservation { + /// Returns progress; Interrupted is never completion or proof of absence. + #[must_use] + pub const fn phase(&self) -> NodeLogRotationPhase { + self.phase + } + /// Returns the first original failure, independently of subsequent retries. + #[must_use] + pub fn first_failure(&self) -> Option<&Arc> { + self.first_failure.as_ref() + } + /// Returns the latest original failure; successful retry does not erase it. + #[must_use] + pub fn latest_failure(&self) -> Option<&Arc> { + self.latest_failure.as_ref() + } + /// Returns member confirmation after successful canonical old-epoch closure. + #[must_use] + pub fn retirement(&self) -> Option<&Arc> { + self.retirement.as_ref() + } + /// Returns local completion only after the replacement binding is installed. + #[must_use] + pub fn completion(&self) -> Option<&Arc> { + self.completion.as_ref() + } +} + +/// Weak, epoch-bound inspection handle. Dropping it cannot cancel accepted work. +/// The host retains one pending and one most-recent completed request. +#[derive(Clone, Debug)] +pub struct NodeLogRotationRequest { + epoch: u64, + record: Weak, +} +impl NodeLogRotationRequest { + /// Returns the exact original epoch named at acceptance. + #[must_use] + pub const fn log_epoch(&self) -> u64 { + self.epoch + } + /// Observes retained local progress. Eviction is not proof of completion. + pub fn observe(&self) -> cellule_runtime::Result { + let record = self.record.upgrade().ok_or(Error::Control( + "node-log rotation receipt no longer retained", + ))?; + Ok(record.lock()?.clone()) + } +} + +pub(super) struct RotationRecord { + pub(super) epoch: u64, + pub(super) durability: Arc, + progress: Mutex, + _bytes: NodeByteReservation, +} +impl RotationRecord { + fn lock( + &self, + ) -> cellule_runtime::Result> { + self.progress + .lock() + .map_err(|_| Error::Control("node-log rotation receipt lock poisoned")) + } + fn handle(self: &Arc) -> NodeLogRotationRequest { + NodeLogRotationRequest { + epoch: self.epoch, + record: Arc::downgrade(self), + } + } + pub(super) fn phase(&self, phase: NodeLogRotationPhase) -> cellule_runtime::Result<()> { + self.lock()?.phase = phase; + Ok(()) + } + pub(super) fn failed( + &self, + error: Arc, + ) -> cellule_runtime::Result<()> { + let mut progress = self.lock()?; + progress + .first_failure + .get_or_insert_with(|| Arc::clone(&error)); + progress.latest_failure = Some(error); + Ok(()) + } + pub(super) fn retired( + &self, + proof: Arc, + ) -> cellule_runtime::Result<()> { + let mut progress = self.lock()?; + progress.retirement = Some(proof); + progress.phase = NodeLogRotationPhase::Recruiting; + Ok(()) + } + fn completed(&self, replacement_epoch: u64) -> cellule_runtime::Result<()> { + let mut progress = self.lock()?; + let retirement = progress.retirement.clone().ok_or(Error::Control( + "node-log rotation lacks member confirmation", + ))?; + progress.completion = Some(Arc::new(NodeLogRotationCompletion { + retirement, + replacement_epoch, + })); + progress.phase = NodeLogRotationPhase::Completed; + Ok(()) + } +} + +#[derive(Default)] +struct RequestBank { + pending: Option>, + completed: Option>, + running: Option, + stopped: bool, +} + +pub(crate) struct RotationRequests { + application: ApplicationId, + bank: Mutex, + pub(super) wake: Notify, +} +impl RotationRequests { + pub(super) fn new(application: ApplicationId) -> Self { + Self { + application, + bank: Mutex::new(RequestBank::default()), + wake: Notify::new(), + } + } + fn lock(&self) -> cellule_runtime::Result> { + self.bank + .lock() + .map_err(|_| Error::Control("node-log rotation request lock poisoned")) + } + pub(super) fn inventory(&self) -> cellule_runtime::Result { + let bank = self.lock()?; + let capture = |record: &Arc| -> cellule_runtime::Result<_> { + Ok(NodeLogRotationEntry { + epoch: record.epoch, + progress: record.lock()?.clone(), + }) + }; + Ok(NodeLogRotationInventory { + stopped: bank.stopped, + running_epoch: bank.running, + pending: bank.pending.as_ref().map(capture).transpose()?, + completed: bank.completed.as_ref().map(capture).transpose()?, + }) + } + pub(crate) fn request( + &self, + runtime: &CellRuntime, + epoch: u64, + cancellation: &CancellationToken, + ) -> cellule_runtime::Result { + let mut bank = self.lock()?; + if let Some(record) = bank + .pending + .as_ref() + .filter(|record| record.epoch == epoch) + .or_else(|| { + bank.completed + .as_ref() + .filter(|record| record.epoch == epoch) + }) + { + return Ok(record.handle()); + } + if bank.stopped || cancellation.is_cancelled() || runtime.is_shutting_down() { + return Err(Error::CellDraining); + } + if bank.pending.is_some() { + return Err(Error::Capacity("node-log rotation request already pending")); + } + let (application, durability) = runtime + .node_durability() + .ok_or(Error::Control("node-log durability is not enrolled"))?; + if application != self.application || durability.log_epoch()? != epoch { + return Err(Error::Control("node-log rotation request has stale scope")); + } + // A request cannot retroactively strengthen a best-effort rotation. + // This lock linearizes acceptance with the supervisor's epoch claim. + if bank.running.is_some() { + return Err(Error::Control( + "node-log automatic rotation already running", + )); + } + let record = Arc::new(RotationRecord { + epoch, + durability, + _bytes: runtime.try_reserve_node_bytes(REQUEST_BYTES)?, + progress: Mutex::new(NodeLogRotationObservation { + phase: NodeLogRotationPhase::Queued, + first_failure: None, + latest_failure: None, + retirement: None, + completion: None, + }), + }); + let handle = record.handle(); + bank.pending = Some(record); + self.wake.notify_one(); + Ok(handle) + } + pub(crate) fn lookup( + &self, + epoch: u64, + ) -> cellule_runtime::Result> { + let bank = self.lock()?; + Ok(bank + .pending + .as_ref() + .filter(|record| record.epoch == epoch) + .or_else(|| { + bank.completed + .as_ref() + .filter(|record| record.epoch == epoch) + }) + .map(RotationRecord::handle)) + } + pub(super) fn claim( + &self, + runtime: &CellRuntime, + max_frames: u64, + provider_rotation: bool, + ) -> cellule_runtime::Result> { + let mut bank = self.lock()?; + if bank.stopped || bank.running.is_some() { + return Ok(None); + } + let Some((application, current)) = runtime.node_durability() else { + return Ok(None); + }; + if application != self.application { + return Err(Error::Control( + "CellNode node durability application changed during rotation", + )); + } + let record = bank.pending.clone(); + if let Some(record) = &record { + if !Arc::ptr_eq(¤t, &record.durability) { + return Err(Error::Control( + "requested node-log binding changed outside supervisor", + )); + } + } else if !provider_rotation && !current.needs_rotation(max_frames) { + return Ok(None); + } + let epoch = current.log_epoch()?; + bank.running = Some(epoch); + Ok(Some(RotationWork { + epoch, + durability: current, + record, + })) + } + pub(super) fn complete( + &self, + work: &RotationWork, + replacement_epoch: u64, + ) -> cellule_runtime::Result<()> { + let mut bank = self.lock()?; + if bank.running != Some(work.epoch) { + return Err(Error::Control("node-log rotation claim changed")); + } + if let Some(record) = &work.record { + bank.completed = bank.pending.take(); + // Publish completion after old local history is released. A caller + // observing Completed can rely on the bounded bank having settled. + record.completed(replacement_epoch)?; + } + bank.running = None; + Ok(()) + } + pub(super) fn stop( + &self, + source: Option>, + ) -> cellule_runtime::Result<()> { + let mut bank = self.lock()?; + bank.stopped = true; + if let Some(record) = &bank.pending { + if let Some(source) = source { + record.failed(source)?; + } + record.phase(NodeLogRotationPhase::Interrupted)?; + } + bank.running = None; + Ok(()) + } +} + +pub(super) struct RotationWork { + pub(super) epoch: u64, + pub(super) durability: Arc, + pub(super) record: Option>, +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-host/src/durability/requests/tests.rs b/crates/cellule-host/src/durability/requests/tests.rs new file mode 100644 index 00000000..b23f943b --- /dev/null +++ b/crates/cellule-host/src/durability/requests/tests.rs @@ -0,0 +1,149 @@ +//! Join the real retained task after request-bank bookkeeping fails. +use super::*; +use std::sync::atomic::{AtomicBool, Ordering}; + +struct Provider { + panic: bool, + entered: AtomicBool, +} +impl NodeDurabilityProvider for Provider { + fn rotation_required( + self: Arc, + _live_node_limit: usize, + ) -> Pin> + Send>> { + Box::pin(async { Ok(false) }) + } + + fn recruit( + self: Arc, + _: ReplicaLimits, + _: u64, + _: usize, + ) -> Pin>> + Send>> { + self.entered.store(true, Ordering::Release); + assert!(!self.panic, "original native supervisor panic"); + Box::pin(async { Ok(None) }) + } +} + +async fn joined_poisoned_bank(panic: bool) { + let session = SessionId::from_bytes([7; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 2).unwrap(), 1 << 20, session).unwrap(); + let provider = Arc::new(Provider { + panic, + entered: AtomicBool::new(false), + }); + let config = NodeDurabilitySupervisorConfig::new( + ApplicationId::from_bytes([3; 16]), + ReplicaLimits::default(), + 1, + 2, + Duration::from_millis(10), + Duration::from_secs(3600), + u64::MAX, + ) + .unwrap(); + let owner = Arc::new( + owner::DurabilitySupervisor::new( + provider.clone(), + runtime.clone(), + config, + session, + CancellationToken::new(), + ) + .unwrap(), + ); + let poisoned = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + let _guard = owner.requests.bank.lock().unwrap(); + panic!("poison original request bank"); + })); + assert!(poisoned.is_err()); + let joining = owner.clone(); + let waiter = tokio::spawn(async move { joining.join().await }); + tokio::time::timeout(Duration::from_secs(3), async { + while !provider.entered.load(Ordering::Acquire) { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + owner.cancellation.cancel(); + let first = waiter.await.unwrap().unwrap_err(); + let original = owner.observe(17).unwrap(); + assert_eq!( + original.state, + NodeDurabilitySupervisorState::JoinedUnsettled + ); + let stopped = original.requests_error.unwrap(); + assert!(matches!( + stopped.as_ref(), + Error::Control("node-log rotation request lock poisoned") + )); + assert!(Arc::ptr_eq( + original.rotations.as_ref().unwrap_err(), + &stopped + )); + assert_eq!(original.supervisor_error.is_some(), panic); + let assert_source = |error: &(dyn std::error::Error + 'static)| { + let source = error.source().unwrap(); + // Compare concrete source addresses: trait-object vtables can be + // duplicated across codegen units even for the same original source. + if panic { + let native = original + .supervisor_error + .as_ref() + .unwrap() + .downcast_ref::() + .unwrap(); + assert!(native.is_panic()); + assert!(std::ptr::eq( + source.downcast_ref::().unwrap(), + native + )); + } else { + assert!(std::ptr::eq( + source.downcast_ref::().unwrap(), + stopped.as_ref() + )); + } + }; + assert_source(first.as_ref()); + for _ in 0..3 { + // The JoinHandle was consumed once; a failed stop cannot repoll it. + let retry = owner.drain().await.unwrap_err(); + assert_source(retry.as_ref()); + let observed = owner.observe(18).unwrap(); + assert_eq!( + observed.state, + NodeDurabilitySupervisorState::JoinedUnsettled + ); + assert!(Arc::ptr_eq( + observed.requests_error.as_ref().unwrap(), + &stopped + )); + assert!(Arc::ptr_eq( + observed.rotations.as_ref().unwrap_err(), + &stopped + )); + if let Some(native) = &original.supervisor_error { + assert!(Arc::ptr_eq( + observed.supervisor_error.as_ref().unwrap(), + native + )); + } + } + assert_eq!(runtime.stats().retained_bytes(), 4 * 1024); + drop(owner); + assert_eq!(runtime.stats().retained_bytes(), 0); + runtime.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn successful_task_keeps_original_stop_error_across_repeated_joins() { + joined_poisoned_bank(false).await; +} + +#[tokio::test] +async fn panicked_task_keeps_native_error_separate_from_original_stop_error() { + joined_poisoned_bank(true).await; +} diff --git a/crates/cellule-host/src/durability/supervisor.rs b/crates/cellule-host/src/durability/supervisor.rs new file mode 100644 index 00000000..631ffee3 --- /dev/null +++ b/crates/cellule-host/src/durability/supervisor.rs @@ -0,0 +1,275 @@ +//! One canonical supervisor for automatic and requested epoch rotation. +use super::*; +use requests::{RotationRequests, RotationWork}; + +pub(super) async fn run_node_durability_supervisor( + provider: Arc

, + runtime: CellRuntime, + configuration: NodeDurabilitySupervisorConfig, + session: SessionId, + cancellation: CancellationToken, + requests: Arc, +) -> FacilityResult { + let mut recruit = tokio::time::interval(configuration.recruit_interval); + let mut rotation = tokio::time::interval(configuration.rotation_interval); + loop { + if cancellation.is_cancelled() { + return Ok(()); + } + tokio::select! { + () = cancellation.cancelled() => return Ok(()), + _ = recruit.tick(), if runtime.node_durability().is_none() => { + match provider.clone().recruit(configuration.limits, configuration.required_follower_bytes, configuration.live_node_limit).await { + Ok(Some(config)) => { + if config.identity().0 != session { return Err(Box::new(Error::Control("node-log enrollment boot differs from host"))); } + match config.build() { + Ok(durability) => { + if cancellation.is_cancelled() { + close_replacement(&provider, None, &durability, configuration.recruit_interval).await?; + return Ok(()); + } + if let Err(error) = runtime.install_node_durability(configuration.application, Arc::clone(&durability)) { + provider.rotation_event(NodeDurabilityRotation::Failed); + close_replacement(&provider, None, &durability, configuration.recruit_interval).await?; + return Err(Box::new(error)); + } + } + Err(_error) => provider.rotation_event(NodeDurabilityRotation::Failed), + } + }, + Ok(None) => {} + Err(_) => provider.rotation_event(NodeDurabilityRotation::Failed), + } + } + _ = rotation.tick(), if runtime.node_durability().is_some() => { + rotate(Arc::clone(&provider), &runtime, configuration, &cancellation, &requests).await?; + } + () = requests.wake.notified() => { + rotate(Arc::clone(&provider), &runtime, configuration, &cancellation, &requests).await?; + } + } + } +} + +async fn rotate( + provider: Arc

, + runtime: &CellRuntime, + configuration: NodeDurabilitySupervisorConfig, + cancellation: &CancellationToken, + requests: &RotationRequests, +) -> FacilityResult { + let work = match requests.claim(runtime, configuration.max_issued_frames, false)? { + Some(work) => work, + None => { + match provider + .clone() + .rotation_required(configuration.live_node_limit) + .await + { + Ok(true) => {} + Ok(false) => return Ok(()), + Err(_) => { + provider.rotation_event(NodeDurabilityRotation::Failed); + return Ok(()); + } + } + if cancellation.is_cancelled() { + return Ok(()); + } + // Recheck the bounded request bank after provider I/O. Explicit + // maintenance and expiry-driven rotation share this one claim. + let Some(work) = requests.claim(runtime, configuration.max_issued_frames, true)? else { + return Ok(()); + }; + work + } + }; + provider.rotation_event(NodeDurabilityRotation::Started); + if let Some(record) = &work.record { + record.phase(NodeLogRotationPhase::Retiring)?; + } + loop { + let result = if let Some(record) = &work.record { + work.durability + .shutdown_for_maintenance() + .await + .and_then(|proof| record.retired(proof)) + } else { + work.durability.shutdown().await + }; + match result { + Ok(()) => break, + Err(Error::PendingPublication) => { + provider.rotation_event(NodeDurabilityRotation::Pending); + if !retry(cancellation, configuration.recruit_interval).await { + return Ok(()); + } + } + Err(error) => { + provider.rotation_event(NodeDurabilityRotation::Failed); + if let Some(record) = &work.record { + // Keep the claim and strict mode across retries. The normal + // timer cannot erase an unconfirmed maintenance obligation. + record.failed(Arc::new(error))?; + if !retry(cancellation, configuration.recruit_interval).await { + return Ok(()); + } + } else if work.durability.requires_confirmed_retirement() { + if !retry(cancellation, configuration.recruit_interval).await { + return Ok(()); + } + } else { + return Err(Box::new(error)); + } + } + } + } + if cancellation.is_cancelled() { + return Ok(()); + } + let replacement = loop { + let result = provider + .clone() + .recruit( + configuration.limits, + configuration.required_follower_bytes, + configuration.live_node_limit, + ) + .await; + match result { + Ok(Some(config)) => { + let identity = config.identity(); + let previous = work.durability.identity()?; + if identity.0 != previous.0 || identity.1 != previous.1 || identity.2 <= previous.2 + { + report_failure( + &provider, + &work, + Box::new(Error::Control( + "node-log replacement identity or epoch differs", + )), + )?; + // Reject foreign scope before building or closing it. An + // application must reconcile any prior recruitment CAS. + if !retry(cancellation, configuration.recruit_interval).await { + return Ok(()); + } + continue; + } + match config.build() { + Ok(durability) => break durability, + Err(error) => { + report_failure(&provider, &work, Box::new(error))?; + } + } + } + Ok(None) => {} + Err(error) => { + report_failure(&provider, &work, error)?; + } + } + if !retry(cancellation, configuration.recruit_interval).await { + return Ok(()); + } + }; + if cancellation.is_cancelled() { + // Recruitment may have committed authority before cancellation. Join + // its canonical close; a host deadline retains this supervisor's handle. + close_replacement( + &provider, + work.record.as_deref(), + &replacement, + configuration.recruit_interval, + ) + .await?; + return Ok(()); + } + let replacement_epoch = replacement.log_epoch()?; + if replacement_epoch <= work.epoch { + let error = Box::new(Error::Control("node-log replacement epoch did not advance")); + let source = report_failure(&provider, &work, error)?; + close_replacement( + &provider, + work.record.as_deref(), + &replacement, + configuration.recruit_interval, + ) + .await?; + return Err(Box::new(RetainedRotationError(source))); + } + match runtime.replace_node_durability( + configuration.application, + &work.durability, + Arc::clone(&replacement), + ) { + Ok(_) => { + requests.complete(&work, replacement_epoch)?; + provider.rotation_event(NodeDurabilityRotation::Completed); + Ok(()) + } + Err(error) => { + let source = report_failure(&provider, &work, Box::new(error))?; + close_replacement( + &provider, + work.record.as_deref(), + &replacement, + configuration.recruit_interval, + ) + .await?; + Err(Box::new(RetainedRotationError(source))) + } + } +} + +fn report_failure( + provider: &Arc

, + work: &RotationWork, + error: Box, +) -> cellule_runtime::Result> { + provider.rotation_event(NodeDurabilityRotation::Failed); + let error: Arc = Arc::from(error); + if let Some(record) = &work.record { + record.failed(Arc::clone(&error))?; + } + Ok(error) +} +#[derive(Debug)] +struct RetainedRotationError(Arc); +impl std::fmt::Display for RetainedRotationError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + self.0.fmt(f) + } +} +impl std::error::Error for RetainedRotationError { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + Some(self.0.as_ref()) + } +} +async fn retry(cancellation: &CancellationToken, interval: Duration) -> bool { + tokio::select! { + () = cancellation.cancelled() => false, + () = tokio::time::sleep(interval) => true, + } +} + +async fn close_replacement( + provider: &Arc

, + record: Option<&requests::RotationRecord>, + replacement: &Arc, + interval: Duration, +) -> FacilityResult { + loop { + match replacement.shutdown().await { + Ok(()) => return Ok(()), + Err(error) => { + provider.rotation_event(NodeDurabilityRotation::Failed); + if let Some(record) = record { + record.failed(Arc::new(error))?; + } + // Recruitment is already accepted. Work cancellation must not + // discard this exact generation or its canonical cleanup join. + tokio::time::sleep(interval).await; + } + } + } +} diff --git a/crates/cellule-host/src/fleet/actions.rs b/crates/cellule-host/src/fleet/actions.rs new file mode 100644 index 00000000..18a6f6bf --- /dev/null +++ b/crates/cellule-host/src/fleet/actions.rs @@ -0,0 +1,628 @@ +use std::sync::{Arc, Mutex}; + +use cellule_runtime::Error; +use cellule_runtime::cell::actor::{CellRuntime, NodeByteReservation}; +use cellule_runtime::fleet::operations::{ + AcceptedFleetAction, DrainBlocker, FleetAction, FleetActionKind, FleetActionOutcome, + FleetInspectionObservation, FleetInspectionRequest, FleetOutcome, FleetScope, + MAX_ACTIVE_ATTEMPTS, MAX_RECORD_BYTES, MaintenanceAction, OperationError, +}; +use cellule_runtime::identity::{Digest, NodeId, SessionId}; +use tokio::{sync::watch, task::JoinHandle}; + +use super::snapshot::{FleetNodeSnapshot, FleetSnapshotRequest, SnapshotOwners}; +use super::{FleetActionAcceptance, FleetActionJournal, FleetCellProvider}; + +/// Retained completion of one accepted canonical effect and journal write. +#[derive(Debug)] +pub struct FleetActionCompletion { + /// Exact original acceptance. Its record does not confer Cell authority. + pub accepted: AcceptedFleetAction, + /// Checked canonical result, or Unknown when an effect cannot be proved. + pub outcome: FleetActionOutcome, + /// True only after the journal confirmed durable result publication. + pub committed: bool, + /// Original runtime error, independently of result publication failure. + pub execution_error: Option>, + /// Original result-publication error. Retry publishes the same retained + /// evidence and never executes the source release again. + pub journal_error: Option>, +} + +pub(super) struct ActionResult { + pub(super) outcome: FleetOutcome, + pub(super) error: Option, +} + +impl ActionResult { + pub(super) fn checked(outcome: FleetOutcome) -> Self { + Self { + outcome, + error: None, + } + } + pub(super) fn refused(blocker: DrainBlocker, error: Error) -> Self { + Self { + outcome: FleetOutcome::Rejected(blocker), + error: Some(error), + } + } +} + +type EffectCompletion = Result, Arc>; +type Completion = Result, Arc>; + +enum JobCompletion { + Effect(Arc), + Inspection(Arc), + Snapshot(Arc), +} + +#[derive(Clone)] +enum JobRequest { + Effect { + action: FleetAction, + now_ms: i64, + }, + Inspection(FleetInspectionRequest), + Snapshot { + request: Box, + owners: Arc, + }, +} + +impl JobRequest { + fn scope(&self) -> FleetScope { + match self { + Self::Effect { action, .. } => action.scope(), + Self::Inspection(request) => request.action().scope(), + Self::Snapshot { request, .. } => request.expected().head().scope(), + } + } + fn validate_endpoint(&self, node: NodeId, session: SessionId) -> Result<(), OperationError> { + match self { + Self::Effect { action, .. } => action.validate_endpoint(node, session), + Self::Inspection(request) => request.validate_endpoint(node, session), + Self::Snapshot { request, .. } + if request.node() == node && request.session() == session => + { + Ok(()) + } + Self::Snapshot { .. } => Err(OperationError::Fenced), + } + } + fn key(&self) -> Result { + match self { + Self::Effect { action, .. } => action.key(), + Self::Inspection(request) => request.key(), + Self::Snapshot { request, .. } => request.key(), + } + } + fn validate_replay(&self, other: &Self) -> Result<(), OperationError> { + match (self, other) { + (Self::Effect { action, .. }, Self::Effect { action: replay, .. }) => { + action.validate_replay(replay) + } + (Self::Inspection(original), Self::Inspection(replay)) if original == replay => Ok(()), + ( + Self::Snapshot { + request: original, .. + }, + Self::Snapshot { + request: replay, .. + }, + ) if original == replay => Ok(()), + _ => Err(OperationError::Conflict), + } + } +} + +pub(crate) struct FleetActionExecutor { + pub(super) runtime: CellRuntime, + pub(super) scope: FleetScope, + pub(super) node: NodeId, + pub(super) session: SessionId, + pub(super) journal: Arc, + pub(super) cells: Arc, + pub(super) registry: Arc, + bank: Mutex, +} + +#[derive(Default)] +struct ActionBank { + draining: bool, + jobs: Vec>, + failure: Option>, +} + +struct ActionJob { + key: Digest, + request: JobRequest, + completion: watch::Receiver>, + task: tokio::sync::Mutex, + _retained: NodeByteReservation, +} + +struct ActionJoin { + task: Option>, + failure: Option>, +} + +impl ActionJoin { + async fn join(&mut self) -> Result<(), Arc> { + if let Some(task) = self.task.as_mut() { + let result = task.await; + self.task = None; + if let Err(source) = result { + self.failure = Some(Arc::new(Error::Facility { + name: "fleet-action-task", + source: Box::new(source), + })); + } + } + match &self.failure { + Some(error) => Err(Arc::clone(error)), + None => Ok(()), + } + } +} + +#[derive(Debug)] +struct RetainedActionFailure(Arc); + +impl std::fmt::Display for RetainedActionFailure { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + std::fmt::Display::fmt(self.0.as_ref(), f) + } +} + +impl std::error::Error for RetainedActionFailure { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + Some(self.0.as_ref()) + } +} + +fn retained_action_failure(error: Arc) -> Error { + Error::Facility { + name: "fleet-action-task", + source: Box::new(RetainedActionFailure(error)), + } +} + +pub(crate) fn operation(error: OperationError) -> Error { + Error::FleetOperation(Box::new(error)) +} + +pub(super) fn journal_error(source: Box) -> Error { + Error::Facility { + name: "fleet-action-journal", + source, + } +} + +impl FleetActionExecutor { + pub(crate) fn new( + runtime: CellRuntime, + scope: FleetScope, + node: NodeId, + session: SessionId, + journal: Arc, + cells: Arc, + registry: Arc, + ) -> cellule_runtime::Result { + cellule_runtime::fleet::operations::FleetHead::new(scope, 0).map_err(operation)?; + if !node.as_bytes().iter().any(|byte| *byte != 0) + || !session.as_bytes().iter().any(|byte| *byte != 0) + { + return Err(Error::Node("invalid fleet executor binding")); + } + Ok(Self { + runtime, + scope, + node, + session, + journal, + cells, + registry, + bank: Mutex::new(ActionBank::default()), + }) + } + + pub(crate) async fn apply( + self: &Arc, + action: FleetAction, + now_ms: i64, + ) -> EffectCompletion { + if matches!(action.kind(), FleetActionKind::Maintenance { action, .. } if *action != MaintenanceAction::Cordon) + { + return Err(Arc::new(Error::Control( + "fleet role settlement and finalization require their host barriers", + ))); + } + if matches!( + action.kind(), + FleetActionKind::Movement { + action: cellule_runtime::fleet::operations::MovementAction::Inspect, + .. + } + ) { + return Err(Arc::new(Error::Control( + "fleet Inspect requires request-bound inspection", + ))); + } + if now_ms < action.issued_at_ms() { + return Err(Arc::new(operation(OperationError::Invalid( + "fleet action time regressed", + )))); + } + let completion = self.submit(JobRequest::Effect { action, now_ms }).await?; + match completion.as_ref() { + JobCompletion::Effect(result) => Ok(Arc::clone(result)), + _ => Err(Arc::new(Error::Control( + "fleet action completion kind mismatch", + ))), + } + } + + pub(crate) async fn observe( + self: &Arc, + request: FleetInspectionRequest, + ) -> Result, Arc> { + if request.node() != self.node || request.session() != self.session { + return Err(Arc::new(Error::Fenced)); + } + if !matches!(request.action().kind(), FleetActionKind::Movement { .. }) { + return Err(Arc::new(Error::Control( + "fleet maintenance inspection requires its host inventory barrier", + ))); + } + let completion = self.submit(JobRequest::Inspection(request)).await?; + match completion.as_ref() { + JobCompletion::Inspection(result) => Ok(Arc::clone(result)), + _ => Err(Arc::new(Error::Control( + "fleet inspection completion kind mismatch", + ))), + } + } + + pub(crate) async fn snapshot( + self: &Arc, + request: FleetSnapshotRequest, + owners: SnapshotOwners, + ) -> Result, Arc> { + let completion = self + .submit(JobRequest::Snapshot { + request: Box::new(request), + owners: Arc::new(owners), + }) + .await?; + match completion.as_ref() { + JobCompletion::Snapshot(snapshot) => Ok(Arc::clone(snapshot)), + _ => Err(Arc::new(Error::Control( + "fleet native snapshot completion kind mismatch", + ))), + } + } + + async fn submit(self: &Arc, request: JobRequest) -> Completion { + if request.scope() != self.scope { + return Err(Arc::new(Error::PeerAuthorization( + "fleet action scope mismatch", + ))); + } + request + .validate_endpoint(self.node, self.session) + .map_err(operation) + .map_err(Arc::new)?; + self.reap().await.map_err(Arc::new)?; + let key = request.key().map_err(operation).map_err(Arc::new)?; + let (mut response, retained_job) = { + let mut bank = self + .bank + .lock() + .map_err(|_| Arc::new(Error::Control("fleet action bank poisoned")))?; + if bank.draining { + return Err(Arc::new(Error::CellDraining)); + } + if let Some(job) = bank.jobs.iter().find(|job| job.key == key) { + job.request + .validate_replay(&request) + .map_err(operation) + .map_err(Arc::new)?; + (job.completion.clone(), Arc::clone(job)) + } else { + if bank.jobs.len() >= MAX_ACTIVE_ATTEMPTS { + return Err(Arc::new(Error::Capacity("fleet action receipt bound"))); + } + // Both the accepted envelope and its checked result remain owned + // after an RPC waiter disappears, using the shared node ledger. + let retained = self + .runtime + .try_reserve_node_bytes(3 * MAX_RECORD_BYTES as usize) + .map_err(Arc::new)?; + let (sender, response) = watch::channel(None); + let executor = Arc::clone(self); + let issued = request.clone(); + let task = tokio::spawn(async move { + let result = match issued { + JobRequest::Effect { action, now_ms } => executor + .execute(action, now_ms) + .await + .map(JobCompletion::Effect), + JobRequest::Inspection(request) => executor + .execute_inspection(request) + .await + .map(JobCompletion::Inspection), + JobRequest::Snapshot { request, owners } => executor + .execute_snapshot(*request, owners) + .await + .map(JobCompletion::Snapshot), + } + .map(Arc::new); + let _ = sender.send(Some(result)); + }); + let job = Arc::new(ActionJob { + key, + request, + completion: response.clone(), + task: tokio::sync::Mutex::new(ActionJoin { + task: Some(task), + failure: None, + }), + _retained: retained, + }); + bank.jobs.push(Arc::clone(&job)); + (response, job) + } + }; + loop { + let completed = response.borrow().clone(); + if let Some(result) = completed { + // Sending completion precedes the task's exit. Join that short + // epilogue so an immediate retry can reap an unconfirmed result + // and retry publication instead of joining the stale receipt. + // The job stays owned if this waiter is dropped during the join. + retained_job.task.lock().await.join().await?; + return result; + } + if response.changed().await.is_err() { + // The waiter retains this exact job even if a concurrent drain + // removes its settled bank receipt. Preserve its own task error. + retained_job.task.lock().await.join().await?; + return Err(Arc::new(Error::Control( + "fleet action completion task ended", + ))); + } + } + } + + async fn execute(&self, action: FleetAction, now_ms: i64) -> EffectCompletion { + let acceptance = self + .journal + .accept_action(&action, self.node, self.session, now_ms) + .await + .map_err(journal_error) + .map_err(Arc::new)?; + let (accepted, result) = match acceptance { + FleetActionAcceptance::Existing { accepted, result } => { + accepted + .validate_replay(&action, self.node, self.session) + .map_err(operation) + .map_err(Arc::new)?; + if let Some(result) = result + && !matches!(result.outcome, FleetOutcome::Unknown) + { + accepted + .validate_result(&result) + .map_err(operation) + .map_err(Arc::new)?; + return Ok(Arc::new(FleetActionCompletion { + accepted, + outcome: *result, + committed: true, + execution_error: None, + journal_error: None, + })); + } + let result = self.inspect_accepted(&accepted).await; + (accepted, result) + } + FleetActionAcceptance::New(accepted) => { + accepted + .validate_replay(&action, self.node, self.session) + .map_err(operation) + .map_err(Arc::new)?; + if accepted.action() != &action || accepted.accepted_at_ms() != now_ms { + return Err(Arc::new(operation(OperationError::Conflict))); + } + let result = self.perform_action(&accepted).await; + (accepted, result) + } + }; + let (outcome, mut execution_error) = match result { + Ok(result) => (result.outcome, result.error.map(Arc::new)), + Err(error) => (FleetOutcome::Unknown, Some(Arc::new(error))), + }; + // Keep canonical proof even if the clock fails after release. An older + // timestamp is conservative, and the original clock error is retained. + let observed_at_ms = match wall_time_ms() { + Ok(now) if now >= accepted.accepted_at_ms() => now, + clock => { + if execution_error.is_none() { + execution_error = Some(Arc::new(match clock { + Err(error) => error, + Ok(_) => Error::Control("fleet action clock regressed"), + })); + } + accepted.accepted_at_ms() + } + }; + let outcome = FleetActionOutcome { + scope: action.scope(), + action_key: action.key().map_err(operation).map_err(Arc::new)?, + node: self.node, + session: self.session, + observed_at_ms, + outcome, + }; + accepted + .validate_result(&outcome) + .map_err(operation) + .map_err(Arc::new)?; + if let Some(error) = &execution_error { + tracing::warn!(action_key = ?outcome.action_key, node = ?self.node, + session = ?self.session, error = ?error, "fleet action retained execution error"); + } + Ok(Arc::new( + self.publish(accepted, outcome, execution_error).await, + )) + } + + async fn publish( + &self, + accepted: AcceptedFleetAction, + outcome: FleetActionOutcome, + execution_error: Option>, + ) -> FleetActionCompletion { + let journal_error = self + .journal + .publish_action_result(&accepted, &outcome) + .await + .err() + .map(journal_error) + .map(Arc::new); + FleetActionCompletion { + accepted, + outcome, + committed: journal_error.is_none(), + execution_error, + journal_error, + } + } + + async fn reap(&self) -> cellule_runtime::Result<()> { + let (jobs, mut first_error) = { + let bank = self + .bank + .lock() + .map_err(|_| Error::Control("fleet action bank poisoned"))?; + ( + bank.jobs.clone(), + bank.failure + .as_ref() + .map(|error| retained_action_failure(Arc::clone(error))), + ) + }; + for job in jobs { + let mut task = job.task.lock().await; + if task.task.as_ref().is_some_and(|task| !task.is_finished()) { + continue; + } + let result = self.finish_job(&job, &mut task).await; + if first_error.is_none() { + first_error = result.err(); + } + } + first_error.map_or(Ok(()), Err) + } + + pub(crate) async fn drain(&self) -> cellule_runtime::Result<()> { + { + let mut bank = self + .bank + .lock() + .map_err(|_| Error::Control("fleet action bank poisoned"))?; + bank.draining = true; + } + let (jobs, mut first_error) = { + let bank = self + .bank + .lock() + .map_err(|_| Error::Control("fleet action bank poisoned"))?; + ( + bank.jobs.clone(), + bank.failure + .as_ref() + .map(|error| retained_action_failure(Arc::clone(error))), + ) + }; + for job in jobs { + let mut task = job.task.lock().await; + // A failed job must not short-circuit the join of a sibling whose + // accepted work is still owned. Keep its original failure separately + // from the settled receipt, then continue all joins/publications. + let result = self.finish_job(&job, &mut task).await; + if first_error.is_none() { + first_error = result.err(); + } + } + first_error.map_or(Ok(()), Err) + } + + fn remember_failure(&self, error: Arc) -> cellule_runtime::Result<()> { + let mut bank = self + .bank + .lock() + .map_err(|_| Error::Control("fleet action bank poisoned"))?; + if bank.failure.is_none() { + bank.failure = Some(error); + } + Ok(()) + } + + async fn finish_job( + &self, + job: &Arc, + task: &mut ActionJoin, + ) -> cellule_runtime::Result<()> { + match task.join().await { + Err(error) => { + self.remember_failure(Arc::clone(&error))?; + self.remove_job(job)?; + Err(retained_action_failure(error)) + } + Ok(()) => self.settle_job(job).await, + } + } + + fn remove_job(&self, job: &Arc) -> cellule_runtime::Result<()> { + self.bank + .lock() + .map_err(|_| Error::Control("fleet action bank poisoned"))? + .jobs + .retain(|entry| !Arc::ptr_eq(entry, job)); + Ok(()) + } + + async fn settle_job(&self, job: &Arc) -> cellule_runtime::Result<()> { + let completion = job.completion.borrow().clone(); + if let Some(Ok(completion)) = completion + && let JobCompletion::Effect(completion) = completion.as_ref() + { + if !completion.committed { + // Retry the retained proof, never the canonical effect. Keep + // this receipt if publication or resource settlement still fails. + self.journal + .publish_action_result(&completion.accepted, &completion.outcome) + .await + .map_err(journal_error)?; + } + self.retire_receiver_receipt(completion).await?; + } + self.remove_job(job)?; + Ok(()) + } +} + +pub(super) fn wall_time_ms() -> cellule_runtime::Result { + let elapsed = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map_err(|source| Error::Facility { + name: "fleet-action-clock", + source: Box::new(source), + })?; + i64::try_from(elapsed.as_millis()).map_err(|source| Error::Facility { + name: "fleet-action-clock", + source: Box::new(source), + }) +} diff --git a/crates/cellule-host/src/fleet/cells.rs b/crates/cellule-host/src/fleet/cells.rs new file mode 100644 index 00000000..91fd5c0e --- /dev/null +++ b/crates/cellule-host/src/fleet/cells.rs @@ -0,0 +1,55 @@ +use super::FleetAdapterFuture; +use cellule_runtime::cell::catalog::CatalogProof; +use cellule_runtime::control::{Owner, authority::CellAuthority}; +use cellule_runtime::fleet::operations::MoveAttemptSpec; +use cellule_runtime::ltx::CellReplica; +use std::path::PathBuf; + +/// Trusted application inputs for one exact receiver or observation. +#[derive(Clone)] +pub struct FleetCellInputs { + /// Verified catalog target; the host checks exact attempt scope. + pub catalog: CatalogProof, + /// Immutable storage operations bound to the Cell and incarnation. + pub replica: CellReplica, + /// Existing canonical Cell authority in that application's storage layout. + pub authority: CellAuthority, + /// Private local SQLite destination supplied during trusted composition. + pub destination: PathBuf, + /// Local leased session and advertised endpoint used by ordinary acquisition. + pub owner: Owner, +} + +/// Canonical failed-session proof and manifest access from ordinary recovery. +#[derive(Clone)] +pub struct FleetRecoveryInputs { + /// Proof obtained only through the node directory/recovery coordinator. + pub takeover: cellule_runtime::node::NodeTakeoverProof, + /// Existing manifest store for the exact control-pinned recovery overlay. + pub manifests: cellule_runtime::recovery::manifest::RecoveryManifestStore, +} + +/// Application-owned lookup of catalog, storage and private local paths. +/// +/// Install this trusted adapter at startup; remote actions never supply local +/// filesystem paths, credentials, or authority constructors. Lookup must not +/// mutate Cell authority, hydrate a database, or activate a writer. Actual +/// preparation, resource admission, and acquisition remain on the runtime path. +pub trait FleetCellProvider: Send + Sync + 'static { + /// Resolves bounded inputs for the immutable movement specification. + fn cell_inputs<'a>( + &'a self, + spec: &'a MoveAttemptSpec, + ) -> FleetAdapterFuture<'a, FleetCellInputs>; + + /// Resolves existing canonical recovery proof. This lookup must not fence + /// a node, seal a log, publish an overlay, or acquire a Cell. The ordinary + /// recovery coordinator establishes those prerequisites independently. + /// Fresh recovered-serving inspection repeats this read-only lookup after + /// acquisition and result publication. Keep the original pinned manifest's + /// canonical backend available; reconstruct inputs without repeating effects. + fn recovery_inputs<'a>( + &'a self, + spec: &'a MoveAttemptSpec, + ) -> FleetAdapterFuture<'a, FleetRecoveryInputs>; +} diff --git a/crates/cellule-host/src/fleet/controller.rs b/crates/cellule-host/src/fleet/controller.rs new file mode 100644 index 00000000..d9159ba8 --- /dev/null +++ b/crates/cellule-host/src/fleet/controller.rs @@ -0,0 +1,139 @@ +use cellule_runtime::fleet::operations::{ + EnrollmentPage, FleetHead, FleetScope, IntentPage, JournalTransition, MaintenanceOperation, + OperationError, OperationId, ProgressPage, RegistryVersion, +}; +use cellule_runtime::identity::{CellId, Digest, IncarnationId, NodeId, SessionId}; + +use super::{FleetActionJournal, FleetAdapterFuture, FleetEnrollmentJournal}; + +/// Head and registry metadata read together in one consistent transaction. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct FleetJournalSnapshot { + head: FleetHead, + registry: RegistryVersion, +} + +impl FleetJournalSnapshot { + /// Binds the same application scope without claiming observation coverage. + pub fn new(head: FleetHead, registry: RegistryVersion) -> Result { + if head.scope() != registry.scope() { + return Err(OperationError::Conflict); + } + Ok(Self { head, registry }) + } + /// Returns the current charged attempts and controller fencing epoch. + #[must_use] + pub const fn head(&self) -> &FleetHead { + &self.head + } + /// Returns retained intent/enrollment revision and operator scheduling mode. + #[must_use] + pub const fn registry(&self) -> RegistryVersion { + self.registry + } +} + +/// Controller and action transactions on one shared, durable application journal. +/// +/// Implementations bind one validated FleetProfile at construction. Every method +/// preserves original backend errors. The action acceptance methods inherited +/// here must use the same transaction domain as these methods, not a separate +/// read-then-write cache. Applications own caller authorization and storage. +pub trait FleetJournal: FleetActionJournal + FleetEnrollmentJournal { + /// Reads the head and registry metadata from one consistent transaction. + fn load_snapshot(&self, scope: FleetScope) -> FleetAdapterFuture<'_, FleetJournalSnapshot>; + + /// CAS-acquires/renews the controller using `FleetHead::claim`. An ambiguous + /// reply requires rereading the same head, never deleting its permits. + fn claim_controller( + &self, + scope: FleetScope, + expected_revision: u64, + claimant: SessionId, + now_ms: i64, + ) -> FleetAdapterFuture<'_, FleetJournalSnapshot>; + + /// Atomically publishes one reducer transition against both current versions. + /// + /// For Allocate, load exact source/receiver intents and invoke the registry's + /// `authorize_allocation` inside this transaction before the head transition. + /// For maintenance, advance and retain that physical node's intent in the + /// same commit. Retain all prior operations for return-to-service proofs. + /// For Retire, publish the exact progress page with permit retirement; a + /// failed CAS cannot expose committed history or release either budget. + /// Finalization must compare the observed registry version again here. + /// ResolveUnaccepted must prove no accepted record exists for its exact + /// effect/attempt/endpoint in this same transaction before changing the head. + /// Existing acceptance (even with no result) rejects resolution; delayed + /// old envelopes then fail the new head revision after successful absence CAS. + fn compare_exchange<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + controller_epoch: u64, + now_ms: i64, + transition: &'a JournalTransition, + ) -> FleetAdapterFuture<'a, FleetJournalSnapshot>; + + /// Loads retained operations, including those older than the current head. + fn load_operation( + &self, + scope: FleetScope, + operation: OperationId, + ) -> FleetAdapterFuture<'_, Option>; + + /// Loads an immutable progress page. Verify scope and canonical page digest + /// against the requested digest; a successful orphan PUT is not committed. + fn load_progress( + &self, + scope: FleetScope, + digest: Digest, + ) -> FleetAdapterFuture<'_, Option>; + + /// Reads the greatest confirmed movement completion time for this exact + /// Cell incarnation from history committed at `expected`. Cancelled attempts + /// do not count as movement. Compare the full snapshot in the same read + /// transaction; an orphan progress page cannot contribute to cooldown. + /// An indexed backend may use an index updated atomically with retirement. + fn last_moved_at<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + cell: CellId, + incarnation: IncarnationId, + ) -> FleetAdapterFuture<'a, Option>; + + /// Greatest committed movement completion time across the scope. Count + /// balancing requires every signed sample to follow this post-batch barrier. + /// Compare the full snapshot and exclude cancellation/orphan pages. + fn last_movement_at<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + ) -> FleetAdapterFuture<'a, Option>; + + /// Returns retained physical-node rows, including every earlier cordon. + /// Require the exact version and a limit in 1..=128 before any allocation. + /// Reject changed revisions rather than returning a filtered current page. + fn intents_page( + &self, + version: RegistryVersion, + after: Option, + limit: usize, + ) -> FleetAdapterFuture<'_, IntentPage>; + + /// Returns all enrollment rows, including failed, pending and retired boots. + /// Require the exact version and bounded limit; never omit expired sessions. + fn enrollments_page( + &self, + version: RegistryVersion, + after: Option, + limit: usize, + ) -> FleetAdapterFuture<'_, EnrollmentPage>; + + /// Atomically applies revision-checked stop/resume through the registry. + /// Stop blocks new Allocate transactions and preserves accepted work, every + /// charged permit, retained intent, and all immutable evidence/history. + fn set_scheduling( + &self, + expected: RegistryVersion, + enabled: bool, + ) -> FleetAdapterFuture<'_, RegistryVersion>; +} diff --git a/crates/cellule-host/src/fleet/coverage/followers.rs b/crates/cellule-host/src/fleet/coverage/followers.rs new file mode 100644 index 00000000..06229f97 --- /dev/null +++ b/crates/cellule-host/src/fleet/coverage/followers.rs @@ -0,0 +1,143 @@ +use super::*; +use cellule_runtime::fleet::operations::{EnrollmentEvent, EnrollmentRole}; +use cellule_runtime::identity::{NodeId, SessionId}; +use cellule_runtime::node::{FollowerLogObservation, LogLeaderState, log_state::NodeLogPhase}; + +pub(super) fn match_authority( + roster: &FleetRoster, + native: &HashMap<(NodeId, SessionId), &FleetNodeInventory>, + foreign: &HashMap, +) -> Result> { + let mut observed = HashMap::<(NodeId, SessionId), &FollowerLogObservation>::new(); + let mut empty_lanes = HashSet::new(); + for references in foreign.values() { + for reference in references.entries() { + if reference.log.phase() == NodeLogPhase::Retired { + // Terminal directory rows remain visible through their existing + // grace boundary. They grant no failed-process join or deletion. + continue; + } + if reference.leader_state != LogLeaderState::Live + || reference.log.phase() != NodeLogPhase::Open + { + return Err(Error::Control( + "failed follower owner requires canonical recovery", + )); + } + let source = native + .get(&(reference.leader_node, reference.leader)) + .ok_or(Error::Control("authoritative follower owner is unobserved"))?; + if source.node_log() + != Some(( + reference.leader, + reference.leader_node, + reference.log.epoch(), + )) + || !source.bindings().follower_producer + || !source.bindings().durability_supervisor + { + return Err(Error::Control( + "authoritative follower owner has no managed binding", + )); + } + if let Some(previous) = + observed.insert((reference.leader_node, reference.leader), reference) + && previous != reference + { + return Err(Error::Control( + "foreign follower authority intervals differ", + )); + } + let progress = source + .follower_enrollments() + .iter() + .find(|progress| progress.epoch == reference.log.epoch()) + .ok_or(Error::Fenced)?; + if !progress.delivered + || !progress.native_started + || progress.native_closed + || progress.enrollment.is_none() + || progress.refusal.is_some() + || progress.members.len() != reference.log.members().len() + { + return Err(Error::Control( + "authoritative follower producer is incomplete", + )); + } + let selected = progress + .members + .iter() + .map(|member| member.spec.target.node) + .collect::>(); + if selected != reference.log.members() { + return Err(Error::Fenced); + } + for member in &progress.members { + let spec = &member.spec; + let row = roster + .enrollments() + .iter() + .find(|row| row.spec() == spec) + .ok_or(Error::Fenced)?; + if row.status() != EnrollmentStatus::Established + || !member.published + || !matches!(member.event, Some(EnrollmentEvent::Established(evidence)) if Some(evidence) == row.established_evidence()) + || member.accepted.as_ref().is_none_or(|accepted| { + accepted.spec() != spec || accepted.accepted_at_ms() != row.accepted_at_ms() + }) + || spec.source.is_none_or(|source| { + source.node != reference.leader_node || source.session != reference.leader + }) + || !matches!(spec.role, EnrollmentRole::Follower { log_epoch } if log_epoch == reference.log.epoch()) + { + return Err(Error::Fenced); + } + let target = + native + .get(&(spec.target.node, spec.target.session)) + .ok_or(Error::Control( + "authoritative follower receiver is unobserved", + ))?; + if !target.bindings().follower_store + || target + .follower_store_state() + .is_none_or(|(_, quarantine)| quarantine != 0) + { + return Err(Error::Control( + "authoritative follower receiver store is incomplete", + )); + } + let counterpart = foreign.get(&spec.target.node).ok_or(Error::Fenced)?; + if !counterpart.entries().iter().any(|other| other == reference) { + return Err(Error::Control( + "complete follower ensemble authority is missing", + )); + } + // The directory enrollment installs the obligation before any + // append opens a persisted lane. This exact cross-node witness + // supplies observation of that obligation, never its absence. + empty_lanes.insert(spec.key().map_err(super::super::operation)?); + } + } + } + for inventory in native.values() { + if let Some((session, node, epoch)) = inventory.node_log() + && observed + .get(&(node, session)) + .is_none_or(|reference| reference.log.epoch() != epoch) + { + return Err(Error::Control( + "native bound follower log has no current authority", + )); + } + if inventory + .follower_store_state() + .is_some_and(|(_, quarantine)| quarantine != 0) + { + return Err(Error::Control( + "native follower inventory has quarantined entries", + )); + } + } + Ok(empty_lanes) +} diff --git a/crates/cellule-host/src/fleet/coverage/mod.rs b/crates/cellule-host/src/fleet/coverage/mod.rs new file mode 100644 index 00000000..be7e8fae --- /dev/null +++ b/crates/cellule-host/src/fleet/coverage/mod.rs @@ -0,0 +1,211 @@ +//! Cross-node original producer, persisted lane and current authority matching. +use super::{FleetFollowerReferences, FleetJournalSnapshot, FleetNodeInventory, FleetRoster}; +use cellule_runtime::fleet::operations::EnrollmentStatus; +use cellule_runtime::{Error, Result, identity::Digest}; +use std::collections::{HashMap, HashSet}; + +mod followers; + +/// Checked native/foreign role graph at a complete retained roster barrier. +/// +/// Collect every required original boot and every retained physical node's +/// foreign references, then recheck all native categories, every exact foreign +/// row and the full journal. Authentication, unexpected advertisement discovery, +/// current Cell authority, replacement policy and failed-process closure remain +/// separate adapter duties. This interval proves coverage, not settled roles, +/// atomicity, nonexecution of Pending work or permission to stop a node. +pub struct FleetRoleCoverage { + snapshot: FleetJournalSnapshot, + roster: Digest, + digest: Digest, + started_at_ms: i64, + finished_at_ms: i64, + native_boots: usize, + physical_nodes: usize, + pending_enrollments: usize, +} + +impl FleetRoleCoverage { + /// Checks a complete original graph after the ordered global rechecks. + /// + /// An Established but not-yet-appended lane is observed only through its + /// exact delivered managed producer, current Open authority and installed + /// source binding. It remains an obligation. Missing/duplicate boots or + /// reference scans, incomplete/dropped rechecks and failed owners block this + /// proof. Buffers stay with the supplied collectors; applications account + /// the bounded temporary indexes (at most 10,000 entries per collection). + pub fn check( + roster: &FleetRoster, + native: &[&FleetNodeInventory], + foreign: &[&FleetFollowerReferences], + now_ms: i64, + ) -> Result { + if roster.snapshot().registry().bootstrap_revision().is_none() + || native.len() > 10_000 + || foreign.len() > 10_000 + { + return Err(Error::Fenced); + } + let roster_digest = roster.digest()?; + let mut boots = HashMap::new(); + for inventory in native { + if !inventory.bindings().managed_startup + || inventory.roster_digest() != roster_digest + || boots + .insert((inventory.node(), inventory.session()), *inventory) + .is_some() + { + return Err(Error::Fenced); + } + roster.boot(inventory.node(), inventory.session())?; + } + let required = roster.required_boots(); + if boots.len() != required.len() + || required + .iter() + .any(|boot| !boots.contains_key(&(boot.node, boot.session))) + { + return Err(Error::Control("required fleet boot inventory is missing")); + } + let mut members = HashMap::new(); + for references in foreign { + references.validate_enrollments(roster)?; + if members.insert(references.member(), *references).is_some() { + return Err(Error::Fenced); + } + } + if members.len() != roster.intents().len() + || roster + .intents() + .iter() + .any(|intent| !members.contains_key(&intent.node())) + { + return Err(Error::Control( + "physical follower reference inventory is missing", + )); + } + let started_at_ms = native + .iter() + .map(|i| i.interval().0) + .chain(foreign.iter().map(|i| i.interval().0)) + .min() + .ok_or(Error::Fenced)?; + let collected = native + .iter() + .map(|i| i.coverage_checkpoint().0) + .chain(foreign.iter().map(|i| i.coverage_checkpoint().0)) + .max() + .ok_or(Error::Fenced)?; + let mut native_finished = collected; + for inventory in native { + let (started, finished) = inventory + .coverage_checkpoint() + .1 + .ok_or(Error::Control("global native recheck is incomplete"))?; + if started < collected || finished < started { + return Err(Error::Fenced); + } + native_finished = native_finished.max(finished); + } + let mut finished_at_ms = native_finished; + for references in foreign { + let (started, finished) = references + .coverage_checkpoint() + .1 + .ok_or(Error::Control("global follower recheck is incomplete"))?; + if started < native_finished || finished < started { + return Err(Error::Fenced); + } + finished_at_ms = finished_at_ms.max(finished); + } + if started_at_ms < 0 + || finished_at_ms > now_ms + || now_ms < started_at_ms + || now_ms - started_at_ms > 30_000 + { + return Err(Error::Fenced); + } + let empty_lanes = followers::match_authority(roster, &boots, &members)?; + for inventory in native { + inventory.validate_enrollments_with(roster, |row| { + row.spec().key().is_ok_and(|key| empty_lanes.contains(&key)) + })?; + } + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-role-coverage.v1\0"); + hash.update(roster_digest.as_bytes()); + hash.update(&started_at_ms.to_be_bytes()); + hash.update(&finished_at_ms.to_be_bytes()); + let mut ordered_native = native.to_vec(); + ordered_native.sort_by_key(|i| (*i.node().as_bytes(), *i.session().as_bytes())); + hash.update(&(native.len() as u64).to_be_bytes()); + for inventory in ordered_native { + hash.update(inventory.coverage_digest()?.as_bytes()); + let (collected, checked) = inventory.coverage_checkpoint(); + let (begin, end) = checked.ok_or(Error::Fenced)?; + for time in [inventory.interval().0, collected, begin, end] { + hash.update(&time.to_be_bytes()); + } + } + let mut ordered_foreign = foreign.to_vec(); + ordered_foreign.sort_by_key(|i| *i.member().as_bytes()); + hash.update(&(foreign.len() as u64).to_be_bytes()); + for references in ordered_foreign { + hash.update(references.coverage_digest().as_bytes()); + let (collected, checked) = references.coverage_checkpoint(); + let (begin, end) = checked.ok_or(Error::Fenced)?; + for time in [references.interval().0, collected, begin, end] { + hash.update(&time.to_be_bytes()); + } + } + Ok(Self { + snapshot: roster.snapshot().clone(), + roster: roster_digest, + digest: Digest::from_bytes(*hash.finalize().as_bytes()), + started_at_ms, + finished_at_ms, + native_boots: native.len(), + physical_nodes: foreign.len(), + pending_enrollments: roster + .enrollments() + .iter() + .filter(|row| row.status() == EnrollmentStatus::Pending) + .count(), + }) + } + /// Exact head/registry bound by every original native request and roster page. + #[must_use] + pub fn snapshot(&self) -> &FleetJournalSnapshot { + &self.snapshot + } + /// Identifies original roster inputs, including terminal rows and timestamps. + #[must_use] + pub const fn roster_digest(&self) -> Digest { + self.roster + } + /// Identifies this interval's graph; it supplies no authentication or authority. + #[must_use] + pub const fn digest(&self) -> Digest { + self.digest + } + /// Original collection through the final exact foreign rechecks, without restamping. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + (self.started_at_ms, self.finished_at_ms) + } + /// Every unresolved original boot, including required earlier process sessions. + #[must_use] + pub const fn native_boots(&self) -> usize { + self.native_boots + } + /// Every retained physical intent, including nodes with no local writers. + #[must_use] + pub const fn physical_nodes(&self) -> usize { + self.physical_nodes + } + /// Pending registry rows remain obligations despite complete inventory coverage. + #[must_use] + pub const fn pending_enrollments(&self) -> usize { + self.pending_enrollments + } +} diff --git a/crates/cellule-host/src/fleet/enrollment.rs b/crates/cellule-host/src/fleet/enrollment.rs new file mode 100644 index 00000000..d5906bcc --- /dev/null +++ b/crates/cellule-host/src/fleet/enrollment.rs @@ -0,0 +1,162 @@ +use cellule_runtime::fleet::operations::{ + EnrollmentEvent, EnrollmentRecord, EnrollmentRole, EnrollmentSpec, EnrollmentStatus, + FleetScope, NodeIntent, OperationError, OperationId, RegistryVersion, +}; +use cellule_runtime::identity::{Digest, NodeId, SessionId}; + +use super::FleetAdapterFuture; + +/// Current physical intent and established boot obligation read atomically. +/// Canonical advertisement evidence is verified by the trusted journal adapter. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct FleetBootObservation { + intent: NodeIntent, + enrollment: EnrollmentRecord, +} + +impl FleetBootObservation { + /// Checks an established boot against the same transaction's current intent. + /// An Active enrollment may finish after a cordon, without reopening it. + pub fn new(intent: NodeIntent, enrollment: EnrollmentRecord) -> Result { + intent.to_bytes()?; + enrollment.to_bytes()?; + let spec = enrollment.spec(); + let EnrollmentRole::Node { mode } = spec.role else { + return Err(OperationError::Conflict); + }; + if enrollment.status() != EnrollmentStatus::Established + || spec.source.is_some() + || spec.scope != intent.scope() + || spec.target.node != intent.node() + || spec.target.session != intent.session() + || spec.target.intent_revision > intent.revision() + || (spec.target.intent_revision == intent.revision() && mode != intent.mode()) + { + return Err(OperationError::Conflict); + } + Ok(Self { intent, enrollment }) + } + /// Returns the current retained physical-node intent. + #[must_use] + pub const fn intent(&self) -> &NodeIntent { + &self.intent + } + /// Returns the established original boot obligation and canonical evidence. + #[must_use] + pub const fn enrollment(&self) -> &EnrollmentRecord { + &self.enrollment + } +} + +/// Atomic pending enrollment or its original retained request. +pub enum FleetEnrollmentAcceptance { + /// Pending acceptance committed with current intent checks. Only this + /// result permits first execution of the ordinary enrollment protocol. + New(EnrollmentRecord), + /// Original request, including lost replies and terminal tombstones. + /// Pending state does not authorize repeating an unobserved enrollment. + Existing(EnrollmentRecord), +} + +/// Enrollment and retained-intent transactions sharing the controller journal. +/// +/// Authentication, bootstrap coverage and canonical role evidence are supplied +/// by the application. Successful writes must survive backend/client restart. +/// Advance the shared RegistryVersion atomically with every actual row change; +/// exact duplicates return original records without refreshing evidence time. +pub trait FleetEnrollmentJournal: Send + Sync + 'static { + /// Atomically settles joined work whose native enrollment never started. + /// Validate complete immutable inputs and trusted nonexecution evidence. + /// If absent, retain `unexecuted_refusal`; if Pending, apply that same refusal. + /// An existing identical refusal replays its original time. Established or + /// differently settled rows conflict. Advance RegistryVersion in the same + /// transaction. A delayed acceptance must return this terminal row, never New. + /// This is not an absence read or authority to cancel an unobserved native CAS. + fn refuse_unexecuted_enrollment<'a>( + &'a self, + spec: &'a EnrollmentSpec, + evidence: Digest, + now_ms: i64, + ) -> FleetAdapterFuture<'a, EnrollmentRecord>; + + /// Loads current intent and the exact established boot in one transaction. + /// Missing rows return None. Pending/refused/retired, foreign-role/session + /// or contradictory rows fail closed. Never return a cached older intent. + fn load_boot( + &self, + scope: FleetScope, + node: NodeId, + key: Digest, + ) -> FleetAdapterFuture<'_, Option>; + + /// Creates an initial Active physical-node row only when absent. An existing + /// identical row is idempotent; any different row conflicts. Never overwrite + /// an older retained cordon with a boot's default configuration. + fn register_initial_intent<'a>( + &'a self, + intent: &'a NodeIntent, + ) -> FleetAdapterFuture<'a, NodeIntent>; + + /// Rebinds an Active intent after application-validated authoritative old + /// withdrawal or complete failed-boot closure, and new boot enrollment. + /// Failed closure includes original process/accepted external work joining + /// and every original role settled; expiry or takeover alone is insufficient. + /// Check the exact original intent and + /// derive the successor with `rebind_active` inside the shared transaction. + fn rebind_active_intent<'a>( + &'a self, + original: &'a NodeIntent, + session: SessionId, + revision: u64, + ) -> FleetAdapterFuture<'a, NodeIntent>; + + /// Loads the exact retained completed operation and current physical-node + /// intent together, then uses `return_to_service`. New boot validation and + /// operator authorization precede this call. A different operation conflicts. + fn return_to_service( + &self, + scope: FleetScope, + node: NodeId, + operation: OperationId, + new_session: SessionId, + revision: u64, + ) -> FleetAdapterFuture<'_, NodeIntent>; + + /// Commits the controlled initial coverage barrier at the exact revision. + /// The caller pauses all producers and imports all existing live/failed + /// obligations first. Do not infer bootstrap from an empty live directory. + fn bootstrap_registry( + &self, + expected: RegistryVersion, + ) -> FleetAdapterFuture<'_, RegistryVersion>; + + /// First acceptance checks all spec intent revisions and Active mode for a + /// new reader/follower role in the same commit as Pending. Boot enrollment + /// must honor its exact retained mode and cannot open readiness. Existing + /// requests compare full original + /// inputs before current intent checks; a cordon cannot erase admitted work. + fn accept_enrollment<'a>( + &'a self, + spec: &'a EnrollmentSpec, + now_ms: i64, + ) -> FleetAdapterFuture<'a, FleetEnrollmentAcceptance>; + + /// Confirms an original acceptance and canonical result in one transaction. + /// Compare immutable spec and original acceptance time, load current progress + /// and apply the event. Evidence must cover the exact role/session/epoch. + /// Lost replies leave pending work charged as an obligation until inspection + /// proves a definite refusal, completion, or canonical closure/retirement. + fn publish_enrollment_result<'a>( + &'a self, + original: &'a EnrollmentRecord, + event: EnrollmentEvent, + now_ms: i64, + ) -> FleetAdapterFuture<'a, EnrollmentRecord>; + + /// Reads retained request progress, including failed-session tombstones. + fn load_enrollment( + &self, + scope: FleetScope, + key: Digest, + ) -> FleetAdapterFuture<'_, Option>; +} diff --git a/crates/cellule-host/src/fleet/failed_boot/mod.rs b/crates/cellule-host/src/fleet/failed_boot/mod.rs new file mode 100644 index 00000000..de1acba8 --- /dev/null +++ b/crates/cellule-host/src/fleet/failed_boot/mod.rs @@ -0,0 +1,296 @@ +//! Exact original boot retirement after canonical and external process closure. +use super::{FleetAdapterFuture, FleetJournal, FleetJournalSnapshot, FleetRoster, operation}; +use cellule_runtime::{ + Error, Result, + fleet::operations::{EnrollmentEvent, EnrollmentRecord, EnrollmentRole, EnrollmentStatus}, + identity::{Digest, SessionId}, + node::{NodeDirectory, NodeSessionClosure, NodeSessionFence}, +}; +use std::{future::Future, sync::Arc}; +use tokio::time::{Instant, timeout_at}; + +mod writers; +pub use writers::{ + FleetOriginalCatalogSet, FleetOriginalCatalogSource, FleetOriginalCatalogs, + FleetOriginalWriterCapture, FleetOriginalWriterInventory, FleetOriginalWriterJournal, +}; +mod process; +mod publication; +mod retained; +pub use process::FleetFailedBootProcessConfirmation; +mod readers; +pub use readers::{ + FleetFailedReaderClosure, FleetFailedReaderPublication, FleetFailedReaderRetirement, +}; +mod records; + +/// Immutable original boot and canonical fence to verify at the process provider. +/// Neither an expired lease nor successful data recovery answers this request. +/// Snapshot, mutable boot status and collection interval are read metadata; +/// providers bind retained original lifetime evidence to [`Self::digest`]. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct FleetFailedBootProcessRequest { + boot: EnrollmentRecord, + canonical: Option, + fence: NodeSessionFence, + digest: Digest, + snapshot: FleetJournalSnapshot, + started_at_ms: i64, + finished_at_ms: i64, +} +impl FleetFailedBootProcessRequest { + /// Full barrier used for the original boot/fence capture. + #[must_use] + pub fn snapshot(&self) -> &FleetJournalSnapshot { + &self.snapshot + } + /// Original process-request collection interval, without restamping. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + (self.started_at_ms, self.finished_at_ms) + } + + /// Original request, acceptance time and establishment evidence. Mutable + /// journal status is not part of the immutable request digest. + #[must_use] + pub const fn boot(&self) -> &EnrollmentRecord { + &self.boot + } + /// Original permanent physical/session fence, independent of log recovery. + #[must_use] + pub const fn fence(&self) -> &NodeSessionFence { + &self.fence + } + /// Original terminal-log observation for a legacy terminal capture. A fenced + /// capture supplies no such assertion; boot retirement checks it separately. + #[must_use] + pub fn canonical(&self) -> Option<&NodeSessionClosure> { + self.canonical.as_ref() + } + /// Stable process request identity across original retirement/reconstruction. + #[must_use] + pub const fn digest(&self) -> Digest { + self.digest + } +} + +/// Application-authenticated durable evidence of this original process lifetime. +/// The witness identifies retained termination/nonexecution evidence, not a PID, +/// timeout, lease expiry, native absence, or recovery result. Construct only +/// after satisfying [`FleetFailedBootProcesses`]'s contract. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct FleetFailedBootProcessEvidence { + request: Digest, + witness: Digest, +} +impl FleetFailedBootProcessEvidence { + /// Binds independently verified process evidence to the complete request. + /// This constructor checks shape; application authentication and actual + /// process/accepted external work joining remain the provider's duties. + pub fn new(request: &FleetFailedBootProcessRequest, witness: Digest) -> Result { + if witness.as_bytes().iter().all(|byte| *byte == 0) { + return Err(Error::Fenced); + } + Ok(Self { + request: request.digest, + witness, + }) + } + /// Exact immutable boot/fence request confirmed by the provider. + #[must_use] + pub const fn request_digest(&self) -> Digest { + self.request + } + /// Stable identity of the application's retained original process evidence. + #[must_use] + pub const fn witness(&self) -> Digest { + self.witness + } +} + +/// Read-only provider of original-boot process termination/nonexecution evidence. +/// +/// Authenticate the exact physical node/session and original establishment, +/// join termination of that process and all its accepted external jobs/producers, +/// and exclude later execution/restart of the same session. Retain the immutable +/// evidence durably outside canonical Cell storage; rereads after adapter or +/// controller restart return the same witness. A successor process, reusable PID, +/// missing inventory, expiry, sealed/Retired data, timeout or lost reply cannot +/// supply this evidence. This method must not start a new kill/drain operation. +/// Own such effects in the application's existing supervised finite work first. +/// Cell relocation and reader/follower policy remain separate evidence. +pub trait FleetFailedBootProcesses: Send + Sync { + /// Reconfirms the original durable evidence, preserving source failures. + fn confirm_stopped<'a>( + &'a self, + request: &'a FleetFailedBootProcessRequest, + ) -> FleetAdapterFuture<'a, FleetFailedBootProcessEvidence>; +} + +/// Original boot capsule after all related enrollment rows and leader log close. +/// Capturing it starts no process/native effects. Applications account bounded +/// metadata and own publication in their accepted finite work. This value does +/// not establish affected-writer relocation or grant maintenance finalization. +pub struct FleetFailedBootRetirement { + snapshot: FleetJournalSnapshot, + request: FleetFailedBootProcessRequest, + canonical: NodeSessionClosure, + started_at_ms: i64, + finished_at_ms: i64, +} +impl FleetFailedBootRetirement { + /// Checks the exact original Established boot (or its replay), complete + /// bootstrapped roster and fresh canonical terminal session. Any unresolved + /// related reader/follower/Pending request prevents capture, even after a + /// new physical intent or successful recovery. Foreign terminal references + /// remain retained under the native grace/collection contracts. + pub async fn capture( + journal: &dyn FleetJournal, + directory: &NodeDirectory, + roster: &FleetRoster, + original: &EnrollmentRecord, + claimant: SessionId, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + let request = FleetFailedBootProcessRequest::capture( + journal, directory, roster, original, claimant, deadline, &mut clock, + ) + .await?; + Self::capture_retained( + journal, directory, roster, &request, claimant, deadline, clock, + ) + .await + } + /// Full original barrier checked before any publication. + #[must_use] + pub fn snapshot(&self) -> &FleetJournalSnapshot { + &self.snapshot + } + /// Exact process request to confirm through the application provider. + #[must_use] + pub const fn request(&self) -> &FleetFailedBootProcessRequest { + &self.request + } + /// Original capture interval, which publication cannot restamp. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + (self.started_at_ms, self.finished_at_ms) + } +} + +/// Durable boot publication plus its independently retained final check. +/// A successful write remains inspectable if a later barrier/provider read fails. +pub struct FleetFailedBootPublication { + process: FleetFailedBootProcessEvidence, + record: std::result::Result>, + closure: std::result::Result>, +} +impl FleetFailedBootPublication { + /// Exact immutable process evidence used for this original publication. + #[must_use] + pub const fn process(&self) -> &FleetFailedBootProcessEvidence { + &self.process + } + /// Original returned retirement row or source error, including a lost reply. + pub fn record(&self) -> std::result::Result<&EnrollmentRecord, Arc> { + self.record.as_ref().map_err(Arc::clone) + } + /// Complete post-publication roster, canonical and process confirmation. + pub fn confirmed(&self) -> Result<&FleetFailedBootClosure> { + self.closure + .as_ref() + .map_err(|source| retained(Arc::clone(source))) + } + /// Original final-check failure without discarding the journal response. + #[must_use] + pub fn closure_error(&self) -> Option> { + self.closure.as_ref().err().cloned() + } +} + +/// Checked original boot retirement. This is interval evidence, not a completed +/// operation: replacements, affected writers and finalization CAS remain required. +pub struct FleetFailedBootClosure { + snapshot: FleetJournalSnapshot, + boot: EnrollmentRecord, + canonical: NodeSessionClosure, + process: FleetFailedBootProcessEvidence, + digest: Digest, + started_at_ms: i64, + finished_at_ms: i64, +} +impl FleetFailedBootClosure { + /// Complete post-publication head and registry, rechecked around all evidence. + #[must_use] + pub fn snapshot(&self) -> &FleetJournalSnapshot { + &self.snapshot + } + /// Original retired request, including first acceptance and establishment. + #[must_use] + pub const fn boot(&self) -> &EnrollmentRecord { + &self.boot + } + /// Permanent canonical physical/session fence and retained terminal log. + #[must_use] + pub const fn canonical(&self) -> &NodeSessionClosure { + &self.canonical + } + /// Application's original durable process lifetime evidence. + #[must_use] + pub const fn process(&self) -> &FleetFailedBootProcessEvidence { + &self.process + } + /// Identifies original retirement/evidence without refreshing timestamps. + #[must_use] + pub const fn digest(&self) -> Digest { + self.digest + } + /// Original capture through fresh final confirmation. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + (self.started_at_ms, self.finished_at_ms) + } +} + +fn interval(start: i64, end: i64) -> Result<()> { + if start < 0 || end < start || end - start > 30_000 { + return Err(Error::Deadline); + } + Ok(()) +} +async fn bounded(deadline: Instant, future: impl Future>) -> Result { + if Instant::now() >= deadline { + return Err(Error::Deadline); + } + timeout_at(deadline, future) + .await + .map_err(|source| Error::Facility { + name: "fleet-failed-boot-deadline", + source: Box::new(source), + })? +} +fn adapter_error(source: Box) -> Error { + Error::Facility { + name: "fleet-failed-boot-adapter", + source, + } +} +#[derive(Debug)] +struct RetainedError(Arc); +impl std::fmt::Display for RetainedError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + self.0.fmt(f) + } +} +impl std::error::Error for RetainedError { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + Some(self.0.as_ref()) + } +} +fn retained(source: Arc) -> Error { + Error::Facility { + name: "fleet-failed-boot-publication", + source: Box::new(RetainedError(source)), + } +} diff --git a/crates/cellule-host/src/fleet/failed_boot/process.rs b/crates/cellule-host/src/fleet/failed_boot/process.rs new file mode 100644 index 00000000..9e0b881c --- /dev/null +++ b/crates/cellule-host/src/fleet/failed_boot/process.rs @@ -0,0 +1,214 @@ +use super::*; + +/// Fresh complete-roster, permanent-fence and original-process confirmation. +/// +/// Original process identity does not depend on recovery progress. This interval +/// proof supplies no affected-Cell inventory, log closure or role settlement. +pub struct FleetFailedBootProcessConfirmation { + snapshot: FleetJournalSnapshot, + fence: NodeSessionFence, + process: FleetFailedBootProcessEvidence, + started_at_ms: i64, + finished_at_ms: i64, +} +impl FleetFailedBootProcessConfirmation { + /// Full journal barrier checked around the process provider reads. + #[must_use] + pub fn snapshot(&self) -> &FleetJournalSnapshot { + &self.snapshot + } + /// Original permanent fence; it makes no terminal-log assertion. + #[must_use] + pub const fn fence(&self) -> &NodeSessionFence { + &self.fence + } + /// Independently authenticated, durable original lifetime evidence. + #[must_use] + pub const fn process(&self) -> &FleetFailedBootProcessEvidence { + &self.process + } + /// Fresh monotonic confirmation interval, without restamping the request. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + (self.started_at_ms, self.finished_at_ms) + } +} + +impl FleetFailedBootProcessRequest { + /// Captures a terminal original boot/log basis. Existing v1 request digests + /// and retirement event identities remain unchanged. Related reader/follower + /// rows may be unresolved; this request supplies no process closure itself. + #[allow(clippy::too_many_arguments)] + pub async fn capture( + journal: &dyn FleetJournal, + directory: &NodeDirectory, + roster: &FleetRoster, + original: &EnrollmentRecord, + claimant: SessionId, + deadline: Instant, + clock: impl FnMut() -> Result, + ) -> Result { + Self::capture_basis( + journal, directory, roster, original, claimant, deadline, clock, true, + ) + .await + } + + /// Captures original process identity immediately after permanent fencing, + /// while the leader log and related roles may still need recovery/retirement. + /// The v2 digest binds the immutable boot and permanent fence, excluding log + /// phase, manifest, claimant, registry and observation times. It remains the + /// same through recovery, claim adoption and terminal retirement. Obtain + /// actual process/accepted-work joining through [`Self::confirm`]; this read + /// starts no process, recovery or native effect. + #[allow(clippy::too_many_arguments)] + pub async fn capture_fenced( + journal: &dyn FleetJournal, + directory: &NodeDirectory, + roster: &FleetRoster, + original: &EnrollmentRecord, + claimant: SessionId, + deadline: Instant, + clock: impl FnMut() -> Result, + ) -> Result { + Self::capture_basis( + journal, directory, roster, original, claimant, deadline, clock, false, + ) + .await + } + + #[allow(clippy::too_many_arguments)] + async fn capture_basis( + journal: &dyn FleetJournal, + directory: &NodeDirectory, + roster: &FleetRoster, + original: &EnrollmentRecord, + claimant: SessionId, + deadline: Instant, + mut clock: impl FnMut() -> Result, + terminal: bool, + ) -> Result { + let started_at_ms = clock()?; + if directory.fleet() != roster.snapshot().head().scope().fleet + || roster.snapshot().registry().bootstrap_revision().is_none() + { + return Err(Error::Fenced); + } + roster.confirm(journal, deadline).await?; + let boot = records::boot(roster, original)?; + let endpoint = boot.spec().target; + let canonical = if terminal { + Some( + bounded( + deadline, + directory.closed_session( + endpoint.node, + endpoint.session, + claimant, + started_at_ms, + ), + ) + .await?, + ) + } else { + None + }; + let fence = match &canonical { + Some(closure) => closure.fence(), + None => { + bounded( + deadline, + directory.fenced_session( + endpoint.node, + endpoint.session, + claimant, + started_at_ms, + ), + ) + .await? + } + }; + roster.confirm(journal, deadline).await?; + let finished_at_ms = clock()?; + interval(started_at_ms, finished_at_ms)?; + let digest = match &canonical { + Some(closure) => records::request_digest(&boot, closure)?, + None => records::fenced_request_digest(&boot, &fence)?, + }; + Ok(Self { + boot, + canonical, + fence, + digest, + snapshot: roster.snapshot().clone(), + started_at_ms, + finished_at_ms, + }) + } + + /// Freshly confirms the original process before dependent recovery/inventory + /// effects. Collects the complete current roster, rechecks its original boot, + /// canonical basis and provider twice, then confirms the full barrier. The + /// read-only provider owns actual authentication and original native/external + /// lifetime joining. Cancellation starts no replacement effect here. + #[allow(clippy::too_many_arguments)] + pub async fn confirm( + &self, + journal: &dyn FleetJournal, + directory: &NodeDirectory, + processes: &dyn FleetFailedBootProcesses, + claimant: SessionId, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + let started_at_ms = clock()?; + let mut last = started_at_ms; + if started_at_ms < self.finished_at_ms { + return Err(Error::Deadline); + } + interval(started_at_ms, started_at_ms)?; + let mut now = || { + let next = clock()?; + if next < last { + return Err(Error::Deadline); + } + interval(started_at_ms, next)?; + last = next; + Ok(next) + }; + let snapshot = bounded(deadline, async { + journal + .load_snapshot(self.snapshot.head().scope()) + .await + .map_err(adapter_error) + }) + .await?; + let roster = FleetRoster::collect(journal, &snapshot, deadline).await?; + if snapshot.registry().bootstrap_revision().is_none() { + return Err(Error::Fenced); + } + records::boot(&roster, &self.boot)?; + self.confirm_canonical(directory, claimant, now()?, deadline) + .await?; + let process = self.confirm_process(processes, deadline).await?; + roster.confirm(journal, deadline).await?; + self.confirm_canonical(directory, claimant, now()?, deadline) + .await?; + if self.confirm_process(processes, deadline).await? != process { + return Err(Error::Control("original failed process evidence changed")); + } + // A suspended second provider read cannot hide a changed journal or + // expired claimant. Keep this final basis check after both reads. + roster.confirm(journal, deadline).await?; + self.confirm_canonical(directory, claimant, now()?, deadline) + .await?; + let finished_at_ms = now()?; + Ok(FleetFailedBootProcessConfirmation { + snapshot, + fence: self.fence.clone(), + process, + started_at_ms, + finished_at_ms, + }) + } +} diff --git a/crates/cellule-host/src/fleet/failed_boot/publication.rs b/crates/cellule-host/src/fleet/failed_boot/publication.rs new file mode 100644 index 00000000..d0f7d4c5 --- /dev/null +++ b/crates/cellule-host/src/fleet/failed_boot/publication.rs @@ -0,0 +1,214 @@ +use super::*; + +impl FleetFailedBootRetirement { + /// Confirms durable original process evidence before publishing boot closure + /// through the existing journal. Rereads complete roster, canonical fence, + /// physical follower references and process evidence afterward. Cancellation + /// or a waiter deadline does not join native/backend work; its existing + /// application/adapter owners retain and join accepted work. Fresh recapture + /// adopts a committed lost reply with unchanged evidence and original times. + #[allow(clippy::too_many_arguments)] + pub async fn publish( + &self, + journal: &dyn FleetJournal, + directory: &NodeDirectory, + processes: &dyn FleetFailedBootProcesses, + claimant: SessionId, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + let mut last = self.finished_at_ms; + let mut clock = || { + let next = clock()?; + if next < last { + return Err(Error::Deadline); + } + interval(self.started_at_ms, next)?; + last = next; + Ok(next) + }; + let now = clock()?; + self.confirm_snapshot(journal, deadline).await?; + self.confirm_terminal(directory, claimant, now, deadline) + .await?; + let process = self.request.confirm_process(processes, deadline).await?; + let evidence = records::retirement(&process); + if self.request.boot.status() == EnrollmentStatus::Retired + && self.request.boot.settlement_evidence() != Some(evidence) + { + return Err(Error::Control("failed boot retirement evidence differs")); + } + // The provider read can suspend. Recheck the entire original barrier + // before committing any evidence; a new related role cannot be ignored. + self.confirm_snapshot(journal, deadline).await?; + let now = clock()?; + self.confirm_terminal(directory, claimant, now, deadline) + .await?; + let record = bounded(deadline, async { + let returned = journal + .publish_enrollment_result( + &self.request.boot, + EnrollmentEvent::Retired(evidence), + now, + ) + .await + .map_err(adapter_error)?; + records::result(&self.request.boot, &returned, evidence)?; + Ok(returned) + }) + .await + .map_err(Arc::new); + let closure = self + .finish( + journal, directory, processes, claimant, deadline, &mut clock, &process, &record, + ) + .await + .map_err(Arc::new); + Ok(FleetFailedBootPublication { + process, + record, + closure, + }) + } + + async fn confirm_snapshot(&self, journal: &dyn FleetJournal, deadline: Instant) -> Result<()> { + let snapshot = bounded(deadline, async { + journal + .load_snapshot(self.snapshot.head().scope()) + .await + .map_err(adapter_error) + }) + .await?; + if snapshot != self.snapshot { + return Err(Error::Fenced); + } + Ok(()) + } + #[allow(clippy::too_many_arguments)] + async fn finish( + &self, + journal: &dyn FleetJournal, + directory: &NodeDirectory, + processes: &dyn FleetFailedBootProcesses, + claimant: SessionId, + deadline: Instant, + clock: &mut impl FnMut() -> Result, + process: &FleetFailedBootProcessEvidence, + record: &std::result::Result>, + ) -> Result { + let returned = record + .as_ref() + .map_err(|source| retained(Arc::clone(source)))?; + let snapshot = bounded(deadline, async { + journal + .load_snapshot(self.snapshot.head().scope()) + .await + .map_err(adapter_error) + }) + .await?; + let roster = FleetRoster::collect(journal, &snapshot, deadline).await?; + let boot = records::select(&roster, &self.request.boot)?; + if &boot != returned { + return Err(Error::Control("failed boot publication changed")); + } + records::result(&self.request.boot, &boot, records::retirement(process))?; + records::references( + directory, + &roster, + boot.spec().target.node, + boot.spec().target.session, + deadline, + clock, + ) + .await?; + self.confirm_terminal(directory, claimant, clock()?, deadline) + .await?; + if &self.request.confirm_process(processes, deadline).await? != process { + return Err(Error::Control("original failed process evidence changed")); + } + roster.confirm(journal, deadline).await?; + let finished_at_ms = clock()?; + Ok(FleetFailedBootClosure { + snapshot, + digest: records::closure_digest(&boot, process)?, + boot, + canonical: self.canonical.clone(), + process: process.clone(), + started_at_ms: self.started_at_ms, + finished_at_ms, + }) + } + + async fn confirm_terminal( + &self, + directory: &NodeDirectory, + claimant: SessionId, + now: i64, + deadline: Instant, + ) -> Result<()> { + self.request + .confirm_canonical(directory, claimant, now, deadline) + .await?; + let endpoint = self.request.boot.spec().target; + let current = bounded( + deadline, + directory.closed_session(endpoint.node, endpoint.session, claimant, now), + ) + .await?; + if current != self.canonical { + return Err(Error::Fenced); + } + Ok(()) + } +} + +impl FleetFailedBootProcessRequest { + pub(super) async fn confirm_canonical( + &self, + directory: &NodeDirectory, + claimant: SessionId, + now: i64, + deadline: Instant, + ) -> Result<()> { + if directory.fleet() != self.snapshot.head().scope().fleet { + return Err(Error::Fenced); + } + let endpoint = self.boot.spec().target; + if let Some(original) = &self.canonical { + let current = bounded( + deadline, + directory.closed_session(endpoint.node, endpoint.session, claimant, now), + ) + .await?; + if ¤t != original { + return Err(Error::Fenced); + } + } else { + let current = bounded( + deadline, + directory.fenced_session(endpoint.node, endpoint.session, claimant, now), + ) + .await?; + if current != self.fence { + return Err(Error::Fenced); + } + } + Ok(()) + } + pub(super) async fn confirm_process( + &self, + processes: &dyn FleetFailedBootProcesses, + deadline: Instant, + ) -> Result { + let process = bounded(deadline, async { + processes.confirm_stopped(self).await.map_err(adapter_error) + }) + .await?; + if process.request != self.digest + || process.witness.as_bytes().iter().all(|byte| *byte == 0) + { + return Err(Error::Fenced); + } + Ok(process) + } +} diff --git a/crates/cellule-host/src/fleet/failed_boot/readers/mod.rs b/crates/cellule-host/src/fleet/failed_boot/readers/mod.rs new file mode 100644 index 00000000..d34d088f --- /dev/null +++ b/crates/cellule-host/src/fleet/failed_boot/readers/mod.rs @@ -0,0 +1,128 @@ +//! Original receiver lifetime closure; replacement policy remains independent. +use super::*; + +mod publication; +mod records; + +/// Exact reader responsibility on a canonically fenced original receiver boot. +/// Capture starts no native effect. Publication requires independently joined +/// original process and accepted-work evidence from the application provider. +/// Source failure cannot authorize retirement on a live receiver. +pub struct FleetFailedReaderRetirement { + request: FleetFailedBootProcessRequest, + reader: EnrollmentRecord, +} +impl FleetFailedReaderRetirement { + /// Captures a Pending, Established or replayed Retired reader at the same + /// complete bootstrap barrier as its original Established receiver boot. + /// Other unresolved roles remain obligations and prevent boot retirement. + #[allow(clippy::too_many_arguments)] + pub async fn capture( + journal: &dyn FleetJournal, + directory: &NodeDirectory, + roster: &FleetRoster, + boot: &EnrollmentRecord, + original: &EnrollmentRecord, + claimant: SessionId, + deadline: Instant, + clock: impl FnMut() -> Result, + ) -> Result { + let request = FleetFailedBootProcessRequest::capture( + journal, directory, roster, boot, claimant, deadline, clock, + ) + .await?; + let reader = records::select(roster, &request, original)?; + Ok(Self { request, reader }) + } + /// Full original head and registry barrier. + #[must_use] + pub fn snapshot(&self) -> &FleetJournalSnapshot { + self.request.snapshot() + } + /// Exact original receiver process request; role publication does not + /// alter its immutable identity or authorize the receiver's boot closure. + #[must_use] + pub const fn request(&self) -> &FleetFailedBootProcessRequest { + &self.request + } + /// Original reader request, acceptance and retained establishment history. + #[must_use] + pub const fn reader(&self) -> &EnrollmentRecord { + &self.reader + } + /// Original collection interval, without restamping on publication. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + self.request.interval() + } +} + +/// Retains the journal response and final confirmation independently. +/// A committed row remains inspectable after a final-check failure; a lost +/// reply retains its original source error for reconstruction and adoption. +pub struct FleetFailedReaderPublication { + process: FleetFailedBootProcessEvidence, + record: std::result::Result>, + closure: std::result::Result>, +} +impl FleetFailedReaderPublication { + /// Durable original receiver lifetime witness used by publication. + #[must_use] + pub const fn process(&self) -> &FleetFailedBootProcessEvidence { + &self.process + } + /// Original returned row or retained publication source error. + pub fn record(&self) -> std::result::Result<&EnrollmentRecord, Arc> { + self.record.as_ref().map_err(Arc::clone) + } + /// Checked original reader retirement after all final reads. + pub fn confirmed(&self) -> Result<&FleetFailedReaderClosure> { + self.closure + .as_ref() + .map_err(|source| retained(Arc::clone(source))) + } + /// Original final-check failure without discarding the journal response. + #[must_use] + pub fn closure_error(&self) -> Option> { + self.closure.as_ref().err().cloned() + } +} + +/// Interval evidence that one original reader's receiver cannot execute again. +/// This does not prove replacement redundancy, writer relocation, other role +/// closure, boot retirement or completed physical maintenance. +pub struct FleetFailedReaderClosure { + snapshot: FleetJournalSnapshot, + reader: EnrollmentRecord, + process: FleetFailedBootProcessEvidence, + digest: Digest, + started_at_ms: i64, + finished_at_ms: i64, +} +impl FleetFailedReaderClosure { + /// Complete post-publication barrier, confirmed around authority/provider reads. + #[must_use] + pub fn snapshot(&self) -> &FleetJournalSnapshot { + &self.snapshot + } + /// Original retired reader, including acceptance and establishment history. + #[must_use] + pub const fn reader(&self) -> &EnrollmentRecord { + &self.reader + } + /// Application-authenticated original receiver lifetime evidence. + #[must_use] + pub const fn process(&self) -> &FleetFailedBootProcessEvidence { + &self.process + } + /// Stable identity of the original terminal row and process witness. + #[must_use] + pub const fn digest(&self) -> Digest { + self.digest + } + /// Original capture through final confirmation, bounded to thirty seconds. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + (self.started_at_ms, self.finished_at_ms) + } +} diff --git a/crates/cellule-host/src/fleet/failed_boot/readers/publication.rs b/crates/cellule-host/src/fleet/failed_boot/readers/publication.rs new file mode 100644 index 00000000..96b08376 --- /dev/null +++ b/crates/cellule-host/src/fleet/failed_boot/readers/publication.rs @@ -0,0 +1,134 @@ +use super::*; + +impl FleetFailedReaderRetirement { + /// Confirms the exact original receiver's durable process and accepted-work + /// closure, then publishes through the existing enrollment journal. No + /// native drain or new task starts here. The adapter owns and joins accepted + /// backend work across cancellation; fresh recapture adopts lost replies + /// with the same event, original timestamps and establishment history. + pub async fn publish( + &self, + journal: &dyn FleetJournal, + directory: &NodeDirectory, + processes: &dyn FleetFailedBootProcesses, + claimant: SessionId, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + let (started_at_ms, mut last) = self.interval(); + let mut clock = || { + let next = clock()?; + if next < last { + return Err(Error::Deadline); + } + interval(started_at_ms, next)?; + last = next; + Ok(next) + }; + let now = clock()?; + self.confirm_snapshot(journal, deadline).await?; + self.request + .confirm_canonical(directory, claimant, now, deadline) + .await?; + let process = self.request.confirm_process(processes, deadline).await?; + let evidence = records::retirement(&self.reader, &process)?; + if self.reader.status() == EnrollmentStatus::Retired + && self.reader.settlement_evidence() != Some(evidence) + { + return Err(Error::Control("failed reader retirement evidence differs")); + } + // The provider can suspend. Any change to the original complete barrier + // requires recapture before effects; expiry is never process evidence. + self.confirm_snapshot(journal, deadline).await?; + let now = clock()?; + self.request + .confirm_canonical(directory, claimant, now, deadline) + .await?; + let record = bounded(deadline, async { + let returned = journal + .publish_enrollment_result(&self.reader, EnrollmentEvent::Retired(evidence), now) + .await + .map_err(adapter_error)?; + records::result(&self.reader, &returned, evidence)?; + Ok(returned) + }) + .await + .map_err(Arc::new); + let closure = self + .finish( + journal, directory, processes, claimant, deadline, &mut clock, &process, &record, + ) + .await + .map_err(Arc::new); + Ok(FleetFailedReaderPublication { + process, + record, + closure, + }) + } + + async fn confirm_snapshot(&self, journal: &dyn FleetJournal, deadline: Instant) -> Result<()> { + let snapshot = bounded(deadline, async { + journal + .load_snapshot(self.snapshot().head().scope()) + .await + .map_err(adapter_error) + }) + .await?; + if &snapshot != self.snapshot() { + return Err(Error::Fenced); + } + Ok(()) + } + + #[allow(clippy::too_many_arguments)] + async fn finish( + &self, + journal: &dyn FleetJournal, + directory: &NodeDirectory, + processes: &dyn FleetFailedBootProcesses, + claimant: SessionId, + deadline: Instant, + clock: &mut impl FnMut() -> Result, + process: &FleetFailedBootProcessEvidence, + record: &std::result::Result>, + ) -> Result { + let returned = record + .as_ref() + .map_err(|source| retained(Arc::clone(source)))?; + let snapshot = bounded(deadline, async { + journal + .load_snapshot(self.snapshot().head().scope()) + .await + .map_err(adapter_error) + }) + .await?; + let roster = FleetRoster::collect(journal, &snapshot, deadline).await?; + super::super::records::boot(&roster, &self.request.boot)?; + let reader = records::select(&roster, &self.request, &self.reader)?; + if &reader != returned { + return Err(Error::Control("failed reader publication changed")); + } + records::result( + &self.reader, + &reader, + records::retirement(&self.reader, process)?, + )?; + self.request + .confirm_canonical(directory, claimant, clock()?, deadline) + .await?; + if &self.request.confirm_process(processes, deadline).await? != process { + return Err(Error::Control("original failed receiver evidence changed")); + } + roster.confirm(journal, deadline).await?; + let finished_at_ms = clock()?; + Ok(FleetFailedReaderClosure { + snapshot, + digest: records::closure_digest(&reader, process)?, + reader, + process: process.clone(), + started_at_ms: self.request.started_at_ms, + finished_at_ms, + }) + } +} diff --git a/crates/cellule-host/src/fleet/failed_boot/readers/records.rs b/crates/cellule-host/src/fleet/failed_boot/readers/records.rs new file mode 100644 index 00000000..c08c35c6 --- /dev/null +++ b/crates/cellule-host/src/fleet/failed_boot/readers/records.rs @@ -0,0 +1,86 @@ +use super::*; + +pub(super) fn select( + roster: &FleetRoster, + request: &FleetFailedBootProcessRequest, + original: &EnrollmentRecord, +) -> Result { + original.to_bytes().map_err(operation)?; + let spec = original.spec(); + if !matches!(spec.role, EnrollmentRole::Reader { .. }) + || spec.scope != roster.snapshot().head().scope() + || spec.target.node != request.boot.spec().target.node + || spec.target.session != request.boot.spec().target.session + || spec.target.intent_revision < request.boot.spec().target.intent_revision + || original.status() == EnrollmentStatus::Refused + { + return Err(Error::Fenced); + } + let key = spec.key().map_err(operation)?; + let current = roster + .enrollments() + .iter() + .find(|row| row.spec().key().is_ok_and(|candidate| candidate == key)) + .ok_or(Error::Fenced)?; + history(original, current)?; + if original.status() == EnrollmentStatus::Retired && original != current { + return Err(Error::Fenced); + } + Ok(current.clone()) +} + +fn history(original: &EnrollmentRecord, returned: &EnrollmentRecord) -> Result<()> { + returned + .validate_replay(original.spec()) + .map_err(operation)?; + if returned.accepted_at_ms() != original.accepted_at_ms() + || returned.updated_at_ms() < original.updated_at_ms() + || returned.status() == EnrollmentStatus::Refused + || original + .established_evidence() + .is_some_and(|evidence| returned.established_evidence() != Some(evidence)) + { + return Err(Error::Fenced); + } + Ok(()) +} + +pub(super) fn retirement( + original: &EnrollmentRecord, + process: &FleetFailedBootProcessEvidence, +) -> Result { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-failed-reader-retirement.v1\0"); + hash.update(process.request.as_bytes()); + hash.update(process.witness.as_bytes()); + hash.update(&original.spec().to_bytes().map_err(operation)?); + hash.update(&original.accepted_at_ms().to_be_bytes()); + // Establishment may complete after the original capture. The journal's + // canonical transition preserves that history; it cannot change this event. + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) +} + +pub(super) fn result( + original: &EnrollmentRecord, + returned: &EnrollmentRecord, + evidence: Digest, +) -> Result<()> { + history(original, returned)?; + if returned.status() != EnrollmentStatus::Retired + || returned.settlement_evidence() != Some(evidence) + { + return Err(Error::Fenced); + } + Ok(()) +} + +pub(super) fn closure_digest( + reader: &EnrollmentRecord, + process: &FleetFailedBootProcessEvidence, +) -> Result { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-failed-reader-closure.v1\0"); + hash.update(&reader.to_bytes().map_err(operation)?); + hash.update(retirement(reader, process)?.as_bytes()); + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) +} diff --git a/crates/cellule-host/src/fleet/failed_boot/records.rs b/crates/cellule-host/src/fleet/failed_boot/records.rs new file mode 100644 index 00000000..62e010e3 --- /dev/null +++ b/crates/cellule-host/src/fleet/failed_boot/records.rs @@ -0,0 +1,189 @@ +use super::*; +use crate::fleet::FleetFollowerReferences; +use cellule_runtime::{identity::NodeId, node::log_state::NodeLogPhase}; + +pub(super) fn boot(roster: &FleetRoster, original: &EnrollmentRecord) -> Result { + original.to_bytes().map_err(operation)?; + let spec = original.spec(); + if !matches!(spec.role, EnrollmentRole::Node { .. }) + || spec.source.is_some() + || spec.scope != roster.snapshot().head().scope() + || original.established_evidence().is_none() + || !matches!( + original.status(), + EnrollmentStatus::Established | EnrollmentStatus::Retired + ) + { + return Err(Error::Fenced); + } + let mut boots = roster.enrollments().iter().filter(|row| { + matches!(row.spec().role, EnrollmentRole::Node { .. }) + && row.spec().target.node == spec.target.node + && row.spec().target.session == spec.target.session + && row.status() != EnrollmentStatus::Refused + }); + let current = boots.next().ok_or(Error::Fenced)?; + current.validate_replay(spec).map_err(operation)?; + if boots.next().is_some() + || current.accepted_at_ms() != original.accepted_at_ms() + || current.established_evidence() != original.established_evidence() + || current.updated_at_ms() < original.updated_at_ms() + || !matches!( + current.status(), + EnrollmentStatus::Established | EnrollmentStatus::Retired + ) + { + return Err(Error::Fenced); + } + Ok(current.clone()) +} + +pub(super) fn select( + roster: &FleetRoster, + original: &EnrollmentRecord, +) -> Result { + let current = boot(roster, original)?; + let spec = current.spec(); + for row in roster.enrollments() { + if row.spec() == spec || !row.unresolved() { + continue; + } + if row + .spec() + .source + .into_iter() + .chain(std::iter::once(row.spec().target)) + .any(|endpoint| { + endpoint.node == spec.target.node && endpoint.session == spec.target.session + }) + { + return Err(Error::Control( + "failed boot has unresolved enrollment responsibilities", + )); + } + } + Ok(current) +} + +pub(super) async fn references( + directory: &NodeDirectory, + roster: &FleetRoster, + node: NodeId, + session: SessionId, + deadline: Instant, + clock: &mut impl FnMut() -> Result, +) -> Result<()> { + let references = + FleetFollowerReferences::collect(directory, roster, node, 128, deadline, clock).await?; + references.validate_enrollments(roster)?; + for reference in references.entries() { + if reference.log.phase() == NodeLogPhase::Retired { + continue; + } + let matching = |row: &&EnrollmentRecord| { + row.spec().target.node == node + && row.spec().source.is_some_and(|source| { + source.node == reference.leader_node && source.session == reference.leader + }) + && matches!(row.spec().role, EnrollmentRole::Follower { log_epoch } if log_epoch == reference.log.epoch()) + }; + let mut original = false; + let mut replacement = false; + for row in roster.enrollments().iter().filter(matching) { + original |= + row.spec().target.session == session && row.status() != EnrollmentStatus::Refused; + replacement |= row.spec().target.session != session && row.unresolved(); + } + // A successor may carry other epochs on the same physical node. It + // cannot substitute for this original boot's unresolved native lane or + // make a contradictory Retired registry row agree with live authority. + if original || !replacement { + return Err(Error::Control( + "failed boot retains a foreign follower responsibility", + )); + } + } + Ok(()) +} + +pub(super) fn request_digest( + boot: &EnrollmentRecord, + canonical: &NodeSessionClosure, +) -> Result { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-failed-boot-process-request.v1\0"); + hash.update(&boot.spec().to_bytes().map_err(operation)?); + hash.update(&boot.accepted_at_ms().to_be_bytes()); + hash.update(boot.established_evidence().ok_or(Error::Fenced)?.as_bytes()); + hash.update(canonical.node().as_bytes()); + hash.update(canonical.session().as_bytes()); + hash.update(&canonical.expires_at_ms().to_be_bytes()); + hash.update(&canonical.retired_at_ms().to_be_bytes()); + hash.update(&[u8::from(canonical.log().is_some())]); + if let Some(log) = canonical.log() { + hash.update(&log.epoch().to_be_bytes()); + hash.update(b"retired\0"); + hash.update(&[u8::from(log.active())]); + hash.update(&log.tiered_through().to_be_bytes()); + hash.update(&(log.members().len() as u64).to_be_bytes()); + for member in log.members() { + hash.update(member.as_bytes()); + } + hash.update(&[u8::from(log.recovery_manifest().is_some())]); + if let Some(manifest) = log.recovery_manifest() { + hash.update(manifest.as_bytes()); + } + } + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) +} +pub(super) fn retirement(process: &FleetFailedBootProcessEvidence) -> Digest { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-failed-boot-retirement.v1\0"); + hash.update(process.request.as_bytes()); + hash.update(process.witness.as_bytes()); + Digest::from_bytes(*hash.finalize().as_bytes()) +} + +pub(super) fn fenced_request_digest( + boot: &EnrollmentRecord, + fence: &NodeSessionFence, +) -> Result { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-failed-boot-process-request.v2\0"); + hash.update(&boot.spec().to_bytes().map_err(operation)?); + hash.update(&boot.accepted_at_ms().to_be_bytes()); + hash.update(boot.established_evidence().ok_or(Error::Fenced)?.as_bytes()); + hash.update(fence.node().as_bytes()); + hash.update(fence.session().as_bytes()); + hash.update(&fence.expires_at_ms().to_be_bytes()); + hash.update(&fence.retired_at_ms().to_be_bytes()); + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) +} +pub(super) fn result( + original: &EnrollmentRecord, + returned: &EnrollmentRecord, + evidence: Digest, +) -> Result<()> { + returned + .validate_replay(original.spec()) + .map_err(operation)?; + if returned.status() != EnrollmentStatus::Retired + || returned.accepted_at_ms() != original.accepted_at_ms() + || returned.established_evidence() != original.established_evidence() + || returned.updated_at_ms() < original.updated_at_ms() + || returned.settlement_evidence() != Some(evidence) + { + return Err(Error::Fenced); + } + Ok(()) +} +pub(super) fn closure_digest( + boot: &EnrollmentRecord, + process: &FleetFailedBootProcessEvidence, +) -> Result { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-failed-boot-closure.v1\0"); + hash.update(&boot.to_bytes().map_err(operation)?); + hash.update(retirement(process).as_bytes()); + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) +} diff --git a/crates/cellule-host/src/fleet/failed_boot/retained.rs b/crates/cellule-host/src/fleet/failed_boot/retained.rs new file mode 100644 index 00000000..74a40981 --- /dev/null +++ b/crates/cellule-host/src/fleet/failed_boot/retained.rs @@ -0,0 +1,90 @@ +use super::*; + +impl FleetFailedBootRetirement { + /// Reuses the immutable original process request after recovery/role closure. + /// + /// A fenced request keeps its v2 identity and original collection interval; + /// a terminal request keeps its v1 identity. This fresh capture still requires + /// a canonically Retired leader log, every related role settled, complete + /// physical references and the same original boot/fence. Publication rechecks + /// the provider and terminal authority independently. Neither an old process + /// confirmation nor this capture can complete affected-Cell relocation. + #[allow(clippy::too_many_arguments)] + pub async fn capture_retained( + journal: &dyn FleetJournal, + directory: &NodeDirectory, + roster: &FleetRoster, + request: &FleetFailedBootProcessRequest, + claimant: SessionId, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + let started_at_ms = clock()?; + if started_at_ms < request.finished_at_ms { + return Err(Error::Deadline); + } + interval(started_at_ms, started_at_ms)?; + let mut last = started_at_ms; + let mut clock = || { + let next = clock()?; + if next < last { + return Err(Error::Deadline); + } + interval(started_at_ms, next)?; + last = next; + Ok(next) + }; + if roster.snapshot().registry().bootstrap_revision().is_none() + || roster.snapshot().head().scope() != request.snapshot.head().scope() + || directory.fleet() != roster.snapshot().head().scope().fleet + { + return Err(Error::Fenced); + } + roster.confirm(journal, deadline).await?; + let boot = records::select(roster, &request.boot)?; + let endpoint = boot.spec().target; + request + .confirm_canonical(directory, claimant, clock()?, deadline) + .await?; + let canonical = bounded( + deadline, + directory.closed_session(endpoint.node, endpoint.session, claimant, clock()?), + ) + .await?; + if canonical.fence() != request.fence { + return Err(Error::Fenced); + } + records::references( + directory, + roster, + endpoint.node, + endpoint.session, + deadline, + &mut clock, + ) + .await?; + roster.confirm(journal, deadline).await?; + request + .confirm_canonical(directory, claimant, clock()?, deadline) + .await?; + // Foreign reference collection can suspend. Terminal original authority + // must agree after it, as well as before it. + if bounded( + deadline, + directory.closed_session(endpoint.node, endpoint.session, claimant, clock()?), + ) + .await? + != canonical + { + return Err(Error::Fenced); + } + let finished_at_ms = clock()?; + Ok(Self { + snapshot: roster.snapshot().clone(), + request: request.clone(), + canonical, + started_at_ms, + finished_at_ms, + }) + } +} diff --git a/crates/cellule-host/src/fleet/failed_boot/writers/capture.rs b/crates/cellule-host/src/fleet/failed_boot/writers/capture.rs new file mode 100644 index 00000000..86497d0a --- /dev/null +++ b/crates/cellule-host/src/fleet/failed_boot/writers/capture.rs @@ -0,0 +1,324 @@ +use super::*; +use cellule_runtime::{ + fleet::operations::{ + MAX_ORIGINAL_WRITERS, MaintenancePhase, OriginalCatalogWitness, + OriginalWriterInventoryBasis, OriginalWriterObservation, + }, + identity::CellTarget, + node::NodeMode, +}; + +impl FleetOriginalWriterCapture { + /// Joins the original process through its existing provider, traverses every + /// authenticated catalog, retains all original ownership epochs, then rechecks + /// source configuration, process/fence, catalogs and the complete journal. + /// Missing legacy/restore owner history is an error, never an empty set. + #[allow(clippy::too_many_arguments)] + pub async fn capture( + journal: &dyn FleetOriginalWriterJournal, + directory: &NodeDirectory, + processes: &dyn FleetFailedBootProcesses, + catalogs: &dyn FleetOriginalCatalogs, + request: &FleetFailedBootProcessRequest, + claimant: SessionId, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + let started = clock()?; + let mut last = started; + let mut now = || { + let next = clock()?; + interval(started, next)?; + if next < last { + return Err(Error::Deadline); + } + last = next; + Ok(next) + }; + bounded(deadline, async { + let joined = request + .confirm(journal, directory, processes, claimant, deadline, &mut now) + .await?; + let snapshot = joined.snapshot().clone(); + let operation = snapshot.head().maintenance().ok_or(Error::Fenced)?.clone(); + let roster = FleetRoster::collect(journal, &snapshot, deadline).await?; + check_operation(&snapshot, &roster, request, now()?)?; + let boot = records::boot(&roster, request.boot())?; + let sources = catalogs + .catalogs(request, &operation, &snapshot) + .await + .map_err(adapter_error)?; + if sources.request != request.digest() || sources.operation != operation { + return Err(Error::Fenced); + } + let mut owners = Vec::new(); + let mut witnesses = Vec::new(); + let mut receipts = Vec::new(); + let mut cell_count = 0; + let mut history_count = 0; + for source in &sources.sources { + let catalog = CellCatalog::new(source.layout.clone(), source.tenant); + let authority = CellAuthority::new(source.layout.clone()); + let mut scan = catalog + .scan_all((MAX_ORIGINAL_WRITERS - cell_count).max(1)) + .await?; + let before = owners.len(); + let mut histories = blake3::Hasher::new(); + histories.update(b"cellule.original-writer-catalog-history.v1\0"); + while let Some(page) = scan.next_page().await? { + if page.entries().len() > MAX_ORIGINAL_WRITERS - cell_count { + return Err(Error::Capacity( + "complete original catalogs exceed Cell bound", + )); + } + for proof in page.entries() { + now()?; + cell_count += 1; + let entry = proof.entry(); + let target = CellTarget::new( + source.tenant, + source.application(), + entry.namespace(), + entry.partition(), + )?; + histories.update(target.cell_id().as_bytes()); + if authority.load(entry.cell()).await?.is_none() { + // Provisioning can commit a catalog entry before its + // first Control. Original accepted metadata work is + // already joined; absence is retained explicitly. + histories.update(&[0]); + continue; + } + histories.update(&[1]); + let history = authority + .owner_history( + entry.cell(), + (MAX_ORIGINAL_WRITERS - history_count).max(1), + ) + .await?; + if history.owners().len() > MAX_ORIGINAL_WRITERS - history_count { + return Err(Error::Capacity( + "complete original owner histories exceed bound", + )); + } + history_count += history.owners().len(); + hash_control(&mut histories, history.current())?; + histories.update(&(history.owners().len() as u64).to_be_bytes()); + for control in history.owners() { + hash_control(&mut histories, control)?; + if control + .owner + .as_ref() + .is_some_and(|owner| owner.session == request.fence().session()) + { + owners.push(OriginalWriterObservation { + target: target.clone(), + control: control.clone(), + }); + } + } + } + } + let receipt = scan.finish().await?; + let mut heads = blake3::Hasher::new(); + heads.update(b"cellule.original-writer-catalog-heads.v1\0"); + heads.update(source.application().as_bytes()); + heads.update(source.tenant.as_bytes()); + for shard in 0..=u8::MAX { + heads.update(&[shard]); + heads.update(&receipt.revision(shard).to_be_bytes()); + let pages = receipt.page_digests(shard); + heads.update(&(pages.len() as u64).to_be_bytes()); + for page in pages { + heads.update(page.as_bytes()); + } + } + witnesses.push(OriginalCatalogWitness { + application: source.application(), + tenant: source.tenant, + source: source.identity, + heads: Digest::from_bytes(*heads.finalize().as_bytes()), + histories: Digest::from_bytes(*histories.finalize().as_bytes()), + cells: receipt.entry_count() as u64, + owners: (owners.len() - before) as u64, + }); + receipts.push(receipt); + } + if !sources.matches( + &catalogs + .catalogs(request, &operation, &snapshot) + .await + .map_err(adapter_error)?, + ) { + return Err(Error::Control("original catalog configuration changed")); + } + for receipt in &receipts { + now()?; + receipt.revalidate().await?; + } + let final_join = request + .confirm(journal, directory, processes, claimant, deadline, &mut now) + .await?; + if final_join.snapshot() != &snapshot || final_join.process() != joined.process() { + return Err(Error::Fenced); + } + roster.confirm(journal, deadline).await?; + let finished = now()?; + let (record, pages) = OriginalWriterInventoryRecord::new( + OriginalWriterInventoryBasis { + operation, + head_digest: Digest::from_bytes( + *blake3::hash(&snapshot.head().to_bytes().map_err(operation_error)?) + .as_bytes(), + ), + registry: snapshot.registry(), + boot, + process_request: request.digest(), + process_witness: joined.process().witness(), + catalog_witness: sources.witness, + interval: (started, finished), + }, + witnesses, + owners, + ) + .map_err(operation_error)?; + Ok(Self { + snapshot, + request: request.clone(), + process: joined.process().clone(), + sources, + catalogs: receipts, + record, + pages, + }) + }) + .await + } + + /// Confirms the original basis before first publication in the existing + /// accepted journal owner. A committed exact replay returns original history + /// without repeating provider reads or refreshing times. Success proves + /// retention only; every current successor still requires separate evidence. + #[allow(clippy::too_many_arguments)] + pub async fn publish( + &self, + journal: &dyn FleetOriginalWriterJournal, + directory: &NodeDirectory, + processes: &dyn FleetFailedBootProcesses, + catalogs: &dyn FleetOriginalCatalogs, + claimant: SessionId, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + bounded(deadline, async { + let current = journal + .load_snapshot(self.snapshot.head().scope()) + .await + .map_err(adapter_error)?; + if let Some(original) = FleetOriginalWriterInventory::load( + journal, + ¤t, + self.record.basis().operation.id(), + self.request.digest(), + deadline, + ) + .await? + { + if original.record() != &self.record || original.pages() != self.pages.as_slice() { + return Err(Error::Fenced); + } + return Ok(original.record().clone()); + } + if current != self.snapshot { + return Err(Error::Fenced); + } + let start = self.record.basis().interval.0; + let mut last = self.record.basis().interval.1; + let mut now = || { + let next = clock()?; + interval(start, next)?; + if next < last { + return Err(Error::Deadline); + } + last = next; + Ok(next) + }; + let joined = self + .request + .confirm(journal, directory, processes, claimant, deadline, &mut now) + .await?; + if joined.snapshot() != &self.snapshot || joined.process() != &self.process { + return Err(Error::Fenced); + } + let source_set = catalogs + .catalogs( + &self.request, + &self.record.basis().operation, + &self.snapshot, + ) + .await + .map_err(adapter_error)?; + if !self.sources.matches(&source_set) { + return Err(Error::Fenced); + } + for receipt in &self.catalogs { + now()?; + receipt.revalidate().await?; + } + // Provider/configuration I/O cannot hide a registry or claimant change. + let roster = FleetRoster::collect(journal, &self.snapshot, deadline).await?; + check_operation(&self.snapshot, &roster, &self.request, now()?)?; + self.request + .confirm_canonical(directory, claimant, now()?, deadline) + .await?; + if self.request.confirm_process(processes, deadline).await? != self.process { + return Err(Error::Fenced); + } + roster.confirm(journal, deadline).await?; + self.request + .confirm_canonical(directory, claimant, now()?, deadline) + .await?; + journal + .persist_original_writers(&self.snapshot, &self.record, &self.pages, now()?) + .await + .map_err(adapter_error) + }) + .await + } +} +fn hash_control( + hash: &mut blake3::Hasher, + control: &cellule_runtime::control::Control, +) -> Result<()> { + let bytes = control.encode()?; + hash.update(&(bytes.len() as u64).to_be_bytes()); + hash.update(&bytes); + Ok(()) +} +fn check_operation( + snapshot: &FleetJournalSnapshot, + roster: &FleetRoster, + request: &FleetFailedBootProcessRequest, + now: i64, +) -> Result<()> { + let current = snapshot.head().maintenance().ok_or(Error::Fenced)?; + let lease = snapshot.head().controller().ok_or(Error::Fenced)?; + let intent = roster + .intents() + .iter() + .find(|row| row.node() == current.node()) + .ok_or(Error::Fenced)?; + records::boot(roster, request.boot())?; + if current.phase() == MaintenancePhase::Completed + || current.node() != request.fence().node() + || current.session() != request.fence().session() + || intent.session() != current.session() + || intent.revision() != current.intent_revision() + || intent.mode() != NodeMode::Draining + || now >= lease.expires_at_ms + || now >= current.deadline_ms() + { + return Err(Error::Fenced); + } + Ok(()) +} diff --git a/crates/cellule-host/src/fleet/failed_boot/writers/inventory.rs b/crates/cellule-host/src/fleet/failed_boot/writers/inventory.rs new file mode 100644 index 00000000..c858adfa --- /dev/null +++ b/crates/cellule-host/src/fleet/failed_boot/writers/inventory.rs @@ -0,0 +1,84 @@ +//! Complete reconstruction of immutable original inputs after controller restart. +use super::*; +use cellule_runtime::fleet::operations::OriginalWriterObservation; + +/// Complete committed original writer set reconstructed through its canonical +/// journal pointer and every exact page. Missing commitment is distinct from an +/// explicitly authenticated empty set; missing pages never yield a partial set. +/// +/// This retains original scopes, epochs, controls and times. It is historical +/// input, not current successor serving, a root pin or maintenance settlement. +/// The application authenticates the journal, accounts bounded collector memory +/// and owns accepted adapter work through the enclosing finite deadline. +pub struct FleetOriginalWriterInventory { + record: OriginalWriterInventoryRecord, + pages: Vec, +} + +impl FleetOriginalWriterInventory { + /// Loads the immutable original set at the supplied complete journal read + /// barrier. The canonical record bounds every page and owner before this + /// value escapes; wrong scope, key, page identity/order or counts refuse. + /// `None` means no committed set at that read, never zero original writers + /// or proof that an accepted capture cannot still publish. + pub async fn load( + journal: &dyn FleetOriginalWriterJournal, + expected: &FleetJournalSnapshot, + operation: OperationId, + process_request: Digest, + deadline: Instant, + ) -> Result> { + if operation.as_bytes() == &[0; 16] || process_request.as_bytes() == &[0; 32] { + return Err(Error::Control("invalid original writer inventory key")); + } + bounded(deadline, async { + let Some(record) = journal + .original_writers(expected, operation, process_request) + .await + .map_err(adapter_error)? + else { + return Ok(None); + }; + if record.basis().registry.scope() != expected.head().scope() + || record.basis().operation.id() != operation + || record.basis().process_request != process_request + { + return Err(Error::Fenced); + } + // Validate the bounded header before reserving the page vector. + // Pages are immutable history; they cannot refresh this read's + // barrier or supply current native-role evidence. + record.digest().map_err(operation_error)?; + let mut pages = Vec::with_capacity(record.pages().len()); + for digest in record.pages() { + pages.push( + journal + .original_writer_page(expected.head().scope(), *digest) + .await + .map_err(adapter_error)? + .ok_or(Error::Fenced)?, + ); + } + record.validate_pages(&pages).map_err(operation_error)?; + Ok(Some(Self { record, pages })) + }) + .await + } + + /// Original immutable manifest, including complete source witnesses. + #[must_use] + pub const fn record(&self) -> &OriginalWriterInventoryRecord { + &self.record + } + + /// Every exact original page in canonical order. + #[must_use] + pub fn pages(&self) -> &[OriginalWriterInventoryPage] { + &self.pages + } + + /// All original owner epochs across the fully validated page set. + pub fn writers(&self) -> impl Iterator { + self.pages.iter().flat_map(|page| page.entries().iter()) + } +} diff --git a/crates/cellule-host/src/fleet/failed_boot/writers/mod.rs b/crates/cellule-host/src/fleet/failed_boot/writers/mod.rs new file mode 100644 index 00000000..0b978408 --- /dev/null +++ b/crates/cellule-host/src/fleet/failed_boot/writers/mod.rs @@ -0,0 +1,178 @@ +//! Original Cell capture and persistence through the existing fleet journal. +use super::*; +use cellule_runtime::{ + cell::catalog::{CatalogScanReceipt, CellCatalog}, + control::authority::CellAuthority, + fleet::operations::{ + FleetScope, MaintenanceOperation, OperationId, OriginalWriterInventoryPage, + OriginalWriterInventoryRecord, + }, + identity::{ApplicationId, TenantId}, + ltx::CellStorageLayout, +}; + +mod capture; +mod inventory; +pub use inventory::FleetOriginalWriterInventory; + +/// One application-authenticated canonical catalog source. Construct from the +/// same storage layout for catalog and authority; never accept remote credentials. +#[derive(Clone)] +pub struct FleetOriginalCatalogSource { + identity: Digest, + tenant: TenantId, + layout: CellStorageLayout, +} +impl FleetOriginalCatalogSource { + /// Shape validation only. The provider attests canonical source identity. + pub fn new(identity: Digest, tenant: TenantId, layout: CellStorageLayout) -> Result { + if identity.as_bytes() == &[0; 32] { + return Err(Error::Control("invalid original catalog source")); + } + Ok(Self { + identity, + tenant, + layout, + }) + } + /// Original trusted canonical source identity. + #[must_use] + pub const fn identity(&self) -> Digest { + self.identity + } + /// Original tenant scope. + #[must_use] + pub const fn tenant(&self) -> TenantId { + self.tenant + } + /// Original application scope. + #[must_use] + pub fn application(&self) -> ApplicationId { + ApplicationId::from_bytes(*self.layout.application_id()) + } +} + +/// Application attestation of every canonical catalog the original physical +/// boot could write, across every application/tenant. Empty is an explicit +/// authenticated no-writer configuration, never inferred from missing storage. +pub struct FleetOriginalCatalogSet { + request: Digest, + operation: MaintenanceOperation, + witness: Digest, + sources: Vec, +} +impl FleetOriginalCatalogSet { + /// Binds a bounded complete source set to the exact original request and + /// operation. Providers retain the source-set witness across reconstruction. + pub fn new( + request: &FleetFailedBootProcessRequest, + operation: MaintenanceOperation, + witness: Digest, + mut sources: Vec, + ) -> Result { + if witness.as_bytes() == &[0; 32] + || sources.len() > cellule_runtime::fleet::operations::MAX_ORIGINAL_CATALOGS + || operation.node() != request.boot().spec().target.node + || operation.session() != request.boot().spec().target.session + { + return Err(Error::Fenced); + } + operation.to_bytes().map_err(operation_error)?; + sources.sort_by_key(|source| (*source.application().as_bytes(), *source.tenant.as_bytes())); + if sources.windows(2).any(|pair| { + (pair[0].application(), pair[0].tenant) == (pair[1].application(), pair[1].tenant) + }) { + return Err(Error::Control("original catalog scopes are duplicated")); + } + Ok(Self { + request: request.digest(), + operation, + witness, + sources, + }) + } + fn matches(&self, other: &Self) -> bool { + self.request == other.request + && self.operation == other.operation + && self.witness == other.witness + && self.sources.len() == other.sources.len() + && self.sources.iter().zip(&other.sources).all(|(a, b)| { + a.identity == b.identity + && a.tenant == b.tenant + && a.application() == b.application() + && a.layout.application_prefix() == b.layout.application_prefix() + }) + } +} + +/// Read-only application provider for original boot's complete authorized +/// catalog set. Authenticate every original physical/session scope and canonical +/// storage mapping; exclude omitted applications/tenants and session reuse. A +/// filtered resident list, current-owner scan or successful recovery is insufficient. +/// Reads must return the same durable witness and canonical mappings. Accepted +/// provider work uses the application's existing finite owner; no effect starts here. +pub trait FleetOriginalCatalogs: Send + Sync { + /// Confirms complete original configuration under the full operation barrier. + fn catalogs<'a>( + &'a self, + request: &'a FleetFailedBootProcessRequest, + operation: &'a MaintenanceOperation, + expected: &'a FleetJournalSnapshot, + ) -> FleetAdapterFuture<'a, FleetOriginalCatalogSet>; +} + +/// Immutable original set in the same registry transaction domain as all other +/// fleet work. First publication advances the registry atomically. Identical +/// replay returns original bytes/times; a second different original set conflicts. +pub trait FleetOriginalWriterJournal: FleetJournal { + /// Checks full barrier, bootstrap, live controller/current operation/intent + /// and original boot row, then commits every page plus its one original-set + /// pointer atomically. Orphan pages cannot establish a retained inventory. + fn persist_original_writers<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + record: &'a OriginalWriterInventoryRecord, + pages: &'a [OriginalWriterInventoryPage], + now_ms: i64, + ) -> FleetAdapterFuture<'a, OriginalWriterInventoryRecord>; + /// Reads the committed original set at a complete current barrier. + fn original_writers<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + operation: OperationId, + process_request: Digest, + ) -> FleetAdapterFuture<'a, Option>; + /// Reads one digest-verified historical page; validate all manifest pages. + fn original_writer_page( + &self, + scope: FleetScope, + digest: Digest, + ) -> FleetAdapterFuture<'_, Option>; +} + +/// Opaque fully collected original set. Persist it before dependent effects. +/// Collection alone grants no relocation, successor readiness or role settlement. +pub struct FleetOriginalWriterCapture { + snapshot: FleetJournalSnapshot, + request: FleetFailedBootProcessRequest, + process: FleetFailedBootProcessEvidence, + sources: FleetOriginalCatalogSet, + catalogs: Vec, + record: OriginalWriterInventoryRecord, + pages: Vec, +} +impl FleetOriginalWriterCapture { + /// Complete immutable original manifest, without publication rights. + #[must_use] + pub const fn record(&self) -> &OriginalWriterInventoryRecord { + &self.record + } + /// Complete immutable original pages. + #[must_use] + pub fn pages(&self) -> &[OriginalWriterInventoryPage] { + &self.pages + } +} +fn operation_error(source: cellule_runtime::fleet::operations::OperationError) -> Error { + operation(source) +} diff --git a/crates/cellule-host/src/fleet/follower_evacuation/mod.rs b/crates/cellule-host/src/fleet/follower_evacuation/mod.rs new file mode 100644 index 00000000..fa1ffb7d --- /dev/null +++ b/crates/cellule-host/src/fleet/follower_evacuation/mod.rs @@ -0,0 +1,58 @@ +//! Durable follower replacement history in the existing fleet journal domain. +use super::*; +use cellule_runtime::{ + fleet::operations::{FollowerEvacuationRecord, FollowerReplacementPolicy, OperationId}, + identity::Digest, +}; + +mod native; +mod publication; +mod refresh; +mod verification; +pub use publication::FleetFollowerEvacuationPublication; +pub use refresh::FleetFollowerEvacuationCandidate; +pub use verification::{FleetFollowerEvacuationCheck, FleetFollowerEvacuationVerifier}; + +/// Immutable follower captures and authoritative policy share the fleet registry. +/// Each mutation must use the same transaction domain and accepted backend owner +/// as enrollment/actions. Applications authorize policy changes and account buffers. +pub trait FleetFollowerEvacuationJournal: FleetJournal { + /// Loads current application redundancy policy at the complete expected barrier. + /// Absence is an explicit blocker; it never means zero required redundancy. + fn follower_replacement_policy<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + ) -> FleetAdapterFuture<'a, Option>; + /// Atomically compares the complete barrier and live controller, then stores + /// policy revision one or exactly the current revision plus one. Advance the + /// registry in the same transaction; never silently lower application policy. + fn set_follower_replacement_policy<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + policy: FollowerReplacementPolicy, + now_ms: i64, + ) -> FleetAdapterFuture<'a, FollowerReplacementPolicy>; + /// Commits complete original/replacement rows and the latest request pointer + /// after full barrier/current operation/controller/policy and intent checks. + /// Exact historical replay returns original bytes without advancing the + /// registry or restoring a superseded pointer; original times never refresh. + fn persist_follower_evacuation<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + record: &'a FollowerEvacuationRecord, + now_ms: i64, + ) -> FleetAdapterFuture<'a, FollowerEvacuationRecord>; + /// Loads immutable exact history; validate its requested scope and full digest. + fn load_follower_evacuation( + &self, + scope: cellule_runtime::fleet::operations::FleetScope, + digest: Digest, + ) -> FleetAdapterFuture<'_, Option>; + /// Reads the current pointer at the full barrier. Orphan PUTs grant no evidence. + fn latest_follower_evacuation<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + operation: OperationId, + original: Digest, + ) -> FleetAdapterFuture<'a, Option>; +} diff --git a/crates/cellule-host/src/fleet/follower_evacuation/native.rs b/crates/cellule-host/src/fleet/follower_evacuation/native.rs new file mode 100644 index 00000000..401dcd19 --- /dev/null +++ b/crates/cellule-host/src/fleet/follower_evacuation/native.rs @@ -0,0 +1,188 @@ +//! Fresh opaque native inventories through the canonical snapshot task owner. +use super::*; +use crate::NodeDurabilitySupervisorState; +use cellule_runtime::{ + Error, Result, + identity::{NodeId, SessionId}, + node::NodeMode, +}; +use tokio::time::Instant; + +impl FleetFollowerEvacuationVerifier { + pub(super) async fn collect_native( + &self, + roster: &FleetRoster, + record: &FollowerEvacuationRecord, + deadline: Instant, + clock: &mut impl FnMut() -> Result, + ) -> Result> { + let source = record.source().map_err(operation)?; + let mut inventories = Vec::with_capacity(3); + inventories.push( + self.inventory(roster, source.node, source.session, deadline, clock) + .await?, + ); + let leader = &inventories[0]; + if leader.mode() != NodeMode::Active + || !leader.bindings().follower_producer + || !leader.bindings().durability_supervisor + || leader.node_log() != Some((source.session, source.node, record.replacement_epoch())) + || leader + .follower_producer_state() + .is_none_or(|(busy, draining, pending)| busy || draining || pending.is_some()) + || leader.supervisor().is_none_or(|supervisor| { + supervisor.state != NodeDurabilitySupervisorState::Running + || supervisor.cancellation_requested + || supervisor.supervisor_error.is_some() + || supervisor.requests_error.is_some() + }) + { + return Err(Error::Fenced); + } + leader.validate_enrollments(roster)?; + let progress = leader + .follower_enrollments() + .iter() + .find(|progress| progress.epoch == record.replacement_epoch()) + .ok_or(Error::Fenced)?; + if !progress.delivered + || !progress.native_started + || progress.native_closed + || progress.refusal.is_some() + || progress.enrollment.is_none() + || progress.members.len() != record.replacements().len() + || progress + .members + .iter() + .zip(record.replacements()) + .any(|(member, entry)| { + member.spec != *entry.enrollment.spec() + || !member.published + || member.accepted.as_ref().is_none_or(|accepted| { + accepted.accepted_at_ms() != entry.enrollment.accepted_at_ms() + }) + }) + { + return Err(Error::Fenced); + } + for entry in record.replacements() { + let endpoint = entry.enrollment.spec().target; + let receiver = self + .inventory(roster, endpoint.node, endpoint.session, deadline, clock) + .await?; + if receiver.mode() != NodeMode::Active + || !receiver.bindings().follower_store + || receiver + .follower_store_state() + .is_none_or(|(_, quarantined)| quarantined != 0) + { + return Err(Error::Fenced); + } + // The original native producer and current canonical complete ensemble + // account for a member before its first persisted append lane exists. + receiver.validate_enrollments_with(roster, |row| { + record + .replacements() + .iter() + .any(|entry| &entry.enrollment == row) + })?; + inventories.push(receiver); + } + Ok(inventories) + } + async fn inventory( + &self, + roster: &FleetRoster, + node: NodeId, + session: SessionId, + deadline: Instant, + clock: &mut impl FnMut() -> Result, + ) -> Result { + let mut scan = FleetNodeInventoryScan::new(roster, node, session)?; + while let Some(subject) = scan.next_subject()? { + let request = self.request(roster, node, session, subject, deadline, clock)?; + let page = self + .transport + .capture(&request, deadline) + .await + .map_err(|source| Error::Facility { + name: "fleet-native-snapshot-transport", + source, + })?; + scan.accept(&request, &page, clock()?)?; + } + scan.finish() + } + pub(super) async fn recheck_native( + &self, + roster: &FleetRoster, + inventories: &mut [FleetNodeInventory], + deadline: Instant, + clock: &mut impl FnMut() -> Result, + ) -> Result<()> { + for inventory in inventories { + let node = inventory.node(); + let session = inventory.session(); + let mut recheck = inventory.recheck(); + while let Some(subject) = recheck.next_subject()? { + let request = self.request(roster, node, session, subject, deadline, clock)?; + let page = self + .transport + .capture(&request, deadline) + .await + .map_err(|source| Error::Facility { + name: "fleet-native-snapshot-transport", + source, + })?; + recheck.accept(&request, &page, clock()?)?; + } + recheck.finish()?; + } + Ok(()) + } + fn request( + &self, + roster: &FleetRoster, + node: NodeId, + session: SessionId, + subject: FleetSnapshotSubject, + deadline: Instant, + clock: &mut impl FnMut() -> Result, + ) -> Result { + let now = clock()?; + let remaining = deadline + .checked_duration_since(Instant::now()) + .ok_or(Error::Deadline)?; + let millis = i64::try_from(remaining.as_millis()).map_err(|_| Error::Deadline)?; + let end = now + .checked_add(millis.min(30_000)) + .ok_or(Error::Deadline)? + .min( + roster + .snapshot() + .head() + .controller() + .ok_or(Error::Fenced)? + .expires_at_ms, + ); + let mut nonce = blake3::Hasher::new(); + nonce.update(b"cellule.follower-policy-native.v1\0"); + nonce.update(uuid::Uuid::now_v7().as_bytes()); + let limit = if matches!(subject, FleetSnapshotSubject::FollowerEnrollments(_)) { + 32 + } else { + 128 + }; + FleetSnapshotRequest::new( + roster.snapshot().clone(), + Digest::from_bytes(*nonce.finalize().as_bytes()), + node, + session, + subject, + limit, + now, + end, + ) + .map_err(operation) + } +} diff --git a/crates/cellule-host/src/fleet/follower_evacuation/publication.rs b/crates/cellule-host/src/fleet/follower_evacuation/publication.rs new file mode 100644 index 00000000..5c175554 --- /dev/null +++ b/crates/cellule-host/src/fleet/follower_evacuation/publication.rs @@ -0,0 +1,127 @@ +use super::verification::{adapter, bounded, interval}; +use super::*; +use crate::FollowerEvacuation; +use cellule_runtime::{Error, Result}; +use std::sync::Arc; +use tokio::time::Instant; + +/// Original durable response and independent fresh confirmation, including errors. +pub struct FleetFollowerEvacuationPublication { + record: std::result::Result>, + check: std::result::Result>, +} +impl FleetFollowerEvacuationPublication { + /// Publishes the native rotation's exact full ensemble under current policy. + /// Accepted backend work remains owned after cancellation or ambiguous replies. + pub async fn publish( + capture: &FollowerEvacuation, + policy: FollowerReplacementPolicy, + journal: &dyn FleetFollowerEvacuationJournal, + verifier: &FleetFollowerEvacuationVerifier, + deadline: Instant, + clock: impl FnMut() -> Result, + ) -> Result { + let record = capture.durable_record(policy)?; + Self::publish_record( + capture.snapshot(), + &record, + journal, + verifier, + deadline, + clock, + ) + .await + } + /// Commits fresh policy/ensemble metadata without repeating original retirement. + pub async fn publish_refreshed( + candidate: &FleetFollowerEvacuationCandidate, + journal: &dyn FleetFollowerEvacuationJournal, + verifier: &FleetFollowerEvacuationVerifier, + deadline: Instant, + clock: impl FnMut() -> Result, + ) -> Result { + Self::publish_record( + candidate.snapshot(), + candidate.record(), + journal, + verifier, + deadline, + clock, + ) + .await + } + async fn publish_record( + snapshot: &FleetJournalSnapshot, + record: &FollowerEvacuationRecord, + journal: &dyn FleetFollowerEvacuationJournal, + verifier: &FleetFollowerEvacuationVerifier, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + if let Some(original) = bounded(deadline, async { + journal + .load_follower_evacuation( + record.policy().scope(), + record.digest().map_err(operation)?, + ) + .await + .map_err(adapter) + }) + .await? + { + if &original != record { + return Err(Error::Fenced); + } + let check = verifier + .recheck(journal, &original, deadline, &mut clock) + .await + .map_err(Arc::new); + return Ok(Self { + record: Ok(original), + check, + }); + } + let (started, mut last) = record.interval(); + let mut clock = || { + let now = clock()?; + interval(started, last, now)?; + last = now; + Ok(now) + }; + verifier + .candidate(journal, record, snapshot, deadline, &mut clock) + .await?; + let now = clock()?; + let returned = bounded(deadline, async { + let returned = journal + .persist_follower_evacuation(snapshot, record, now) + .await + .map_err(adapter)?; + if &returned != record { + return Err(Error::Fenced); + } + Ok(returned) + }) + .await + .map_err(Arc::new); + let check = match &returned { + Ok(record) => verifier + .recheck(journal, record, deadline, &mut clock) + .await + .map_err(Arc::new), + Err(error) => Err(Arc::clone(error)), + }; + Ok(Self { + record: returned, + check, + }) + } + /// Original returned history or unchanged shared source failure. + pub fn record(&self) -> std::result::Result<&FollowerEvacuationRecord, Arc> { + self.record.as_ref().map_err(Arc::clone) + } + /// Independent current confirmation; historical success alone cannot settle a role. + pub fn confirmed(&self) -> std::result::Result<&FleetFollowerEvacuationCheck, Arc> { + self.check.as_ref().map_err(Arc::clone) + } +} diff --git a/crates/cellule-host/src/fleet/follower_evacuation/refresh.rs b/crates/cellule-host/src/fleet/follower_evacuation/refresh.rs new file mode 100644 index 00000000..601e276e --- /dev/null +++ b/crates/cellule-host/src/fleet/follower_evacuation/refresh.rs @@ -0,0 +1,177 @@ +use super::verification::{adapter, bounded, interval}; +use super::*; +use crate::durability::enrollment::maintenance::boot_identity; +use cellule_runtime::{ + Error, Result, + fleet::operations::{ + EnrollmentRole, EnrollmentStatus, FollowerReplacementWitness, MaintenancePhase, + }, + node::log_state::NodeLogPhase, +}; +use tokio::time::Instant; + +/// Fresh current ensemble/policy capture retaining the original native retirement. +pub struct FleetFollowerEvacuationCandidate { + snapshot: FleetJournalSnapshot, + record: FollowerEvacuationRecord, +} +impl FleetFollowerEvacuationCandidate { + /// Complete original barrier used by this fresh candidate. + #[must_use] + pub fn snapshot(&self) -> &FleetJournalSnapshot { + &self.snapshot + } + /// New immutable metadata; all original retired rows/timestamps stay unchanged. + #[must_use] + pub fn record(&self) -> &FollowerEvacuationRecord { + &self.record + } +} +impl FleetFollowerEvacuationVerifier { + /// Refreshes committed live-owner history after policy, ensemble or operation + /// adoption. Ordinary recruitment supplies current replacements. This starts + /// no rotation, native retirement, recovery or producer work. A failed original + /// leader still requires canonical recovery and separate affected-Cell evidence. + pub async fn refresh( + &self, + journal: &dyn FleetFollowerEvacuationJournal, + original: &FollowerEvacuationRecord, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + let started = clock()?; + let mut last = started; + let mut clock = || { + let now = clock()?; + interval(started, last, now)?; + last = now; + Ok(now) + }; + bounded(deadline, async { + let scope = original.policy().scope(); + if journal + .load_follower_evacuation(scope, original.digest().map_err(operation)?) + .await + .map_err(adapter)? + .as_ref() + != Some(original) + { + return Err(Error::Fenced); + } + let snapshot = journal.load_snapshot(scope).await.map_err(adapter)?; + let maintenance = snapshot.head().maintenance().ok_or(Error::Fenced)?; + if maintenance.id() != original.operation().id() + || maintenance.node() != original.operation().node() + || !matches!( + maintenance.phase(), + MaintenancePhase::Evacuating | MaintenancePhase::Closing + ) + { + return Err(Error::Fenced); + } + let policy = journal + .follower_replacement_policy(&snapshot) + .await + .map_err(adapter)? + .ok_or(Error::Fenced)?; + let roster = FleetRoster::collect(journal, &snapshot, deadline).await?; + for row in original.retired() { + if !roster.enrollments().iter().any(|current| current == row) { + return Err(Error::Fenced); + } + } + let source = original.source().map_err(operation)?; + let leader = self + .directory + .load_if_live(source.session, clock()?) + .await? + .ok_or(Error::Fenced)?; + let leader = leader.advertisement(); + if leader.node() != source.node || boot_identity(leader)? != original.source_boot() { + return Err(Error::Fenced); + } + let original_epoch = original.original_epoch().map_err(operation)?; + let log = leader + .log() + .filter(|log| log.phase() == NodeLogPhase::Open && log.epoch() > original_epoch) + .ok_or(Error::Fenced)?; + let mut replacements = Vec::with_capacity(log.members().len()); + for member in log.members() { + let mut rows = roster.enrollments().iter().filter(|row| { + row.status() == EnrollmentStatus::Established + && row.spec().source.is_some_and(|endpoint| { + endpoint.node == source.node && endpoint.session == source.session + }) + && row.spec().target.node == *member + && row.spec().role + == (EnrollmentRole::Follower { + log_epoch: log.epoch(), + }) + }); + let row = rows.next().ok_or(Error::Fenced)?; + if rows.next().is_some() { + return Err(Error::Fenced); + } + let boot = self + .directory + .load_if_live(row.spec().target.session, clock()?) + .await? + .ok_or(Error::Fenced)?; + replacements.push(FollowerReplacementWitness { + enrollment: row.clone(), + boot_identity: boot_identity(boot.advertisement())?, + }); + } + // This is a fresh canonical observation identity, distinct from the + // original enrollment proof and from native retirement history. + let mut evidence = blake3::Hasher::new(); + evidence.update(b"cellule.follower-maintenance-observation.v1\0"); + evidence.update(boot_identity(leader)?.as_bytes()); + evidence.update(&leader.generation().to_le_bytes()); + evidence.update(&leader.issued_at_ms().to_le_bytes()); + evidence.update(&leader.expires_at_ms().to_le_bytes()); + evidence.update(&log.epoch().to_le_bytes()); + evidence.update(&log.tiered_through().to_le_bytes()); + for member in log.members() { + evidence.update(member.as_bytes()); + } + let evidence = Digest::from_bytes(*evidence.finalize().as_bytes()); + let record = FollowerEvacuationRecord::new( + maintenance.clone(), + ( + Digest::from_bytes( + *blake3::hash(&snapshot.head().to_bytes().map_err(operation)?).as_bytes(), + ), + snapshot.registry(), + ), + policy, + (original.original_key(), original.original_digest()), + original.retired().to_vec(), + original.covered_through(), + original.source_boot(), + (log.epoch(), evidence), + replacements, + (started, clock()?), + ) + .map_err(operation)?; + self.candidate(journal, &record, &snapshot, deadline, &mut clock) + .await?; + // Include the complete revalidation interval without changing native history. + let record = FollowerEvacuationRecord::new( + maintenance.clone(), + (record.head_digest(), record.registry()), + policy, + (record.original_key(), record.original_digest()), + record.retired().to_vec(), + record.covered_through(), + record.source_boot(), + (record.replacement_epoch(), record.replacement_evidence()), + record.replacements().to_vec(), + (started, clock()?), + ) + .map_err(operation)?; + Ok(FleetFollowerEvacuationCandidate { snapshot, record }) + }) + .await + } +} diff --git a/crates/cellule-host/src/fleet/follower_evacuation/verification.rs b/crates/cellule-host/src/fleet/follower_evacuation/verification.rs new file mode 100644 index 00000000..805560c5 --- /dev/null +++ b/crates/cellule-host/src/fleet/follower_evacuation/verification.rs @@ -0,0 +1,382 @@ +use super::*; +use crate::durability::enrollment::maintenance::boot_identity; +use cellule_runtime::{ + Error, Result, + fleet::operations::MaintenancePhase, + node::{NodeAdvertisement, NodeDirectory, NodeMode, log_state::NodeLogPhase}, +}; +use tokio::time::{Instant, timeout_at}; + +/// Read-only canonical revalidation; uses no second supervisor or native task. +#[derive(Clone)] +pub struct FleetFollowerEvacuationVerifier { + pub(super) directory: NodeDirectory, + pub(super) transport: std::sync::Arc, +} +impl FleetFollowerEvacuationVerifier { + /// Binds the existing signed canonical directory. Applications pin signing + /// keys and authentic management endpoints before consuming observations. + #[must_use] + pub fn new( + directory: NodeDirectory, + transport: std::sync::Arc, + ) -> Self { + Self { + directory, + transport, + } + } + /// Reloads latest history, current policy, complete roster and exact signed + /// ensemble twice. Fresh interval metadata never renews historical times. + /// Native/foreign inventories and failed processes remain separate barriers. + pub async fn recheck( + &self, + journal: &dyn FleetFollowerEvacuationJournal, + record: &FollowerEvacuationRecord, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + let started = clock()?; + let mut last = started; + let mut clock = || { + let now = clock()?; + interval(started, last, now)?; + last = now; + Ok(now) + }; + bounded(deadline, async { + let snapshot = journal + .load_snapshot(record.policy().scope()) + .await + .map_err(adapter)?; + if journal + .latest_follower_evacuation( + &snapshot, + record.operation().id(), + record.original_key(), + ) + .await + .map_err(adapter)? + .as_ref() + != Some(record) + { + return Err(Error::Fenced); + } + self.confirm(journal, record, &snapshot, deadline, &mut clock, started) + .await + }) + .await + } + pub(super) async fn candidate( + &self, + journal: &dyn FleetFollowerEvacuationJournal, + record: &FollowerEvacuationRecord, + snapshot: &FleetJournalSnapshot, + deadline: Instant, + clock: &mut impl FnMut() -> Result, + ) -> Result { + let started = clock()?; + bounded( + deadline, + self.confirm(journal, record, snapshot, deadline, clock, started), + ) + .await + } + #[allow(clippy::too_many_arguments)] + pub(super) async fn confirm( + &self, + journal: &dyn FleetFollowerEvacuationJournal, + record: &FollowerEvacuationRecord, + snapshot: &FleetJournalSnapshot, + deadline: Instant, + clock: &mut impl FnMut() -> Result, + started: i64, + ) -> Result { + record.to_bytes().map_err(operation)?; + let original = record.operation(); + let current = snapshot.head().maintenance().ok_or(Error::Fenced)?; + if self.directory.fleet() != record.policy().scope().fleet + || snapshot.head().scope() != record.policy().scope() + || snapshot.registry().bootstrap_revision().is_none() + || current.id() != original.id() + || current.node() != original.node() + || current.session() != original.session() + || current.intent_revision() != original.intent_revision() + || current.deadline_ms() != original.deadline_ms() + || !matches!( + current.phase(), + MaintenancePhase::Evacuating | MaintenancePhase::Closing + ) + { + return Err(Error::Fenced); + } + let roster = FleetRoster::collect(journal, snapshot, deadline).await?; + let intent = roster + .intents() + .iter() + .find(|row| row.node() == current.node()) + .ok_or(Error::Fenced)?; + if intent.session() != current.session() + || intent.revision() != current.intent_revision() + || intent.mode() != NodeMode::Draining + { + return Err(Error::Fenced); + } + for row in record.retired() { + if !roster.enrollments().iter().any(|current| current == row) { + return Err(Error::Fenced); + } + } + let source = record.source().map_err(operation)?; + for (epoch, expected) in [ + ( + record.original_epoch().map_err(operation)?, + record.retired().iter().collect::>(), + ), + ( + record.replacement_epoch(), + record + .replacements() + .iter() + .map(|entry| &entry.enrollment) + .collect::>(), + ), + ] { + let all = roster + .enrollments() + .iter() + .filter(|row| { + row.spec().source.is_some_and(|endpoint| { + endpoint.node == source.node && endpoint.session == source.session + }) && row.spec().role + == (cellule_runtime::fleet::operations::EnrollmentRole::Follower { + log_epoch: epoch, + }) + }) + .collect::>(); + if all.len() != expected.len() || all.iter().any(|row| !expected.contains(row)) { + return Err(Error::Fenced); + } + } + let mut authority: Option = None; + let mut native = Vec::new(); + let mut last = started; + for _ in 0..2 { + let now = clock()?; + interval(started, last, now)?; + last = now; + if now >= current.deadline_ms() { + return Err(Error::Deadline); + } + if journal + .follower_replacement_policy(snapshot) + .await + .map_err(adapter)? + != Some(record.policy()) + { + return Err(Error::Fenced); + } + native = self + .collect_native(&roster, record, deadline, clock) + .await?; + let observed = self.observe_directory(&roster, record, clock).await?; + if authority + .as_ref() + .is_some_and(|previous| !nonregressing(previous, &observed)) + { + return Err(Error::Fenced); + } + if journal + .follower_replacement_policy(snapshot) + .await + .map_err(adapter)? + != Some(record.policy()) + { + return Err(Error::Fenced); + } + roster.confirm(journal, deadline).await?; + // Reobserve every original boot after suspended journal reads. The + // source is read again after all members, including the final pass. + self.recheck_native(&roster, &mut native, deadline, clock) + .await?; + let after = self.observe_directory(&roster, record, clock).await?; + if !nonregressing(&observed, &after) { + return Err(Error::Fenced); + } + authority = Some(after); + if journal + .follower_replacement_policy(snapshot) + .await + .map_err(adapter)? + != Some(record.policy()) + { + return Err(Error::Fenced); + } + roster.confirm(journal, deadline).await?; + } + let finished = clock()?; + interval(started, last, finished)?; + if finished >= current.deadline_ms() { + return Err(Error::Deadline); + } + Ok(FleetFollowerEvacuationCheck { + snapshot: snapshot.clone(), + record: record.digest().map_err(operation)?, + roster_digest: roster.digest()?, + authority: authority.ok_or(Error::Fenced)?, + started_at_ms: started, + finished_at_ms: finished, + native, + }) + } + async fn observe_directory( + &self, + roster: &FleetRoster, + record: &FollowerEvacuationRecord, + clock: &mut impl FnMut() -> Result, + ) -> Result { + let source = record.source().map_err(operation)?; + let expected_members = record + .replacements() + .iter() + .map(|entry| entry.enrollment.spec().target.node) + .collect::>(); + let source_boot = roster.boot(source.node, source.session)?; + if source_boot.intent().mode() != NodeMode::Active { + return Err(Error::Fenced); + } + let valid = |leader: &NodeAdvertisement, now: i64| -> Result { + Ok(leader.node() == source.node + && boot_identity(leader)? == record.source_boot() + && leader.accepts_new_roles(now) + && leader.log().is_some_and(|log| { + log.phase() == NodeLogPhase::Open + && log.epoch() == record.replacement_epoch() + && log.members() == expected_members + })) + }; + let leader = self + .directory + .load_if_live(source.session, clock()?) + .await? + .ok_or(Error::Fenced)?; + if !valid(leader.advertisement(), clock()?)? { + return Err(Error::Fenced); + } + for entry in record.replacements() { + let row = &entry.enrollment; + let endpoint = row.spec().target; + if !roster.enrollments().iter().any(|current| current == row) + || row.spec().source.is_none_or(|endpoint| { + endpoint.intent_revision != source_boot.intent().revision() + }) + { + return Err(Error::Fenced); + } + let boot = roster.boot(endpoint.node, endpoint.session)?; + let signed = self + .directory + .load_if_live(endpoint.session, clock()?) + .await? + .ok_or(Error::Fenced)?; + if boot.intent().mode() != NodeMode::Active + || boot.intent().revision() != endpoint.intent_revision + || signed.advertisement().node() != endpoint.node + || boot_identity(signed.advertisement())? != entry.boot_identity + || !signed.advertisement().accepts_new_roles(clock()?) + { + return Err(Error::Fenced); + } + } + let after = self + .directory + .load_if_live(source.session, clock()?) + .await? + .ok_or(Error::Fenced)?; + if !valid(after.advertisement(), clock()?)? + || !nonregressing(leader.advertisement(), after.advertisement()) + { + return Err(Error::Fenced); + } + Ok(after.advertisement().clone()) + } +} +/// Separate fresh canonical confirmation of one persisted follower obligation. +pub struct FleetFollowerEvacuationCheck { + snapshot: FleetJournalSnapshot, + record: Digest, + roster_digest: Digest, + authority: NodeAdvertisement, + started_at_ms: i64, + finished_at_ms: i64, + native: Vec, +} +impl FleetFollowerEvacuationCheck { + /// Original complete source/member native traversals with all-category + /// rechecks after authority discovery. Other physical roles remain separate. + #[must_use] + pub fn native(&self) -> &[FleetNodeInventory] { + &self.native + } + /// Full final barrier; dependent transactions compare it again. + #[must_use] + pub fn snapshot(&self) -> &FleetJournalSnapshot { + &self.snapshot + } + /// Exact immutable record independently confirmed here. + #[must_use] + pub const fn record_digest(&self) -> Digest { + self.record + } + /// Complete retained roster identity, including Pending and terminal rows. + #[must_use] + pub const fn roster_digest(&self) -> Digest { + self.roster_digest + } + /// Current signed canonical leader/log, without ownership permission. + #[must_use] + pub fn authority(&self) -> &NodeAdvertisement { + &self.authority + } + /// Fresh confirmation interval, separate from original capture times. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + (self.started_at_ms, self.finished_at_ms) + } +} +pub(super) fn adapter(source: Box) -> Error { + Error::Facility { + name: "fleet-follower-evacuation-journal", + source, + } +} +pub(super) fn interval(start: i64, last: i64, now: i64) -> Result<()> { + if start < 0 || now < last || now - start > 30_000 { + return Err(Error::Deadline); + } + Ok(()) +} +pub(super) async fn bounded( + deadline: Instant, + future: impl std::future::Future>, +) -> Result { + if Instant::now() >= deadline { + return Err(Error::Deadline); + } + timeout_at(deadline, future) + .await + .map_err(|source| Error::Facility { + name: "fleet-follower-evacuation-deadline", + source: Box::new(source), + })? +} + +fn nonregressing(before: &NodeAdvertisement, after: &NodeAdvertisement) -> bool { + after.generation() >= before.generation() + && after.issued_at_ms() >= before.issued_at_ms() + && after.log().is_some_and(|log| { + before + .log() + .is_some_and(|old| log.tiered_through() >= old.tiered_through()) + }) +} diff --git a/crates/cellule-host/src/fleet/inspection.rs b/crates/cellule-host/src/fleet/inspection.rs new file mode 100644 index 00000000..f58298f5 --- /dev/null +++ b/crates/cellule-host/src/fleet/inspection.rs @@ -0,0 +1,59 @@ +use std::sync::Arc; + +use cellule_runtime::Error; +use cellule_runtime::fleet::operations::{ + FleetActionKind, FleetActionOutcome, FleetInspectionObservation, FleetInspectionRequest, + MovementAction, OperationError, +}; + +use super::actions::{FleetActionExecutor, journal_error, operation, wall_time_ms}; + +impl FleetActionExecutor { + pub(super) async fn execute_inspection( + &self, + request: FleetInspectionRequest, + ) -> Result, Arc> { + let started = wall_time_ms().map_err(Arc::new)?; + if started < request.action().issued_at_ms() || started >= request.deadline_ms() { + return Err(Arc::new(operation(OperationError::Deadline))); + } + self.journal + .authorize_inspection(&request, started) + .await + .map_err(journal_error) + .map_err(Arc::new)?; + let FleetActionKind::Movement { + action: MovementAction::Inspect, + attempt, + } = request.action().kind() + else { + return Err(Arc::new(Error::Control("fleet inspection is not movement"))); + }; + // This path never reads a cached Inspect result and never invokes + // acquisition or cleanup. Current serving comes from actor + authority. + let result = self.inspect(attempt).await.map_err(Arc::new)?; + if let Some(error) = result.error { + return Err(Arc::new(error)); + } + let finished = wall_time_ms().map_err(Arc::new)?; + if finished >= request.deadline_ms() { + return Err(Arc::new(operation(OperationError::Deadline))); + } + let outcome = FleetActionOutcome { + scope: request.action().scope(), + action_key: request + .action() + .key() + .map_err(operation) + .map_err(Arc::new)?, + node: self.node, + session: self.session, + observed_at_ms: finished, + outcome: result.outcome, + }; + FleetInspectionObservation::new(request, started, outcome) + .map(Arc::new) + .map_err(operation) + .map_err(Arc::new) + } +} diff --git a/crates/cellule-host/src/fleet/inventory/mod.rs b/crates/cellule-host/src/fleet/inventory/mod.rs new file mode 100644 index 00000000..ff91fd23 --- /dev/null +++ b/crates/cellule-host/src/fleet/inventory/mod.rs @@ -0,0 +1,250 @@ +//! Canonical full native traversal beneath the application's fleet observer. +use super::{ + FleetJournalSnapshot, FleetNodeSnapshot, FleetOwnedCell, FleetRoster, FleetSnapshotBindings, + FleetSnapshotNativePage, FleetSnapshotRequest, FleetSnapshotSubject, operation, +}; +use crate::read_replicas::{ReaderEnrollmentCompletion, ReaderEnrollmentJobs}; +use crate::{NodeDurabilitySupervisorObservation, NodeState}; +use cellule_runtime::{ + Error, Result, + client::ReadReplicaLifecycleObservation, + follower::FollowerLaneObservation, + identity::{CellId, Digest, NodeId, SessionId}, + node::NodeMode, +}; +use std::collections::HashSet; + +mod pages; +mod recheck; +mod roles; +mod traversal; +pub use recheck::FleetNodeInventoryRecheck; + +const MAX_ENTRIES: usize = 10_000; +const CATEGORIES: usize = 7; +// Four 10,000-row categories, at most 32 producer epochs, Host/supervisor, +// and both local and fleet-wide rechecks. Further bounded rounds may consume +// spare capacity; exhaustion requires a fresh traversal, never nonce eviction. +const MAX_CAPTURES: usize = 4 * MAX_ENTRIES + 32 + 2 + 2 * CATEGORIES; + +/// Fully traversed local owners at an exact durable roster barrier. +/// +/// This is native coverage, not fleet settlement. Authentication, unexpected +/// directory-record discovery, current Cell/log authority, failed-process +/// closure and replacement policy remain required. Unbound categories retain +/// their missing-binding identity; they are never interpreted as empty roles. +/// Applications account these bounded copied collector buffers. Native page +/// charges stay with their original responses and can be released after accept. +pub struct FleetNodeInventory { + node: NodeId, + session: SessionId, + roster: Digest, + snapshot: FleetJournalSnapshot, + nonces: HashSet, + started_at_ms: i64, + finished_at_ms: i64, + collected_at_ms: i64, + rechecked: Option<(i64, i64)>, + host: Host, + headers: [Option

; CATEGORIES], + cells: Vec, + transitioning: Vec, + role_state: RoleState, + readers: Vec, + reader_enrollments: Vec, + follower_lanes: Vec, + follower_enrollments: Vec, + supervisor: Option, +} + +impl FleetNodeInventory { + /// Exact authenticated physical endpoint selected by the collector. + #[must_use] + pub const fn node(&self) -> NodeId { + self.node + } + /// Original boot; a successor cannot replace it during this traversal. + #[must_use] + pub const fn session(&self) -> SessionId { + self.session + } + /// Original fully traversed journal inputs, including terminal records. + #[must_use] + pub const fn roster_digest(&self) -> Digest { + self.roster + } + /// Actual first-to-last capture interval, including all final rechecks. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + (self.started_at_ms, self.finished_at_ms) + } + pub(crate) fn coverage_checkpoint(&self) -> (i64, Option<(i64, i64)>) { + (self.collected_at_ms, self.rechecked) + } + pub(crate) fn coverage_digest(&self) -> Result { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-native-coverage.v1\0"); + hash.update(self.roster.as_bytes()); + hash.update(self.node.as_bytes()); + hash.update(self.session.as_bytes()); + hash.update(&[self.host.state as u8, self.host.mode as u8]); + for bound in [ + self.host.bindings.managed_startup, + self.host.bindings.readers, + self.host.bindings.follower_store, + self.host.bindings.follower_producer, + self.host.bindings.durability_supervisor, + ] { + hash.update(&[u8::from(bound)]); + } + hash.update(&[u8::from(self.host.node_log.is_some())]); + if let Some((session, node, epoch)) = self.host.node_log { + hash.update(session.as_bytes()); + hash.update(node.as_bytes()); + hash.update(&epoch.to_be_bytes()); + } + for header in self.headers { + let header = header.ok_or(Error::Fenced)?; + hash.update(&[u8::from(header.bound)]); + hash.update(&(header.total as u64).to_be_bytes()); + hash.update(header.extra.as_bytes()); + hash.update(&[u8::from(header.topology.is_some())]); + if let Some(topology) = header.topology { + hash.update(topology.as_bytes()); + } + } + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) + } + /// Sealed host composition observed throughout this traversal. + #[must_use] + pub const fn bindings(&self) -> FleetSnapshotBindings { + self.host.bindings + } + /// Shared irreversible admission mode observed throughout capture. + #[must_use] + pub const fn mode(&self) -> NodeMode { + self.host.mode + } + /// Local active node-log binding; remote authority must still be loaded. + #[must_use] + pub const fn node_log(&self) -> Option<(SessionId, NodeId, u64)> { + self.host.node_log + } + /// Generation-bound writers. Transitional obligations are retained separately. + #[must_use] + pub fn cells(&self) -> &[FleetOwnedCell] { + &self.cells + } + /// Outstanding actor activation/close/release keys, never absent writers. + #[must_use] + pub fn transitioning_cells(&self) -> &[CellId] { + &self.transitioning + } + /// Actual reader admission closure; None retains a missing native binding. + #[must_use] + pub const fn readers_closed(&self) -> Option { + self.role_state.readers_closed + } + /// Original accepted producer jobs, including preparation without a request. + #[must_use] + pub fn reader_jobs(&self) -> Option<&ReaderEnrollmentJobs> { + self.role_state.reader_jobs.as_ref() + } + /// Counts unretired lanes and quarantined files. None means unbound. + #[must_use] + pub const fn follower_store_state(&self) -> Option<(usize, usize)> { + self.role_state.follower_store + } + /// Protocol busy, draining and original pending epoch. None means unbound. + #[must_use] + pub const fn follower_producer_state(&self) -> Option<(bool, bool, Option)> { + self.role_state.follower_producer + } + /// Begins all-category revalidation after remote authority/membership scans. + /// The returned collector must finish; its interval starts at the original + /// traversal. Repeated nonces are rejected across all revalidation rounds. + pub fn recheck(&mut self) -> FleetNodeInventoryRecheck<'_> { + FleetNodeInventoryRecheck::new(self) + } + /// Original shared native lifecycle observations for every managed view. + #[must_use] + pub fn readers(&self) -> &[ReadReplicaLifecycleObservation] { + &self.readers + } + /// Every original producer request and its retained native/publication errors. + #[must_use] + pub fn reader_enrollments(&self) -> &[ReaderEnrollmentCompletion] { + &self.reader_enrollments + } + /// All persisted lanes, including sealed, retired and cold lanes. + #[must_use] + pub fn follower_lanes(&self) -> &[FollowerLaneObservation] { + &self.follower_lanes + } + /// Original selected ensembles and producer progress, including unknown CAS. + #[must_use] + pub fn follower_enrollments(&self) -> &[crate::FollowerEnrollmentProgress] { + &self.follower_enrollments + } + /// Original supervisor lifecycle and accepted automatic/requested rotations. + #[must_use] + pub fn supervisor(&self) -> Option<&NodeDurabilitySupervisorObservation> { + self.supervisor.as_ref() + } +} + +#[derive(Clone, Copy, PartialEq, Eq)] +struct Host { + state: NodeState, + mode: NodeMode, + bindings: FleetSnapshotBindings, + node_log: Option<(SessionId, NodeId, u64)>, +} +#[derive(Clone, Copy, PartialEq, Eq)] +struct Header { + topology: Option, + total: usize, + minimum: usize, + extra: Digest, + bound: bool, +} + +#[derive(Default)] +struct RoleState { + actor_counts: Option<(usize, usize)>, + readers_closed: Option, + reader_jobs: Option, + follower_store: Option<(usize, usize)>, + follower_producer: Option<(bool, bool, Option)>, +} + +/// Streams every canonical native category and then rechecks its full fingerprint. +/// +/// Use fresh authenticated requests for `next_subject`, validate responses with +/// the exact request, and release each original page after acceptance. A failed +/// acceptance poisons this scan; restart against a fresh full journal barrier. +/// Matching fingerprints are interval evidence. They do not make open native +/// work atomic or establish remote authority and maintenance finalization. +pub struct FleetNodeInventoryScan<'a> { + roster: &'a FleetRoster, + node: NodeId, + session: SessionId, + stage: usize, + next: Option, + headers: [Option
; CATEGORIES], + counts: [usize; CATEGORIES], + nonces: HashSet, + host: Option, + started_at_ms: Option, + finished_at_ms: i64, + poisoned: bool, + cells: Vec, + transitioning: Vec, + last_cell: Option, + role_state: RoleState, + readers: Vec, + reader_enrollments: Vec, + follower_lanes: Vec, + follower_enrollments: Vec, + supervisor: Option, +} diff --git a/crates/cellule-host/src/fleet/inventory/pages.rs b/crates/cellule-host/src/fleet/inventory/pages.rs new file mode 100644 index 00000000..a68c47ea --- /dev/null +++ b/crates/cellule-host/src/fleet/inventory/pages.rs @@ -0,0 +1,255 @@ +use super::*; +use cellule_runtime::cell::actor::CellInventoryEntry; + +pub(super) fn header( + page: &FleetSnapshotNativePage, +) -> Result<(Header, Option, usize)> { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-native-category.v1\0"); + let mut field = |n: usize| { + hash.update(&(n as u64).to_be_bytes()); + }; + let (topology, total, next, count, bound) = match page { + FleetSnapshotNativePage::Host => (None, 0, None, 0, true), + FleetSnapshotNativePage::Unbound => (None, 0, None, 0, false), + FleetSnapshotNativePage::Cells(p) => { + field(p.owned_cells()); + field(p.transitioning_cells()); + ( + Some(p.topology()), + p.owned_cells() + .checked_add(p.transitioning_cells()) + .ok_or(Error::Capacity("native actor count overflow"))?, + p.next().map(|c| FleetSnapshotSubject::Cells(Some(c))), + p.entries().len(), + true, + ) + } + FleetSnapshotNativePage::Readers(p) => { + field(p.mode() as usize); + field(usize::from(p.closed())); + ( + Some(p.topology()), + p.total_views(), + p.next().map(|c| FleetSnapshotSubject::Readers(Some(c))), + p.entries().len(), + true, + ) + } + FleetSnapshotNativePage::ReaderEnrollments(p) => { + field(p.mode() as usize); + let jobs = p.jobs(); + for n in [ + jobs.retained(), + jobs.running(), + jobs.unobserved(), + jobs.joining(), + usize::from(jobs.draining()), + usize::from(jobs.task_failure().is_some()), + usize::from(jobs.protocol_failure().is_some()), + ] { + field(n); + } + ( + Some(p.topology()), + p.total_enrollments(), + p.next() + .map(|c| FleetSnapshotSubject::ReaderEnrollments(Some(c))), + p.entries().len(), + true, + ) + } + FleetSnapshotNativePage::FollowerLanes(p) => { + field(p.mode() as usize); + field(p.unretired_lanes()); + field(p.quarantined_entries()); + ( + Some(p.topology()), + p.total_lanes(), + p.next() + .map(|c| FleetSnapshotSubject::FollowerLanes(Some(c))), + p.entries().len(), + true, + ) + } + FleetSnapshotNativePage::FollowerEnrollments(p) => { + field(p.mode() as usize); + field(usize::from(p.protocol_busy())); + field(usize::from(p.draining())); + hash.update(&p.pending_epoch().unwrap_or(0).to_be_bytes()); + ( + Some(p.topology()), + p.total_epochs(), + p.next() + .map(|c| FleetSnapshotSubject::FollowerEnrollments(Some(c))), + p.entries().len(), + true, + ) + } + FleetSnapshotNativePage::DurabilitySupervisor(p) => { + field(p.state as usize); + field(usize::from(p.cancellation_requested)); + field(usize::from(p.supervisor_error.is_some())); + field(usize::from(p.requests_error.is_some())); + match &p.rotations { + Err(_) => { + hash.update(&[0]); + } + Ok(rotations) => { + hash.update(&[1, u8::from(rotations.stopped)]); + hash.update(&rotations.running_epoch.unwrap_or(0).to_be_bytes()); + for entry in [&rotations.pending, &rotations.completed] { + hash.update(&[u8::from(entry.is_some())]); + if let Some(entry) = entry { + let progress = &entry.progress; + hash.update(&entry.epoch.to_be_bytes()); + hash.update(&[ + progress.phase() as u8, + u8::from(progress.first_failure().is_some()), + u8::from(progress.latest_failure().is_some()), + ]); + hash.update(&[u8::from(progress.retirement().is_some())]); + if let Some(proof) = progress.retirement() { + retirement(&mut hash, proof); + } + hash.update(&[u8::from(progress.completion().is_some())]); + if let Some(completion) = progress.completion() { + retirement(&mut hash, completion.retirement()); + hash.update(&completion.replacement_epoch().to_be_bytes()); + } + } + } + } + } + (None, 1, None, 1, true) + } + }; + if total > MAX_ENTRIES { + return Err(Error::Capacity("native inventory category bound")); + } + Ok(( + Header { + topology, + total, + minimum: match page { + FleetSnapshotNativePage::Cells(p) => p.owned_cells().max(p.transitioning_cells()), + _ => total, + }, + extra: Digest::from_bytes(*hash.finalize().as_bytes()), + bound, + }, + next, + count, + )) +} + +impl FleetNodeInventoryScan<'_> { + pub(super) fn append(&mut self, page: &FleetSnapshotNativePage) -> Result<()> { + match page { + FleetSnapshotNativePage::Host | FleetSnapshotNativePage::Unbound => {} + FleetSnapshotNativePage::Cells(p) => { + self.role_state.actor_counts = Some((p.owned_cells(), p.transitioning_cells())); + for entry in p.entries() { + ordered( + self.last_cell.map(|cell| *cell.as_bytes()), + *entry.cell().as_bytes(), + )?; + self.last_cell = Some(entry.cell()); + match entry { + CellInventoryEntry::Owned(row) => { + if row.target.application() + != self.roster.snapshot().head().scope().application + { + return Err(Error::Fenced); + } + self.cells.push(FleetOwnedCell { + node: self.node, + session: self.session, + observation: (**row).clone(), + }); + } + CellInventoryEntry::Transitioning { cell } => { + self.transitioning.push(*cell) + } + } + } + } + FleetSnapshotNativePage::Readers(p) => { + self.role_state.readers_closed = Some(p.closed()); + for row in p.entries() { + ordered( + self.readers.last().map(|r| *r.receipt().cell.as_bytes()), + *row.receipt().cell.as_bytes(), + )?; + self.readers.push(*row); + } + } + FleetSnapshotNativePage::ReaderEnrollments(p) => { + self.role_state.reader_jobs = Some(p.jobs().clone()); + for row in p.entries() { + ordered( + self.reader_enrollments + .last() + .map(|r| *r.source.description().cell.as_bytes()), + *row.source.description().cell.as_bytes(), + )?; + self.reader_enrollments.push(row.clone()); + } + } + FleetSnapshotNativePage::FollowerLanes(p) => { + self.role_state.follower_store = + Some((p.unretired_lanes(), p.quarantined_entries())); + for row in p.entries() { + let key = |r: &FollowerLaneObservation| (*r.leader.as_bytes(), r.epoch); + ordered(self.follower_lanes.last().map(key), key(row))?; + self.follower_lanes.push(*row); + } + } + FleetSnapshotNativePage::FollowerEnrollments(p) => { + self.role_state.follower_producer = + Some((p.protocol_busy(), p.draining(), p.pending_epoch())); + for row in p.entries() { + ordered(self.follower_enrollments.last().map(|r| r.epoch), row.epoch)?; + self.follower_enrollments + .push(crate::FollowerEnrollmentProgress { + epoch: row.epoch, + attempt: row.attempt, + members: row.members.clone(), + native_started: row.native_started, + no_effect: row.no_effect, + delivered: row.delivered, + enrollment: row.enrollment, + refusal: row.refusal, + retirement: row.retirement.clone(), + native_closed: row.native_closed, + execution_error: row.execution_error.clone(), + journal_error: row.journal_error.clone(), + }); + } + } + FleetSnapshotNativePage::DurabilitySupervisor(p) => self.supervisor = Some(p.clone()), + } + Ok(()) + } +} + +fn ordered(previous: Option, next: K) -> Result<()> { + if previous.is_some_and(|previous| previous >= next) { + return Err(Error::Node("native inventory rows repeat or regress")); + } + Ok(()) +} + +fn retirement( + hash: &mut blake3::Hasher, + proof: &cellule_runtime::node::log::NodeLogRetirementProof, +) { + let barrier = proof.barrier(); + hash.update(barrier.leader_session().as_bytes()); + hash.update(&barrier.log_epoch().to_be_bytes()); + hash.update(&barrier.covered_through().to_be_bytes()); + hash.update(&(barrier.members().len() as u64).to_be_bytes()); + for member in barrier.members() { + hash.update(member.as_bytes()); + } +} diff --git a/crates/cellule-host/src/fleet/inventory/recheck.rs b/crates/cellule-host/src/fleet/inventory/recheck.rs new file mode 100644 index 00000000..326a7340 --- /dev/null +++ b/crates/cellule-host/src/fleet/inventory/recheck.rs @@ -0,0 +1,94 @@ +use super::traversal::subject; +use super::*; + +/// Rechecks all complete category fingerprints after fleet-wide discovery. +/// +/// Success supplies interval evidence only. Reconfirm the journal and exact +/// current Cell/log authority and replacement policy before committing any +/// maintenance proof. A failed or dropped recheck supplies no confirmation. +pub struct FleetNodeInventoryRecheck<'a> { + inventory: &'a mut FleetNodeInventory, + stage: usize, + poisoned: bool, + started_at_ms: Option, +} + +impl<'a> FleetNodeInventoryRecheck<'a> { + pub(super) fn new(inventory: &'a mut FleetNodeInventory) -> Self { + inventory.rechecked = None; + Self { + inventory, + stage: 0, + poisoned: false, + started_at_ms: None, + } + } + + /// Requests fresh first pages: their fingerprints cover the whole category. + pub fn next_subject(&self) -> Result> { + if self.poisoned { + return Err(Error::Control("native inventory recheck failed")); + } + Ok((self.stage < CATEGORIES).then(|| subject(self.stage))) + } + + /// Accepts a fresh exact request-bound page without replacing original rows. + pub fn accept( + &mut self, + request: &FleetSnapshotRequest, + response: &FleetNodeSnapshot, + now_ms: i64, + ) -> Result<()> { + let expected = self + .next_subject()? + .ok_or(Error::Control("native inventory recheck already complete"))?; + self.poisoned = true; + response.validate(request, now_ms)?; + let original = &mut self.inventory; + if request.expected() != &original.snapshot + || request.node() != original.node + || request.session() != original.session + || request.subject() != &expected + || !original.nonces.insert(request.nonce()) + || original.nonces.len() > MAX_CAPTURES + { + return Err(Error::Fenced); + } + let host = Host { + state: response.state_before(), + mode: response.mode(), + bindings: response.bindings(), + node_log: response.node_log(), + }; + if response.state_after() != host.state || host != original.host { + return Err(Error::Node("native inventory host binding changed")); + } + if response.started_at_ms() < original.finished_at_ms + || response.finished_at_ms() - original.started_at_ms > 30_000 + { + return Err(Error::Node("native inventory recheck interval differs")); + } + let (header, _, _) = pages::header(response.page())?; + if original.headers[self.stage] != Some(header) { + return Err(Error::Node("native inventory category changed")); + } + original.finished_at_ms = response.finished_at_ms(); + self.started_at_ms.get_or_insert(response.started_at_ms()); + self.stage += 1; + self.poisoned = false; + Ok(()) + } + + /// Returns the actual original-to-recheck interval after all seven categories. + /// Individual accepted pages never grant a partial confirmation. + pub fn finish(self) -> Result<(i64, i64)> { + if self.poisoned || self.stage != CATEGORIES { + return Err(Error::Control("native inventory recheck incomplete")); + } + self.inventory.rechecked = Some(( + self.started_at_ms.ok_or(Error::Fenced)?, + self.inventory.finished_at_ms, + )); + Ok(self.inventory.interval()) + } +} diff --git a/crates/cellule-host/src/fleet/inventory/roles.rs b/crates/cellule-host/src/fleet/inventory/roles.rs new file mode 100644 index 00000000..d0de5716 --- /dev/null +++ b/crates/cellule-host/src/fleet/inventory/roles.rs @@ -0,0 +1,220 @@ +use super::*; +use cellule_runtime::fleet::operations::{ + EnrollmentEvent, EnrollmentRecord, EnrollmentRole, EnrollmentStatus, +}; +use cellule_runtime::follower::FollowerLaneState; + +impl FleetNodeInventory { + /// Matches original producer inputs and local roles to the exact full roster. + /// Missing bindings/requests refuse coverage; failed boots, remote authority, + /// producer preparation, replacement policy and finalization still require + /// their own evidence. This performs no enrollment or settlement effect. + pub fn validate_enrollments(&self, roster: &FleetRoster) -> Result<()> { + self.validate_enrollments_with(roster, |_| false) + } + pub(crate) fn validate_enrollments_with( + &self, + roster: &FleetRoster, + empty_enrolled_lane: impl Fn(&EnrollmentRecord) -> bool, + ) -> Result<()> { + if roster.snapshot() != &self.snapshot || roster.digest()? != self.roster { + return Err(Error::Fenced); + } + let host = self.host; + for (index, required) in [ + (2, host.bindings.readers), + (3, host.bindings.readers), + (4, host.bindings.follower_store), + (5, host.bindings.follower_producer), + (6, host.bindings.durability_supervisor), + ] { + if self.headers[index].is_none_or(|header| header.bound != required) { + return Err(Error::Control("native inventory owner is unbound")); + } + } + let mut reader_keys = HashSet::new(); + for completion in &self.reader_enrollments { + let spec = &completion.spec; + let EnrollmentRole::Reader { target, position } = &spec.role else { + return Err(Error::Fenced); + }; + let source = spec.source.ok_or(Error::Fenced)?; + let original = &completion.source; + if spec.scope != roster.snapshot().head().scope() + || spec.target.node != self.node + || spec.target.session != self.session + || original.target() != target + || original.description().incarnation != position.incarnation + || original.epoch() != position.epoch + || original.root() != &position.root + || original.fleet() != spec.scope.fleet + || original.node() != source.node + || original.owner().session != source.session + { + return Err(Error::Fenced); + } + let current = roster + .enrollments() + .iter() + .find(|row| row.spec() == spec) + .ok_or(Error::Control( + "native reader request is missing from the roster", + ))?; + check_publication( + current, + completion.accepted.as_ref(), + completion.event, + completion.published, + )?; + if !reader_keys.insert(spec.key().map_err(operation)?) { + return Err(Error::Fenced); + } + let view = self + .readers + .iter() + .find(|r| r.receipt().cell == target.cell_id()); + if let Some(view) = view { + let receipt = view.receipt(); + if !completion.opening_started + || !completion.opening_joined + || completion.execution_error.is_some() + || receipt.incarnation != position.incarnation + || receipt.commit_sequence < position.root.commit_sequence + || (matches!(completion.event, Some(EnrollmentEvent::Retired(_))) + && !view.locally_joined()) + { + return Err(Error::Fenced); + } + } else if matches!(completion.event, Some(EnrollmentEvent::Established(_))) { + return Err(Error::Control( + "Established reader has no observed native view", + )); + } + } + for view in &self.readers { + if !self.reader_enrollments.iter().any(|r| { + matches!(&r.spec.role, EnrollmentRole::Reader {target, position} + if target.cell_id() == view.receipt().cell && position.incarnation == view.receipt().incarnation) + }) { return Err(Error::Control("native reader lacks its original producer")); } + } + let mut follower_keys = HashSet::new(); + for progress in &self.follower_enrollments { + if progress.epoch == 0 || progress.members.is_empty() || progress.members.len() > 16 { + return Err(Error::Fenced); + } + for member in &progress.members { + let spec = &member.spec; + let source = spec.source.ok_or(Error::Fenced)?; + if spec.scope != roster.snapshot().head().scope() + || source.node != self.node + || source.session != self.session + || !matches!(spec.role, EnrollmentRole::Follower { log_epoch } if log_epoch == progress.epoch) + { + return Err(Error::Fenced); + } + let current = roster + .enrollments() + .iter() + .find(|row| row.spec() == spec) + .ok_or(Error::Control( + "native follower request is missing from the roster", + ))?; + check_publication( + current, + member.accepted.as_ref(), + member.event, + member.published, + )?; + if !follower_keys.insert(spec.key().map_err(operation)?) { + return Err(Error::Fenced); + } + } + } + for lane in &self.follower_lanes { + // Reopened stores can retain an earlier receiving boot's lane. Keep + // that original target session in the roster; never substitute this + // envelope's current boot as evidence that the old process joined. + if !roster.enrollments().iter().any(|row| { + row.spec().target.node == self.node + && row.spec().source.is_some_and(|s| s.session == lane.leader) + && matches!(row.spec().role, EnrollmentRole::Follower { log_epoch } if log_epoch == lane.epoch) + && (row.unresolved() || lane.state == FollowerLaneState::Retired) + }) { return Err(Error::Control("persisted follower lane lacks original enrollment")); } + } + for row in roster.enrollments().iter().filter(|row| row.unresolved()) { + let spec = row.spec(); + let key = spec.key().map_err(operation)?; + match spec.role { + EnrollmentRole::Reader { .. } + if spec.target.node == self.node && spec.target.session == self.session => + { + if !reader_keys.contains(&key) { + return Err(Error::Control("registered reader is unobserved")); + } + } + EnrollmentRole::Follower { log_epoch } => { + if spec + .source + .is_some_and(|s| s.node == self.node && s.session == self.session) + && !follower_keys.contains(&key) + { + return Err(Error::Control("registered follower producer is unobserved")); + } + if row.status() == EnrollmentStatus::Established + && spec.target.node == self.node + && spec.target.session == self.session + && !self.follower_lanes.iter().any(|lane| { + spec.source.is_some_and(|s| s.session == lane.leader) + && lane.epoch == log_epoch + }) + && !empty_enrolled_lane(row) + { + return Err(Error::Control( + "Established follower has no persisted native lane", + )); + } + } + _ => {} + } + } + Ok(()) + } +} + +fn check_publication( + current: &EnrollmentRecord, + accepted: Option<&EnrollmentRecord>, + event: Option, + published: bool, +) -> Result<()> { + if let Some(accepted) = accepted { + if accepted.spec() != current.spec() + || accepted.accepted_at_ms() != current.accepted_at_ms() + { + return Err(Error::Fenced); + } + } else if published { + return Err(Error::Fenced); + } + if published { + let agrees = match event { + Some(EnrollmentEvent::Established(d)) => { + current.status() == EnrollmentStatus::Established + && current.established_evidence() == Some(d) + } + Some(EnrollmentEvent::Retired(d)) => { + current.status() == EnrollmentStatus::Retired + && current.settlement_evidence() == Some(d) + } + Some(EnrollmentEvent::Refused(d)) => { + current.status() == EnrollmentStatus::Refused + && current.settlement_evidence() == Some(d) + } + None => false, + }; + if !agrees { + return Err(Error::Fenced); + } + } + Ok(()) +} diff --git a/crates/cellule-host/src/fleet/inventory/traversal.rs b/crates/cellule-host/src/fleet/inventory/traversal.rs new file mode 100644 index 00000000..913453bc --- /dev/null +++ b/crates/cellule-host/src/fleet/inventory/traversal.rs @@ -0,0 +1,189 @@ +use super::*; + +impl<'a> FleetNodeInventoryScan<'a> { + /// Already accepted generation-bound writers, including after a failed scan. + /// These are partial advisory rows. Independently recheck current authority + /// and actor identity before pressure relief; they supply no complete counts. + #[must_use] + pub fn cells(&self) -> &[FleetOwnedCell] { + &self.cells + } + /// Already accepted transitional keys, never interpreted as absent owners. + #[must_use] + pub fn transitioning_cells(&self) -> &[CellId] { + &self.transitioning + } + /// Pins an exact current Established managed boot in the supplied full roster. + pub fn new(roster: &'a FleetRoster, node: NodeId, session: SessionId) -> Result { + roster.boot(node, session)?; + if roster.snapshot().registry().bootstrap_revision().is_none() { + return Err(Error::Control( + "native inventory requires bootstrapped roster", + )); + } + Ok(Self { + roster, + node, + session, + stage: 0, + next: None, + headers: [None; CATEGORIES], + counts: [0; CATEGORIES], + nonces: HashSet::new(), + host: None, + started_at_ms: None, + finished_at_ms: 0, + poisoned: false, + cells: Vec::new(), + transitioning: Vec::new(), + last_cell: None, + role_state: RoleState::default(), + readers: Vec::new(), + reader_enrollments: Vec::new(), + follower_lanes: Vec::new(), + follower_enrollments: Vec::new(), + supervisor: None, + }) + } + /// Returns the exact next category/cursor. None means every recheck completed. + pub fn next_subject(&self) -> Result> { + if self.poisoned { + return Err(Error::Control("native inventory scan failed")); + } + Ok(if self.stage == 2 * CATEGORIES { + None + } else { + Some( + self.next + .clone() + .unwrap_or_else(|| subject(self.stage % CATEGORIES)), + ) + }) + } + /// Accepts one original request-bound response. This performs no I/O/effect. + pub fn accept( + &mut self, + request: &FleetSnapshotRequest, + response: &FleetNodeSnapshot, + now_ms: i64, + ) -> Result<()> { + let expected = self + .next_subject()? + .ok_or(Error::Control("native inventory already complete"))?; + self.poisoned = true; + response.validate(request, now_ms)?; + if request.expected() != self.roster.snapshot() + || request.node() != self.node + || request.session() != self.session + || request.subject() != &expected + || !self.nonces.insert(request.nonce()) + || self.nonces.len() > MAX_CAPTURES + { + return Err(Error::Fenced); + } + let host = Host { + state: response.state_before(), + mode: response.mode(), + bindings: response.bindings(), + node_log: response.node_log(), + }; + if response.state_after() != host.state + || !host.bindings.managed_startup + || !matches!(host.state, NodeState::Ready | NodeState::Maintenance) + || self.host.is_some_and(|old| old != host) + || self.roster.boot(self.node, self.session)?.intent().mode() != host.mode + || response.started_at_ms() < self.finished_at_ms + { + return Err(Error::Node("native inventory host binding changed")); + } + let started = *self.started_at_ms.get_or_insert(response.started_at_ms()); + if response.finished_at_ms() - started > 30_000 { + return Err(Error::Node("native inventory capture interval exceeded")); + } + self.host = Some(host); + self.finished_at_ms = response.finished_at_ms(); + let index = self.stage % CATEGORIES; + let (header, next, count) = pages::header(response.page())?; + if self.headers[index].is_some_and(|old| old != header) { + return Err(Error::Node("native inventory category changed")); + } + if self.stage < CATEGORIES { + self.headers[index] = Some(header); + let total = self.counts[index] + .checked_add(count) + .filter(|total| *total <= header.total && *total <= MAX_ENTRIES) + .ok_or(Error::Capacity("native inventory row bound"))?; + if (next.is_none() && total < header.minimum) || (next.is_some() && count == 0) { + return Err(Error::Node("native inventory continuation is incomplete")); + } + self.append(response.page())?; + if index == 1 + && next.is_none() + && self + .role_state + .actor_counts + .is_none_or(|(_, transitions)| transitions != self.transitioning.len()) + { + return Err(Error::Node("native actor transition count differs")); + } + self.counts[index] = total; + self.next = next; + if self.next.is_none() { + self.stage += 1; + } + } else { + // The first page fingerprints the complete category, including rows + // outside it. It is a fresh recheck, never a cached initial response. + self.stage += 1; + self.next = None; + } + self.poisoned = false; + Ok(()) + } + /// Produces a bounded traversal after every category and recheck. + /// Call `validate_enrollments` to match the durable responsibilities. Then + /// recheck after remote authority/policy discovery and reconfirm the roster. + /// No combination of these local calls alone proves fleet settlement. + pub fn finish(self) -> Result { + if self.poisoned || self.stage != 2 * CATEGORIES { + return Err(Error::Control("native inventory traversal incomplete")); + } + Ok(FleetNodeInventory { + node: self.node, + session: self.session, + roster: self.roster.digest()?, + snapshot: self.roster.snapshot().clone(), + nonces: self.nonces, + started_at_ms: self + .started_at_ms + .ok_or(Error::Control("native inventory interval missing"))?, + finished_at_ms: self.finished_at_ms, + collected_at_ms: self.finished_at_ms, + rechecked: None, + host: self + .host + .ok_or(Error::Control("native inventory host missing"))?, + headers: self.headers, + cells: self.cells, + transitioning: self.transitioning, + role_state: self.role_state, + readers: self.readers, + reader_enrollments: self.reader_enrollments, + follower_lanes: self.follower_lanes, + follower_enrollments: self.follower_enrollments, + supervisor: self.supervisor, + }) + } +} + +pub(super) fn subject(index: usize) -> FleetSnapshotSubject { + match index { + 0 => FleetSnapshotSubject::Host, + 1 => FleetSnapshotSubject::Cells(None), + 2 => FleetSnapshotSubject::Readers(None), + 3 => FleetSnapshotSubject::ReaderEnrollments(None), + 4 => FleetSnapshotSubject::FollowerLanes(None), + 5 => FleetSnapshotSubject::FollowerEnrollments(None), + _ => FleetSnapshotSubject::DurabilitySupervisor, + } +} diff --git a/crates/cellule-host/src/fleet/journal.rs b/crates/cellule-host/src/fleet/journal.rs new file mode 100644 index 00000000..763bbbfc --- /dev/null +++ b/crates/cellule-host/src/fleet/journal.rs @@ -0,0 +1,136 @@ +use std::{future::Future, pin::Pin}; + +use cellule_runtime::fleet::operations::{ + AcceptedFleetAction, AcquisitionBasis, AttemptId, FleetAction, FleetActionOutcome, + FleetInspectionRequest, FleetScope, MovementAction, RecoveryBasis, RecoveryEvidence, +}; +use cellule_runtime::identity::{NodeId, SessionId}; + +/// Provider-neutral fleet adapter call preserving the original source error. +pub type FleetAdapterFuture<'a, T> = + Pin>> + Send + 'a>>; + +/// Atomic first acceptance or a previously committed exact action. +pub enum FleetActionAcceptance { + /// The backend atomically validated the head and committed this acceptance. + /// Only this result permits starting its effect for the first time. + New(AcceptedFleetAction), + /// The exact action was already accepted. A missing result is unresolved; + /// it is never permission to repeat an unobserved effect. + Existing { + /// Original immutable acceptance, retained across controller changes. + accepted: AcceptedFleetAction, + /// Latest committed checked result, if available. + result: Option>, + }, +} + +/// Strongly consistent application journal for local fleet actions. +/// +/// First acceptance must linearize its fresh head, controller epoch, permit, +/// intent revision, endpoint and deadline checks with record publication. +/// Use `AcceptedFleetAction::new` inside that transaction. An unconditional +/// write following a separate head read is insufficient. Compare complete +/// execution inputs with `validate_replay`, not only the stable action key. +/// Applications authenticate the caller before invoking the node. +pub trait FleetActionJournal: Send + Sync + 'static { + /// Checks a native page request against its full current head/registry and + /// endpoint intent in one transaction using `request.authorize_against`. + /// Applications authenticate the transport. This read can publish no effect. + fn authorize_snapshot<'a>( + &'a self, + request: &'a super::FleetSnapshotRequest, + now_ms: i64, + ) -> FleetAdapterFuture<'a, ()>; + + /// Checks a read-only inspection against the current head, registry version + /// and endpoint intent in one transaction. Use `request.authorize_against`; never satisfy this + /// with a cached acceptance or effect result. No effect or result is published. + fn authorize_inspection<'a>( + &'a self, + request: &'a FleetInspectionRequest, + now_ms: i64, + ) -> FleetAdapterFuture<'a, ()>; + + /// Atomically accepts new work or returns its original durable acceptance. + /// Ambiguous writes are reconciled by the same identity on a later call. + fn accept_action<'a>( + &'a self, + action: &'a FleetAction, + node: NodeId, + session: SessionId, + now_ms: i64, + ) -> FleetAdapterFuture<'a, FleetActionAcceptance>; + + /// Publishes a checked result bound to the original acceptance. + /// + /// Identical terminal results are idempotent; incompatible terminal results + /// conflict. Unknown observations may be replaced by checked terminal + /// evidence. A successful return means the result is durably reachable, + /// including after backend/client reconstruction. Retiring a fleet permit + /// remains a separate controller transition requiring all cleanup evidence. + fn publish_action_result<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + result: &'a FleetActionOutcome, + ) -> FleetAdapterFuture<'a, ()>; + + /// Looks up original accepted work by immutable attempt, effect and endpoint. + /// Return Existing records only, with the original result or unresolved + /// marker. Inspection does not authorize first acceptance. + fn load_movement_action<'a>( + &'a self, + scope: FleetScope, + attempt: AttemptId, + effect: MovementAction, + node: NodeId, + session: SessionId, + ) -> FleetAdapterFuture<'a, Option>; + + /// Durably records the exact checked input before receiver acquisition. + /// + /// Bind it to the original acceptance atomically. Identical accepted/control + /// inputs return the original record and capture time. Changed controls + /// conflict; do not erase the basis after later owner publication. A lost + /// response must prevent takeover until the same basis is confirmed. + fn record_acquisition_basis<'a>( + &'a self, + basis: &'a AcquisitionBasis, + ) -> FleetAdapterFuture<'a, AcquisitionBasis>; + + /// Loads the retained basis without inferring it from a successor's root. + fn load_acquisition_basis<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + ) -> FleetAdapterFuture<'a, Option>; + + /// Atomically binds immutable recovery input to the original acceptance. + /// Identical inputs return their original capture time; changed controls + /// conflict. Unknown write replies prevent ownership CAS until confirmed. + fn record_recovery_basis<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + basis: &'a RecoveryBasis, + ) -> FleetAdapterFuture<'a, RecoveryBasis>; + + /// Loads original recovery input without reconstructing it from a successor. + fn load_recovery_basis<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + ) -> FleetAdapterFuture<'a, Option>; + + /// Confirms the canonical materialized recovery position before admission. + /// Bind to the exact original basis; retain it through later publication. + /// Identical writes return the original record/time, incompatible ones fail. + fn record_recovery_evidence<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + evidence: &'a RecoveryEvidence, + ) -> FleetAdapterFuture<'a, RecoveryEvidence>; + + /// Loads the confirmed pre-admission recovery position, or explicit absence. + fn load_recovery_evidence<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + ) -> FleetAdapterFuture<'a, Option>; +} diff --git a/crates/cellule-host/src/fleet/maintenance.rs b/crates/cellule-host/src/fleet/maintenance.rs new file mode 100644 index 00000000..58f7cf32 --- /dev/null +++ b/crates/cellule-host/src/fleet/maintenance.rs @@ -0,0 +1,48 @@ +//! Monotonic admission closure under an exact accepted physical-node intent. + +use super::actions::{ActionResult, FleetActionExecutor}; +use cellule_runtime::Error; +use cellule_runtime::fleet::operations::{ + AcceptedFleetAction, FleetActionKind, FleetOutcome, MaintenanceAction, +}; +use cellule_runtime::node::NodeMode; + +impl FleetActionExecutor { + pub(super) async fn perform_action( + &self, + accepted: &AcceptedFleetAction, + ) -> cellule_runtime::Result { + match accepted.action().kind() { + FleetActionKind::Movement { .. } => self.perform_movement(accepted).await, + FleetActionKind::Maintenance { .. } => self.perform_maintenance(accepted), + } + } + + pub(super) fn perform_maintenance( + &self, + accepted: &AcceptedFleetAction, + ) -> cellule_runtime::Result { + let FleetActionKind::Maintenance { + action: MaintenanceAction::Cordon, + operation, + } = accepted.action().kind() + else { + return Err(Error::Control("unsupported fleet maintenance effect")); + }; + if accepted.action().scope() != self.scope + || operation.node() != self.node + || operation.session() != self.session + { + return Err(Error::Fenced); + } + // The journal accepted the retained Draining intent. This closes the + // same gate used by writers, readers and followers, without releasing + // any existing responsibility or taking the node shutdown lane. + let gate = self.runtime.node_admission(); + gate.begin_drain()?; + if gate.mode()? != NodeMode::Draining { + return Err(Error::CellDraining); + } + Ok(ActionResult::checked(FleetOutcome::Cordoned)) + } +} diff --git a/crates/cellule-host/src/fleet/mod.rs b/crates/cellule-host/src/fleet/mod.rs new file mode 100644 index 00000000..5e15fcb0 --- /dev/null +++ b/crates/cellule-host/src/fleet/mod.rs @@ -0,0 +1,70 @@ +//! Journal-bound fleet execution through the node's canonical runtime. +//! +//! Applications authenticate management requests and supply a strongly +//! consistent journal. Local execution owns accepted work independently of +//! transport waiters. Controller scheduling and deployment remain application +//! responsibilities. + +mod actions; +mod cells; +mod controller; +mod coverage; +mod enrollment; +mod failed_boot; +mod follower_evacuation; +pub use follower_evacuation::{ + FleetFollowerEvacuationCandidate, FleetFollowerEvacuationCheck, FleetFollowerEvacuationJournal, + FleetFollowerEvacuationPublication, FleetFollowerEvacuationVerifier, +}; +mod inspection; +mod inventory; +mod journal; +mod maintenance; +mod movement; +mod reader_evacuation; +mod reconciler; +mod recovered; +pub use reader_evacuation::{ + FleetReaderEvacuationCandidate, FleetReaderEvacuationCheck, FleetReaderEvacuationJournal, + FleetReaderEvacuationPublication, FleetReaderEvacuationVerifier, +}; +mod references; +mod roster; +pub(crate) mod snapshot; +pub(crate) mod withdrawal; + +pub use actions::FleetActionCompletion; +pub use cells::{FleetCellInputs, FleetCellProvider, FleetRecoveryInputs}; +pub use controller::{FleetJournal, FleetJournalSnapshot}; +pub use coverage::FleetRoleCoverage; +pub use enrollment::{FleetBootObservation, FleetEnrollmentAcceptance, FleetEnrollmentJournal}; +pub use failed_boot::{ + FleetFailedBootClosure, FleetFailedBootProcessConfirmation, FleetFailedBootProcessEvidence, + FleetFailedBootProcessRequest, FleetFailedBootProcesses, FleetFailedBootPublication, + FleetFailedBootRetirement, FleetFailedReaderClosure, FleetFailedReaderPublication, + FleetFailedReaderRetirement, FleetOriginalCatalogSet, FleetOriginalCatalogSource, + FleetOriginalCatalogs, FleetOriginalWriterCapture, FleetOriginalWriterInventory, + FleetOriginalWriterJournal, +}; +pub use inventory::{FleetNodeInventory, FleetNodeInventoryRecheck, FleetNodeInventoryScan}; +pub use journal::{FleetActionAcceptance, FleetActionJournal, FleetAdapterFuture}; +pub use reconciler::{ + FleetAttemptFailure, FleetObservation, FleetObserver, FleetOwnedCell, FleetReconcileReport, + FleetReconciler, FleetTransport, +}; +pub use recovered::{ + FleetRecoveredFollowerClosure, FleetRecoveredFollowerMember, FleetRecoveredFollowerPublication, + FleetRecoveredFollowerRetirement, +}; +pub use references::FleetFollowerReferences; +pub use roster::{FleetRoster, FleetRosterBoot}; +pub use snapshot::{ + FleetNodeSnapshot, FleetSnapshotBindings, FleetSnapshotNativePage, FleetSnapshotRequest, + FleetSnapshotSubject, FleetSnapshotTransport, +}; + +pub(crate) use actions::FleetActionExecutor; +pub(crate) use actions::operation; + +/// Stable name of the node-owned finite fleet-action executor. +pub const FLEET_ACTION_COMPONENT: &str = "fleet-actions"; diff --git a/crates/cellule-host/src/fleet/movement/activation.rs b/crates/cellule-host/src/fleet/movement/activation.rs new file mode 100644 index 00000000..982ac3d0 --- /dev/null +++ b/crates/cellule-host/src/fleet/movement/activation.rs @@ -0,0 +1,221 @@ +use super::*; + +impl FleetActionExecutor { + pub(super) async fn activate( + &self, + accepted: &AcceptedFleetAction, + attempt: &MoveAttempt, + ) -> cellule_runtime::Result { + let prepared = match self.runtime.prepared_receiver(attempt.spec().id)? { + Some(_) => { + let prepared = self.prepared(attempt)?; + match prepared.state()? { + ReceiverState::Prepared => Some(prepared), + ReceiverState::Cancelled + if self.confirmed_credit_settlement(attempt).await? => + { + None + } + _ => return Ok(ActionResult::checked(FleetOutcome::Unknown)), + } + } + None if self.confirmed_credit_settlement(attempt).await? => None, + None => { + return Err(Error::Peer( + "fleet receiver credit is absent without cleanup proof", + )); + } + }; + let inputs = self.inputs(attempt).await?; + let observed = inputs + .authority + .load(attempt.spec().target.cell_id()) + .await? + .ok_or(Error::Control("fleet activation authority is absent"))?; + self.check_contract(attempt, &inputs, &observed)?; + if prepared.is_none() + && observed.value().state == ControlState::Serving + && observed + .value() + .owner + .as_ref() + .is_some_and(|owner| owner.session == self.session) + { + // Ordinary acquisition can win after unused credit was joined. It + // needs current serving proof, not a second ownership CAS. + return self + .serving(attempt, &inputs) + .await + .map(ActionResult::checked); + } + let basis = + AcquisitionBasis::new(accepted.clone(), observed.value().clone(), wall_time_ms()?) + .map_err(operation)?; + let retained = self + .journal + .record_acquisition_basis(&basis) + .await + .map_err(journal_error)?; + if retained.accepted() != accepted + || retained.control() != observed.value() + || retained.observed_at_ms() > basis.observed_at_ms() + { + return Err(Error::Peer("journal changed checked acquisition basis")); + } + if let Some(prepared) = prepared { + self.runtime + .activate_prepared_receiver( + &prepared, + inputs.authority.clone(), + observed, + inputs.owner.clone(), + wall_time_ms()?, + ) + .await?; + } else { + // After confirmed unused-credit cleanup, acquire through the normal + // admitted restore path. The fleet permit and exact release remain + // charged; this does not create another movement attempt. + self.runtime + .acquire_idle_restored( + inputs.catalog.clone(), + inputs.replica.clone(), + inputs.authority.clone(), + observed, + inputs.destination.clone(), + inputs.owner.clone(), + ) + .await?; + } + let outcome = self.serving(attempt, &inputs).await?; + let envelope = cellule_runtime::fleet::operations::FleetActionOutcome { + scope: accepted.action().scope(), + action_key: accepted.action().key().map_err(operation)?, + node: self.node, + session: self.session, + observed_at_ms: wall_time_ms()?, + outcome: outcome.clone(), + }; + retained.validate_result(&envelope).map_err(operation)?; + Ok(ActionResult::checked(outcome)) + } + + pub(super) async fn serving( + &self, + attempt: &MoveAttempt, + inputs: &FleetCellInputs, + ) -> cellule_runtime::Result { + let release = attempt.released().ok_or_else(|| { + operation(cellule_runtime::fleet::operations::OperationError::Invalid( + "activation without release", + )) + })?; + let evidence = self + .serving_evidence(attempt, inputs, ServingPrefix::Released(&release.root)) + .await?; + attempt.validate_activation(&evidence).map_err(operation)?; + Ok(FleetOutcome::Activated(evidence)) + } + + pub(super) async fn serving_evidence( + &self, + attempt: &MoveAttempt, + inputs: &FleetCellInputs, + required: ServingPrefix<'_>, + ) -> cellule_runtime::Result { + let spec = attempt.spec(); + let current = inputs + .authority + .load(spec.target.cell_id()) + .await? + .ok_or(Error::Fenced)?; + self.check_contract(attempt, inputs, ¤t)?; + if current.value().state != ControlState::Serving + || current.value().epoch <= spec.source_epoch + || current + .value() + .owner + .as_ref() + .is_none_or(|owner| owner.session != self.session) + { + return Err(Error::Fenced); + } + let handle = self + .runtime + .local_handle(inputs.catalog.clone(), ¤t) + .await? + .ok_or(Error::CellDraining)?; + // The ordinary FIFO query/admission boundary checks the lease and joins + // prior publication. It creates no command or alternative response gate. + handle.query(1, 1, |_| Ok(Vec::new())).await?; + let latest = inputs + .authority + .load(spec.target.cell_id()) + .await? + .ok_or(Error::Fenced)?; + if latest.value().incarnation != spec.incarnation + || latest.value().epoch != current.value().epoch + || latest.value().state != ControlState::Serving + || latest + .value() + .owner + .as_ref() + .is_none_or(|owner| owner.session != self.session) + { + return Err(Error::Fenced); + } + let root = latest.value().ltx_root().ok_or(Error::Fenced)?; + self.verify_serving_prefix(attempt, inputs, required, root) + .await?; + // Origin verification can outlive the first actor query. Recheck the + // same admitted native owner and exact selected root before returning + // serving evidence; historical counters cannot replace this boundary. + handle.query(1, 1, |_| Ok(Vec::new())).await?; + let confirmed = inputs + .authority + .load(spec.target.cell_id()) + .await? + .ok_or(Error::Fenced)?; + self.check_contract(attempt, inputs, &confirmed)?; + if confirmed.value().owner != latest.value().owner + || confirmed.value().epoch != latest.value().epoch + || confirmed.value().state != ControlState::Serving + || confirmed.value().ltx_root() != Some(root) + { + return Err(Error::Fenced); + } + let position = PublishedPosition { + incarnation: latest.value().incarnation, + epoch: latest.value().epoch, + root: latest.value().root.clone().ok_or(Error::Fenced)?, + }; + let mut cursor = None; + loop { + let page = self.runtime.fleet_cells_page(cursor, 128).await?; + if let Some(entry) = page + .entries() + .iter() + .find(|entry| entry.cell() == spec.target.cell_id()) + { + return match entry { + CellInventoryEntry::Owned(owner) + if owner.incarnation == spec.incarnation + && owner.position.as_ref() == Some(&position) => + { + let evidence = ActivationEvidence { + node: self.node, + session: self.session, + position, + }; + Ok(evidence) + } + _ => Err(Error::CellDraining), + }; + } + match page.next() { + Some(next) => cursor = Some(next), + None => return Err(Error::Fenced), + } + } + } +} diff --git a/crates/cellule-host/src/fleet/movement/inspection.rs b/crates/cellule-host/src/fleet/movement/inspection.rs new file mode 100644 index 00000000..8d1c0699 --- /dev/null +++ b/crates/cellule-host/src/fleet/movement/inspection.rs @@ -0,0 +1,295 @@ +use super::*; + +impl FleetActionExecutor { + pub(super) async fn cancel( + &self, + attempt: &MoveAttempt, + ) -> cellule_runtime::Result { + if self.runtime.prepared_receiver(attempt.spec().id)?.is_none() { + return if self.confirmed_credit_settlement(attempt).await? { + Ok(ActionResult::checked(FleetOutcome::ReceiverCleaned)) + } else { + Err(Error::Peer( + "receiver cleanup has no retained resource proof", + )) + }; + } + let prepared = self.prepared(attempt)?; + match prepared.state()? { + ReceiverState::Prepared | ReceiverState::Cancelled => { + self.runtime.cancel_prepared_receiver(&prepared)? + } + ReceiverState::Activated + if matches!( + attempt.phase(), + AttemptPhase::Activated | AttemptPhase::CleaningReceiver + ) => {} + _ => return Ok(ActionResult::checked(FleetOutcome::Unknown)), + } + Ok(ActionResult::checked(FleetOutcome::ReceiverCleaned)) + } + + pub(super) async fn confirmed_credit_settlement( + &self, + attempt: &MoveAttempt, + ) -> cellule_runtime::Result { + for effect in [ + MovementAction::Cancel, + MovementAction::Activate, + MovementAction::Recover, + ] { + if self.confirmed_credit_result(attempt, effect).await? { + return Ok(true); + } + } + Ok(false) + } + + pub(super) async fn confirmed_credit_result( + &self, + attempt: &MoveAttempt, + effect: MovementAction, + ) -> cellule_runtime::Result { + let original = self + .journal + .load_movement_action( + self.scope, + attempt.spec().id, + effect, + self.node, + self.session, + ) + .await + .map_err(journal_error)?; + let Some(FleetActionAcceptance::Existing { + accepted, + result: Some(result), + }) = original + else { + return Ok(false); + }; + if accepted.action().scope() != self.scope + || accepted.node() != self.node + || accepted.session() != self.session + { + return Err(Error::Fenced); + } + accepted.validate_result(&result).map_err(operation)?; + let FleetActionKind::Movement { + action, + attempt: original, + } = accepted.action().kind() + else { + return Err(Error::Fenced); + }; + if *action != effect || original.spec() != attempt.spec() { + return Err(Error::Fenced); + } + if matches!( + (&result.outcome, effect), + (FleetOutcome::ReceiverCleaned, MovementAction::Cancel) + | (FleetOutcome::Activated(_), MovementAction::Activate) + | (FleetOutcome::Recovered(_), MovementAction::Recover) + ) { + return Ok(true); + } + Ok(false) + } + + pub(in crate::fleet) async fn inspect( + &self, + attempt: &MoveAttempt, + ) -> cellule_runtime::Result { + let spec = attempt.spec(); + if self.session == spec.source { + let release_kinds: &[MovementAction] = match attempt.phase() { + AttemptPhase::MaintenanceReleasing => &[MovementAction::ReleaseMaintenance], + AttemptPhase::Releasing => &[MovementAction::Release], + _ => &[MovementAction::ReleaseMaintenance, MovementAction::Release], + }; + for &release_kind in release_kinds { + let original = self + .journal + .load_movement_action( + self.scope, + spec.id, + release_kind, + self.node, + self.session, + ) + .await + .map_err(journal_error)?; + if let Some(FleetActionAcceptance::Existing { + accepted, + result: Some(result), + }) = original + { + if accepted.action().scope() != self.scope + || accepted.node() != self.node + || accepted.session() != self.session + { + return Err(Error::Fenced); + } + accepted.validate_result(&result).map_err(operation)?; + if let FleetActionKind::Movement { + action, + attempt: original, + } = accepted.action().kind() + && *action == release_kind + && original.spec() == spec + && matches!(result.outcome, FleetOutcome::Released(_)) + { + return Ok(ActionResult::checked(result.outcome)); + } + } + } + return Ok(ActionResult::checked(FleetOutcome::Unknown)); + } + if (self.node, self.session) != (spec.destination_node, spec.destination) { + // The fresh request/journal boundary permits only known-release + // successor reads here. Never acquire, inspect another endpoint's + // retained effects, or infer settlement of its prepared resources. + let inputs = self.inputs(attempt).await?; + return self + .serving(attempt, &inputs) + .await + .map(ActionResult::checked); + } + if matches!( + attempt.phase(), + AttemptPhase::Recovering | AttemptPhase::Recovered + ) && let Some(FleetActionAcceptance::Existing { accepted, .. }) = self + .journal + .load_movement_action( + self.scope, + spec.id, + MovementAction::Recover, + self.node, + self.session, + ) + .await + .map_err(journal_error)? + { + // Observation cannot start acquisition. Missing historical recovery + // position stays Unknown; only replay of Recover may resume it. + let inputs = self.inputs(attempt).await?; + if self + .journal + .load_recovery_evidence(&accepted) + .await + .map_err(journal_error)? + .is_some() + { + return self.recovered_serving(&accepted, attempt, &inputs).await; + } + return Ok(ActionResult::checked(FleetOutcome::Unknown)); + } + if self.runtime.prepared_receiver(spec.id)?.is_some() { + let prepared = self.prepared(attempt)?; + match prepared.state()? { + ReceiverState::Prepared => return self.inspect_preparation(attempt), + ReceiverState::Cancelled => { + return Ok(ActionResult::checked(FleetOutcome::ReceiverCleaned)); + } + ReceiverState::Activated if attempt.released().is_some() => {} + _ => return Ok(ActionResult::checked(FleetOutcome::Unknown)), + } + } + // The bounded local receipt can retire after result publication. Its + // absence is not proof of resource settlement; use the retained action + // and basis, then establish serving again through the actual actor. + for effect in [MovementAction::Activate, MovementAction::Cancel] { + let original = self + .journal + .load_movement_action(self.scope, spec.id, effect, self.node, self.session) + .await + .map_err(journal_error)?; + let Some(FleetActionAcceptance::Existing { + accepted, + result: Some(result), + }) = original + else { + continue; + }; + if accepted.action().scope() != self.scope + || accepted.node() != self.node + || accepted.session() != self.session + { + return Err(Error::Fenced); + } + accepted.validate_result(&result).map_err(operation)?; + let FleetActionKind::Movement { + action, + attempt: original, + } = accepted.action().kind() + else { + return Err(Error::Fenced); + }; + if *action != effect || original.spec() != spec { + return Err(Error::Fenced); + } + match (&result.outcome, effect) { + (FleetOutcome::Activated(_), MovementAction::Activate) => { + let basis = self + .journal + .load_acquisition_basis(&accepted) + .await + .map_err(journal_error)?; + if let Some(basis) = &basis { + if basis.accepted() != &accepted { + return Err(Error::Fenced); + } + } else if !self + .confirmed_credit_result(attempt, MovementAction::Cancel) + .await? + { + return Err(Error::Peer( + "activation inspection lacks acquisition or cleanup basis", + )); + } + let inputs = self.inputs(attempt).await?; + let outcome = self.serving(attempt, &inputs).await?; + let mut checked = *result; + checked.observed_at_ms = wall_time_ms()?; + checked.outcome = outcome.clone(); + if let Some(basis) = basis { + basis.validate_result(&checked).map_err(operation)?; + } else { + accepted.validate_result(&checked).map_err(operation)?; + } + return Ok(ActionResult::checked(outcome)); + } + (FleetOutcome::ReceiverCleaned, MovementAction::Cancel) => { + return Ok(ActionResult::checked(FleetOutcome::ReceiverCleaned)); + } + _ => {} + } + } + Ok(ActionResult::checked(FleetOutcome::Unknown)) + } + + pub(in crate::fleet) async fn retire_receiver_receipt( + &self, + completion: &FleetActionCompletion, + ) -> cellule_runtime::Result<()> { + if !matches!( + completion.outcome.outcome, + FleetOutcome::Activated(_) | FleetOutcome::Recovered(_) | FleetOutcome::ReceiverCleaned + ) { + return Ok(()); + } + let FleetActionKind::Movement { attempt, .. } = completion.accepted.action().kind() else { + return Ok(()); + }; + if self.session != attempt.spec().destination { + return Ok(()); + } + if let Some(prepared) = self.runtime.prepared_receiver(attempt.spec().id)? { + if prepared.spec()? != *attempt.spec() { + return Err(Error::Fenced); + } + self.runtime.retire_prepared_receiver(&prepared).await?; + } + Ok(()) + } +} diff --git a/crates/cellule-host/src/fleet/movement/mod.rs b/crates/cellule-host/src/fleet/movement/mod.rs new file mode 100644 index 00000000..c6ef9aa8 --- /dev/null +++ b/crates/cellule-host/src/fleet/movement/mod.rs @@ -0,0 +1,187 @@ +//! Canonical receiver admission, movement, and retained evidence. + +use super::actions::{ + ActionResult, FleetActionCompletion, FleetActionExecutor, journal_error, operation, + wall_time_ms, +}; +use super::{FleetActionAcceptance, FleetCellInputs}; +use cellule_runtime::Error; +use cellule_runtime::cell::actor::{CellInventoryEntry, PreparedCellReceiver, ReceiverState}; +use cellule_runtime::control::{ControlState, authority::VersionedControl}; +use cellule_runtime::fleet::operations::{ + AcceptedFleetAction, AcquisitionBasis, ActivationEvidence, AttemptPhase, DrainBlocker, + FleetActionKind, FleetOutcome, MoveAttempt, MovementAction, PublishedPosition, +}; + +mod activation; +mod inspection; +mod prefix; +mod receiver; +mod recovery; + +pub(super) enum ServingPrefix<'a> { + Released(&'a cellule_runtime::control::RootRef), + Recovered(&'a cellule_runtime::fleet::operations::RecoveryEvidence), +} + +impl FleetActionExecutor { + pub(super) async fn perform_movement( + &self, + accepted: &AcceptedFleetAction, + ) -> cellule_runtime::Result { + let FleetActionKind::Movement { action, attempt } = accepted.action().kind() else { + return Err(Error::Control("fleet movement has no attempt")); + }; + let spec = attempt.spec(); + let now = wall_time_ms()?; + match action { + MovementAction::Release => { + if now >= spec.deadline_ms + || attempt.reservation().is_none_or(|r| now >= r.expires_at_ms) + { + return Ok(ActionResult::checked(FleetOutcome::Rejected( + DrainBlocker::Deadline, + ))); + } + match self + .runtime + .release_idle_cell_at( + spec.target.cell_id(), + spec.source, + spec.generation, + spec.incarnation, + spec.source_epoch, + ) + .await + { + Ok(position) => Ok(ActionResult::checked(FleetOutcome::Released(position))), + Err(Error::CellReleaseRefused { blocker, source }) => { + Ok(ActionResult::refused(blocker, *source)) + } + Err(error) => Err(error), + } + } + MovementAction::ReleaseMaintenance => { + let until = attempt + .reservation() + .map_or(spec.deadline_ms, |r| r.expires_at_ms.min(spec.deadline_ms)); + if now >= until { + return Ok(ActionResult::checked(FleetOutcome::Rejected( + DrainBlocker::Deadline, + ))); + } + let remaining = u64::try_from(until - now).map_err(|_| Error::Deadline)?; + let deadline = tokio::time::Instant::now() + .checked_add(std::time::Duration::from_millis(remaining)) + .ok_or(Error::Deadline)?; + match self + .runtime + .release_maintenance_cell_at( + spec.target.cell_id(), + spec.source, + spec.generation, + spec.incarnation, + spec.source_epoch, + deadline, + ) + .await? + { + cellule_runtime::cell::actor::MaintenanceCellRelease::Released(position) => { + Ok(ActionResult::checked(FleetOutcome::Released(position))) + } + cellule_runtime::cell::actor::MaintenanceCellRelease::Refused { + blocker, + error, + } => Ok(ActionResult { + outcome: FleetOutcome::Rejected(blocker), + error, + }), + } + } + MovementAction::Recover => self.recover(accepted, attempt).await, + MovementAction::Prepare => self.prepare(attempt).await, + MovementAction::Activate => self.activate(accepted, attempt).await, + MovementAction::Cancel => self.cancel(attempt).await, + MovementAction::Inspect => Err(Error::Control( + "fleet Inspect requires request-bound inspection", + )), + MovementAction::Retire => Err(Error::Control("fleet retirement is journal-local")), + } + } + + pub(super) async fn inspect_accepted( + &self, + accepted: &AcceptedFleetAction, + ) -> cellule_runtime::Result { + if matches!( + accepted.action().kind(), + FleetActionKind::Maintenance { .. } + ) { + return self.perform_maintenance(accepted); + } + let FleetActionKind::Movement { action, attempt } = accepted.action().kind() else { + return Err(Error::Control("accepted fleet effect is not movement")); + }; + // A runtime-owned Prepared state proves this session has not accepted + // takeover. Repeating preparation is not required; activation's exact + // credit state prevents a second accepted acquisition. + match action { + MovementAction::Recover => self.inspect_recovery(accepted, attempt).await, + MovementAction::Prepare => self.inspect_preparation(attempt), + MovementAction::Activate => { + if self.runtime.prepared_receiver(attempt.spec().id)?.is_none() + && self.confirmed_credit_settlement(attempt).await? + { + // Cleanup retired the receipt, but ordinary acquisition may + // already have served. Authority decides whether to inspect + // that actor or use the retained Idle input for acquisition. + return self.activate(accepted, attempt).await; + } + let prepared = self.prepared(attempt)?; + match prepared.state()? { + ReceiverState::Prepared => self.activate(accepted, attempt).await, + ReceiverState::Cancelled + if self.confirmed_credit_settlement(attempt).await? => + { + self.activate(accepted, attempt).await + } + ReceiverState::Activated => { + let basis = self + .journal + .load_acquisition_basis(accepted) + .await + .map_err(journal_error)? + .ok_or(Error::Peer( + "accepted activation has no retained acquisition basis", + ))?; + if basis.accepted() != accepted { + return Err(Error::Fenced); + } + let inputs = self.inputs(attempt).await?; + let outcome = self.serving(attempt, &inputs).await?; + let envelope = cellule_runtime::fleet::operations::FleetActionOutcome { + scope: accepted.action().scope(), + action_key: accepted.action().key().map_err(operation)?, + node: self.node, + session: self.session, + observed_at_ms: wall_time_ms()?, + outcome: outcome.clone(), + }; + basis.validate_result(&envelope).map_err(operation)?; + Ok(ActionResult::checked(outcome)) + } + _ => Ok(ActionResult::checked(FleetOutcome::Unknown)), + } + } + MovementAction::Cancel => self.cancel(attempt).await, + MovementAction::Inspect => Err(Error::Control( + "fleet Inspect requires request-bound inspection", + )), + MovementAction::Release + | MovementAction::ReleaseMaintenance + | MovementAction::Retire => Err(Error::Peer( + "accepted source effect has no provable retained result", + )), + } + } +} diff --git a/crates/cellule-host/src/fleet/movement/prefix.rs b/crates/cellule-host/src/fleet/movement/prefix.rs new file mode 100644 index 00000000..a46f3090 --- /dev/null +++ b/crates/cellule-host/src/fleet/movement/prefix.rs @@ -0,0 +1,105 @@ +//! Exact release or original sealed-suffix proof for fresh serving observations. +use super::*; + +impl FleetActionExecutor { + pub(super) async fn verify_serving_prefix( + &self, + attempt: &MoveAttempt, + inputs: &FleetCellInputs, + required: ServingPrefix<'_>, + root: cellule_runtime::ltx::RootRef, + ) -> cellule_runtime::Result<()> { + let spec = attempt.spec(); + match required { + ServingPrefix::Released(required) => { + self.runtime + .verify_root_prefix( + &inputs.catalog, + &inputs.authority, + inputs.replica.clone(), + required.to_ltx(spec.target.cell_id(), spec.incarnation), + root, + 10_000, + ) + .await?; + } + ServingPrefix::Recovered(recovery) => { + // Charge canonical acquisition metadata, the 2-MiB manifest + // envelope and decoded row/vector growth before provider I/O. + // Hold the token through the complete suffix/origin proof. + let _recovery_memory = self.runtime.try_reserve_node_bytes(8 << 20)?; + let original = recovery.basis().control(); + let restored = recovery.restored(); + let canonical = inputs + .authority + .acquisition_record(original.cell, original.incarnation, restored.epoch) + .await? + .ok_or(Error::AcquisitionHistoryIncomplete { + cell: original.cell, + incarnation: original.incarnation, + epoch: restored.epoch, + })?; + // Journal shape alone cannot prove native materialization. Bind + // its entire original input/result to canonical acquisition. + if canonical.input() != original || canonical.materialized() != restored { + return Err(Error::Control( + "journal recovery differs from canonical acquisition", + )); + } + if let Some(overlay) = &original.recovery { + // The overlay can survive interrupted earlier claims. Its + // original Cell epoch comes from the digest-verified sealed + // manifest, never from this later acquisition's epoch. + let stores = self.cells.recovery_inputs(spec).await.map_err(|source| { + Error::Facility { + name: "fleet-recovery-provider", + source, + } + })?; + let inventory = stores + .manifests + .load_manifest( + overlay.leader_session, + overlay.log_epoch, + overlay.manifest_digest, + ) + .await?; + let mut rows = inventory.cells().iter().filter(|row| { + row.application == spec.target.application() + && row.cell == original.cell + && row.incarnation == original.incarnation + && &row.recovery == overlay + }); + let suffix = rows.next().ok_or(Error::Control( + "recovered suffix is absent from its manifest", + ))?; + if rows.next().is_some() { + return Err(Error::Control("recovered suffix manifest is ambiguous")); + } + self.runtime + .verify_recovered_prefix( + &inputs.catalog, + &inputs.authority, + inputs.replica.clone(), + suffix, + root, + 10_000, + ) + .await?; + } else { + self.runtime + .verify_root_prefix( + &inputs.catalog, + &inputs.authority, + inputs.replica.clone(), + restored.ltx_root().ok_or(Error::Fenced)?, + root, + 10_000, + ) + .await?; + } + } + } + Ok(()) + } +} diff --git a/crates/cellule-host/src/fleet/movement/receiver.rs b/crates/cellule-host/src/fleet/movement/receiver.rs new file mode 100644 index 00000000..2407c041 --- /dev/null +++ b/crates/cellule-host/src/fleet/movement/receiver.rs @@ -0,0 +1,150 @@ +use super::*; + +impl FleetActionExecutor { + pub(super) async fn inputs( + &self, + attempt: &MoveAttempt, + ) -> cellule_runtime::Result { + let spec = attempt.spec(); + let inputs = self + .cells + .cell_inputs(spec) + .await + .map_err(|source| Error::Facility { + name: "fleet-cell-provider", + source, + })?; + // A verified catalog Cell id commits tenant/application identity. Check + // its explicit namespace and partition as well without constructing a + // second catalog proof or exposing the catalog's private target helper. + if inputs.catalog.entry().cell() != spec.target.cell_id() + || inputs.catalog.entry().namespace() != spec.target.namespace() + || inputs.catalog.entry().partition() != spec.target.partition() + || inputs.replica.scope() + != ( + *spec.target.cell_id().as_bytes(), + *spec.incarnation.as_bytes(), + ) + || inputs.owner.session != self.session + { + return Err(Error::Fenced); + } + Ok(inputs) + } + + pub(super) fn check_contract( + &self, + attempt: &MoveAttempt, + inputs: &FleetCellInputs, + control: &VersionedControl, + ) -> cellule_runtime::Result<()> { + let value = control.value(); + let spec = attempt.spec(); + if value.cell != spec.target.cell_id() || value.incarnation != spec.incarnation { + return Err(Error::Fenced); + } + if !self.registry.supports_cell( + spec.target.namespace(), + inputs.catalog.entry().role(), + value.code, + value.schema, + ) { + return Err(Error::Registry( + "fleet receiver does not support current Cell contract", + )); + } + Ok(()) + } + + pub(super) async fn prepare( + &self, + attempt: &MoveAttempt, + ) -> cellule_runtime::Result { + let prepared = async { + let inputs = self.inputs(attempt).await?; + let spec = attempt.spec(); + let current = inputs + .authority + .load(spec.target.cell_id()) + .await? + .ok_or(Error::Control("fleet source authority is absent"))?; + self.check_contract(attempt, &inputs, ¤t)?; + if current.value().epoch != spec.source_epoch + || current.value().state != ControlState::Serving + || current + .value() + .owner + .as_ref() + .is_none_or(|owner| owner.session != spec.source) + || current.value().root.is_none() + { + return Err(Error::Fenced); + } + let prepared = self.runtime.prepare_receiver( + spec.clone(), + inputs.catalog, + inputs.replica, + inputs.destination, + spec.deadline_ms, + wall_time_ms()?, + )?; + Ok(FleetOutcome::Reserved(prepared.reservation()?)) + } + .await; + // Preparation cannot change Cell authority. A definite refusal has no + // installed credit; partial runtime admission unwinds actual tokens. + Ok(match prepared { + Ok(outcome) => ActionResult::checked(outcome), + Err(error) => { + let blocker = match &error { + Error::Capacity(_) => DrainBlocker::ReceiverCapacity, + Error::Registry(_) => DrainBlocker::IncompatibleRelease, + Error::FleetOperation(e) + if matches!( + e.as_ref(), + cellule_runtime::fleet::operations::OperationError::Deadline + ) => + { + DrainBlocker::Deadline + } + _ => DrainBlocker::IncompleteObservation, + }; + ActionResult::refused(blocker, error) + } + }) + } + + pub(super) fn prepared( + &self, + attempt: &MoveAttempt, + ) -> cellule_runtime::Result { + let prepared = self + .runtime + .prepared_receiver(attempt.spec().id)? + .ok_or(Error::Peer("fleet receiver credit is absent"))?; + let reservation = prepared.reservation()?; + if prepared.spec()? != *attempt.spec() + || attempt + .reservation() + .is_some_and(|expected| reservation != expected) + { + return Err(Error::Fenced); + } + Ok(prepared) + } + + pub(super) fn inspect_preparation( + &self, + attempt: &MoveAttempt, + ) -> cellule_runtime::Result { + let prepared = self.prepared(attempt)?; + let outcome = match prepared.state()? { + ReceiverState::Prepared if prepared.reservation()?.expires_at_ms > wall_time_ms()? => { + FleetOutcome::Reserved(prepared.reservation()?) + } + ReceiverState::Prepared => FleetOutcome::Blocked(DrainBlocker::Deadline), + _ => FleetOutcome::Unknown, + }; + Ok(ActionResult::checked(outcome)) + } +} diff --git a/crates/cellule-host/src/fleet/movement/recovery.rs b/crates/cellule-host/src/fleet/movement/recovery.rs new file mode 100644 index 00000000..c1ce4b0a --- /dev/null +++ b/crates/cellule-host/src/fleet/movement/recovery.rs @@ -0,0 +1,238 @@ +use std::sync::Arc; + +use cellule_runtime::cell::actor::{AcquisitionObservation, AcquisitionObserver}; +use cellule_runtime::control::Control; +use cellule_runtime::fleet::operations::{RecoveredActivation, RecoveryBasis, RecoveryEvidence}; +use cellule_runtime::node::NodeTakeoverProof; + +use super::*; +use crate::fleet::FleetActionJournal; + +struct RecoveryRecorder { + accepted: AcceptedFleetAction, + takeover: NodeTakeoverProof, + journal: Arc, +} + +impl AcquisitionObserver for RecoveryRecorder { + fn before_claim<'a>(&'a self, input: &'a Control) -> AcquisitionObservation<'a> { + Box::pin(async move { + let basis = RecoveryBasis::new( + &self.accepted, + input.clone(), + self.takeover, + wall_time_ms()?, + ) + .map_err(operation)?; + let retained = self + .journal + .record_recovery_basis(&self.accepted, &basis) + .await + .map_err(journal_error)?; + retained + .validate_acceptance(&self.accepted) + .map_err(operation)?; + if retained.control() != input || retained.observed_at_ms() > basis.observed_at_ms() { + return Err(Error::Peer("journal changed checked recovery basis")); + } + Ok(()) + }) + } + fn before_activation<'a>( + &'a self, + input: &'a Control, + restored: &'a Control, + ) -> AcquisitionObservation<'a> { + Box::pin(async move { + let basis = self + .journal + .load_recovery_basis(&self.accepted) + .await + .map_err(journal_error)? + .ok_or(Error::Peer("recovery activation lacks retained input"))?; + basis + .validate_acceptance(&self.accepted) + .map_err(operation)?; + if basis.control() != input { + return Err(Error::Fenced); + } + let evidence = RecoveryEvidence::new(basis, restored.clone(), wall_time_ms()?) + .map_err(operation)?; + let retained = self + .journal + .record_recovery_evidence(&self.accepted, &evidence) + .await + .map_err(journal_error)?; + if retained.basis() != evidence.basis() + || retained.restored() != restored + || retained.recorded_at_ms() > evidence.recorded_at_ms() + { + return Err(Error::Peer("journal changed checked recovery result")); + } + Ok(()) + }) + } +} + +impl FleetActionExecutor { + pub(super) async fn recover( + &self, + accepted: &AcceptedFleetAction, + attempt: &MoveAttempt, + ) -> cellule_runtime::Result { + // Ordinary recovery must not compete with this attempt's unused affine + // worker reservation. Only the tracked Prepared/Cancelled state proves + // that no acquisition used it; never close an active writer for cleanup. + if self.runtime.prepared_receiver(attempt.spec().id)?.is_some() { + let credit = self.prepared(attempt)?; + match credit.state()? { + ReceiverState::Prepared | ReceiverState::Cancelled => { + self.runtime.cancel_prepared_receiver(&credit)? + } + _ => return Ok(ActionResult::checked(FleetOutcome::Unknown)), + } + } else if !self.confirmed_credit_settlement(attempt).await? { + return Err(Error::Peer("recovery lacks unused-credit cleanup proof")); + } + let inputs = self.inputs(attempt).await?; + let recovery = self + .cells + .recovery_inputs(attempt.spec()) + .await + .map_err(|source| Error::Facility { + name: "fleet-recovery-provider", + source, + })?; + let observed = inputs + .authority + .load(attempt.spec().target.cell_id()) + .await? + .ok_or(Error::Fenced)?; + self.check_contract(attempt, &inputs, &observed)?; + // Validate canonical fence/source binding before invoking any acquisition. + RecoveryBasis::new( + accepted, + observed.value().clone(), + recovery.takeover, + wall_time_ms()?, + ) + .map_err(operation)?; + let recorder: Arc = Arc::new(RecoveryRecorder { + accepted: accepted.clone(), + takeover: recovery.takeover, + journal: self.journal.clone(), + }); + if observed.value().state == ControlState::Idle { + self.runtime + .acquire_idle_restored_observed( + inputs.catalog.clone(), + inputs.replica.clone(), + inputs.authority.clone(), + observed, + inputs.destination.clone(), + inputs.owner.clone(), + Some(recorder), + ) + .await?; + } else { + self.runtime + .takeover_restored_observed( + inputs.catalog.clone(), + inputs.replica.clone(), + inputs.authority.clone(), + observed, + recovery.takeover, + recovery.manifests, + inputs.destination.clone(), + inputs.owner.clone(), + Some(recorder), + ) + .await?; + } + self.recovered_serving(accepted, attempt, &inputs).await + } + + pub(super) async fn inspect_recovery( + &self, + accepted: &AcceptedFleetAction, + attempt: &MoveAttempt, + ) -> cellule_runtime::Result { + accepted + .validate_replay(accepted.action(), self.node, self.session) + .map_err(operation)?; + let FleetActionKind::Movement { + action: MovementAction::Recover, + attempt: original, + } = accepted.action().kind() + else { + return Err(Error::Fenced); + }; + if original.spec() != attempt.spec() || accepted.action().scope() != self.scope { + return Err(Error::Fenced); + } + let inputs = self.inputs(attempt).await?; + if self + .journal + .load_recovery_evidence(accepted) + .await + .map_err(journal_error)? + .is_some() + { + // The required position is historical; serving is checked again below. + return self.recovered_serving(accepted, attempt, &inputs).await; + } + if let Some(basis) = self + .journal + .load_recovery_basis(accepted) + .await + .map_err(journal_error)? + { + basis.validate_acceptance(accepted).map_err(operation)?; + let current = inputs + .authority + .load(attempt.spec().target.cell_id()) + .await? + .ok_or(Error::Fenced)?; + if current.value() != basis.control() { + return Ok(ActionResult::checked(FleetOutcome::Unknown)); + } + // Unchanged full control (including monotonic revision/epoch) proves + // this retained input was not acquired; reconfirm it before CAS. + } + self.recover(accepted, attempt).await + } + + pub(super) async fn recovered_serving( + &self, + accepted: &AcceptedFleetAction, + attempt: &MoveAttempt, + inputs: &FleetCellInputs, + ) -> cellule_runtime::Result { + let recovery = self + .journal + .load_recovery_evidence(accepted) + .await + .map_err(journal_error)? + .ok_or(Error::Peer( + "current successor lacks retained recovery evidence", + ))?; + recovery + .basis() + .validate_acceptance(accepted) + .map_err(operation)?; + let serving = self + .serving_evidence(attempt, inputs, ServingPrefix::Recovered(&recovery)) + .await?; + let outcome = FleetOutcome::Recovered(Box::new(RecoveredActivation { recovery, serving })); + let envelope = cellule_runtime::fleet::operations::FleetActionOutcome { + scope: self.scope, + action_key: accepted.action().key().map_err(operation)?, + node: self.node, + session: self.session, + observed_at_ms: wall_time_ms()?, + outcome: outcome.clone(), + }; + accepted.validate_result(&envelope).map_err(operation)?; + Ok(ActionResult::checked(outcome)) + } +} diff --git a/crates/cellule-host/src/fleet/reader_evacuation/mod.rs b/crates/cellule-host/src/fleet/reader_evacuation/mod.rs new file mode 100644 index 00000000..288e69c9 --- /dev/null +++ b/crates/cellule-host/src/fleet/reader_evacuation/mod.rs @@ -0,0 +1,53 @@ +//! Durable reader replacement history and independent current confirmation. +use super::*; +use cellule_runtime::{ + fleet::operations::{OperationId, ReaderEvacuationPage, ReaderEvacuationRecord}, + identity::Digest, +}; + +mod publication; +mod refresh; +pub use refresh::FleetReaderEvacuationCandidate; +mod verification; +pub use publication::FleetReaderEvacuationPublication; +pub use verification::{FleetReaderEvacuationCheck, FleetReaderEvacuationVerifier}; + +/// Reader evidence transactions in the existing fleet journal domain. +/// Immutable manifests/pages and the latest per-operation/request pointer commit +/// together with registry advancement. Historical records never grant settlement. +pub trait FleetReaderEvacuationJournal: FleetJournal { + /// Atomically compares the full original head/registry, current Evacuating/Closing + /// operation, live controller, exact Retired original and Established + /// replacement rows/intents, then commits every checked page and manifest. + /// Advance the shared registry with the latest pointer. Exact duplicate + /// manifests return original history without advancing or restoring an old + /// pointer; ambiguous replies require lookup of that same digest first. + fn persist_reader_evacuation<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + record: &'a ReaderEvacuationRecord, + pages: &'a [ReaderEvacuationPage], + now_ms: i64, + ) -> FleetAdapterFuture<'a, ReaderEvacuationRecord>; + /// Loads immutable historical metadata, checking requested scope/digest. + fn load_reader_evacuation( + &self, + scope: cellule_runtime::fleet::operations::FleetScope, + digest: Digest, + ) -> FleetAdapterFuture<'_, Option>; + /// Loads an immutable page; its digest, basis and ordinal must still be + /// verified against the complete manifest before consuming any entries. + fn load_reader_evacuation_page( + &self, + scope: cellule_runtime::fleet::operations::FleetScope, + digest: Digest, + ) -> FleetAdapterFuture<'_, Option>; + /// Reads the latest committed witness for one original request at the full + /// expected snapshot. No filtered registry view or orphan PUT contributes. + fn latest_reader_evacuation<'a>( + &'a self, + expected: &'a FleetJournalSnapshot, + operation: OperationId, + original: Digest, + ) -> FleetAdapterFuture<'a, Option>; +} diff --git a/crates/cellule-host/src/fleet/reader_evacuation/publication.rs b/crates/cellule-host/src/fleet/reader_evacuation/publication.rs new file mode 100644 index 00000000..a06593ee --- /dev/null +++ b/crates/cellule-host/src/fleet/reader_evacuation/publication.rs @@ -0,0 +1,134 @@ +use super::verification::{adapter, bounded}; +use super::*; +use crate::read_replicas::ReaderEvacuation; +use cellule_runtime::{Error, Result}; +use std::sync::Arc; +use tokio::time::Instant; + +/// Durable response and independent final reader-policy confirmation. +pub struct FleetReaderEvacuationPublication { + record: std::result::Result>, + check: std::result::Result>, +} +impl FleetReaderEvacuationPublication { + /// Publishes the native capsule's exact immutable history after current + /// policy/readiness confirmation. Backend ownership survives cancellation; + /// lookup/adoption uses the same digest after an ambiguous reply. A final + /// failure retains the actual committed record and its separate source error. + pub async fn publish( + capture: &ReaderEvacuation, + journal: &dyn FleetReaderEvacuationJournal, + verifier: &FleetReaderEvacuationVerifier, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + let (record, pages) = capture.durable_record()?; + Self::publish_record( + capture.snapshot(), + &record, + &pages, + journal, + verifier, + deadline, + &mut clock, + ) + .await + } + /// Commits a refreshed policy candidate without repeating original native closure. + pub async fn publish_refreshed( + candidate: &FleetReaderEvacuationCandidate, + journal: &dyn FleetReaderEvacuationJournal, + verifier: &FleetReaderEvacuationVerifier, + deadline: Instant, + clock: impl FnMut() -> Result, + ) -> Result { + Self::publish_record( + candidate.snapshot(), + candidate.record(), + candidate.pages(), + journal, + verifier, + deadline, + clock, + ) + .await + } + #[allow(clippy::too_many_arguments)] + async fn publish_record( + snapshot: &FleetJournalSnapshot, + record: &ReaderEvacuationRecord, + pages: &[ReaderEvacuationPage], + journal: &dyn FleetReaderEvacuationJournal, + verifier: &FleetReaderEvacuationVerifier, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + if let Some(original) = bounded(deadline, async { + journal + .load_reader_evacuation( + record.retired().spec().scope, + record.digest().map_err(operation)?, + ) + .await + .map_err(adapter) + }) + .await? + { + if &original != record { + return Err(Error::Fenced); + } + let check = verifier + .recheck(journal, &original, deadline, &mut clock) + .await + .map_err(Arc::new); + return Ok(Self { + record: Ok(original), + check, + }); + } + let (started, mut last) = record.interval(); + let mut clock = || { + let next = clock()?; + if next < last || started < 0 || next - started > 30_000 { + return Err(Error::Deadline); + } + last = next; + Ok(next) + }; + verifier + .candidate(journal, record, pages, snapshot, deadline, &mut clock) + .await?; + let now = clock()?; + let returned = bounded(deadline, async { + let returned = journal + .persist_reader_evacuation(snapshot, record, pages, now) + .await + .map_err(adapter)?; + if &returned != record { + return Err(Error::Fenced); + } + Ok(returned) + }) + .await + .map_err(Arc::new); + let check = match &returned { + Ok(returned) => verifier + .recheck(journal, returned, deadline, &mut clock) + .await + .map_err(Arc::new), + Err(error) => Err(Arc::clone(error)), + }; + Ok(Self { + record: returned, + check, + }) + } + /// Original durable response or unchanged shared publication source error. + pub fn record(&self) -> std::result::Result<&ReaderEvacuationRecord, Arc> { + self.record.as_ref().map_err(Arc::clone) + } + /// Complete fresh post-publication confirmation or its original shared error. + pub fn confirmed(&self) -> std::result::Result<&FleetReaderEvacuationCheck, Arc> { + self.check.as_ref().map_err(Arc::clone) + } +} diff --git a/crates/cellule-host/src/fleet/reader_evacuation/refresh.rs b/crates/cellule-host/src/fleet/reader_evacuation/refresh.rs new file mode 100644 index 00000000..a897704e --- /dev/null +++ b/crates/cellule-host/src/fleet/reader_evacuation/refresh.rs @@ -0,0 +1,245 @@ +//! New immutable policy capture from a previously committed native closure. +use super::verification::{adapter, bounded}; +use super::*; +use crate::read_replicas::maintenance::{boot_identity, validate_replacement}; +use cellule_runtime::{ + Error, Result, + client::{CellDescription, Receipt}, + control::ControlState, + fleet::operations::{EnrollmentRole, MaintenancePhase, ReaderReplacementWitness}, +}; +use tokio::time::Instant; + +/// Fresh policy candidate retaining the same original native retirement. +/// This value starts no opening, refresh, closure or task. The normal live-owner +/// recruiter supplies replacements; publication commits only this metadata. +pub struct FleetReaderEvacuationCandidate { + snapshot: FleetJournalSnapshot, + record: ReaderEvacuationRecord, + pages: Vec, +} +impl FleetReaderEvacuationCandidate { + /// Original full barrier used for this new capture. + #[must_use] + pub fn snapshot(&self) -> &FleetJournalSnapshot { + &self.snapshot + } + /// New immutable policy history; original retirement timestamps stay intact. + #[must_use] + pub fn record(&self) -> &ReaderEvacuationRecord { + &self.record + } + /// Complete canonical pages for the new policy capture. + #[must_use] + pub fn pages(&self) -> &[ReaderEvacuationPage] { + &self.pages + } +} +impl FleetReaderEvacuationVerifier { + /// Reobserves replacements after policy/boot/owner changes or operation + /// deadline/session adoption. Requires the exact previously committed + /// native retirement and all its historical pages. It never repeats native + /// closure or treats an absent record as an empty responsibility. Closing + /// operations may restore current redundancy before terminal finalization. + pub async fn refresh( + &self, + journal: &dyn FleetReaderEvacuationJournal, + original: &ReaderEvacuationRecord, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + let started = clock()?; + let mut last = started; + let mut clock = || { + let next = clock()?; + if started < 0 || next < last || next - started > 30_000 { + return Err(Error::Deadline); + } + last = next; + Ok(next) + }; + bounded(deadline, async { + let scope = original.retired().spec().scope; + if journal + .load_reader_evacuation(scope, original.digest().map_err(operation)?) + .await + .map_err(adapter)? + .as_ref() + != Some(original) + { + return Err(Error::Fenced); + } + let mut historical = Vec::with_capacity(original.pages().len()); + for digest in original.pages() { + historical.push( + journal + .load_reader_evacuation_page(scope, *digest) + .await + .map_err(adapter)? + .ok_or(Error::Fenced)?, + ); + } + original.validate_pages(&historical).map_err(operation)?; + let snapshot = journal.load_snapshot(scope).await.map_err(adapter)?; + let maintenance = snapshot.head().maintenance().ok_or(Error::Fenced)?; + if maintenance.id() != original.operation().id() + || maintenance.node() != original.operation().node() + || !matches!( + maintenance.phase(), + MaintenancePhase::Evacuating | MaintenancePhase::Closing + ) + { + return Err(Error::Fenced); + } + let roster = FleetRoster::collect(journal, &snapshot, deadline).await?; + if !roster + .enrollments() + .iter() + .any(|row| row == original.retired()) + { + return Err(Error::Fenced); + } + let EnrollmentRole::Reader { target, position } = &original.retired().spec().role + else { + return Err(Error::Fenced); + }; + let authority = self + .authority + .load(target.cell_id()) + .await? + .ok_or(Error::Fenced)?; + let authority = authority.value(); + if authority.state != ControlState::Serving + || authority.recovery.is_some() + || authority.incarnation != position.incarnation + || authority + .root + .as_ref() + .is_none_or(|root| root.commit_sequence < original.minimum_sequence()) + { + return Err(Error::Fenced); + } + let policy = self + .policy + .load(authority.cell) + .await? + .map(|row| row.value()); + if policy.is_some_and(|policy| policy.incarnation() != authority.incarnation) { + return Err(Error::Fenced); + } + let desired = policy.map_or(0, |policy| policy.desired_readers()); + let minimum = Receipt { + cell: authority.cell, + incarnation: authority.incarnation, + commit_sequence: original.minimum_sequence().max( + authority + .root + .as_ref() + .ok_or(Error::Fenced)? + .commit_sequence, + ), + }; + let selected = self + .directory + .select_readers( + minimum.cell, + authority.owner.as_ref().ok_or(Error::Fenced)?.session, + authority.code, + usize::from(desired), + clock()?, + 10_000, + ) + .await?; + if selected.len() != usize::from(desired) { + return Err(Error::ReplicaUnavailable); + } + let expected = CellDescription { + cell: minimum.cell, + incarnation: minimum.incarnation, + code: authority.code, + schema: authority.schema, + }; + let mut replacements = Vec::with_capacity(selected.len()); + for node in selected { + if node.node() == maintenance.node() { + return Err(Error::Fenced); + } + let (enrollment_key, enrollment_digest) = + validate_replacement(&roster, &node, target, minimum)?; + let (receipt, ready) = self.peer.status(target, node.clone(), expected).await?; + if !ready + || receipt.cell != minimum.cell + || receipt.incarnation != minimum.incarnation + || receipt.commit_sequence < minimum.commit_sequence + { + return Err(Error::ReplicaUnavailable); + } + replacements.push(ReaderReplacementWitness { + node: node.node(), + session: node.session(), + boot_identity: boot_identity(&node)?, + enrollment_key, + enrollment_digest, + commit_sequence: receipt.commit_sequence, + }); + } + roster.confirm(journal, deadline).await?; + let (record, pages) = ReaderEvacuationRecord::new( + maintenance.clone(), + ( + Digest::from_bytes( + *blake3::hash(&snapshot.head().to_bytes().map_err(operation)?).as_bytes(), + ), + snapshot.registry(), + ), + original.retired().clone(), + original.original_digest(), + authority.clone(), + policy.map(|policy| policy.revision()), + desired, + minimum.commit_sequence, + (started, clock()?), + replacements, + ) + .map_err(operation)?; + let check = self + .candidate(journal, &record, &pages, &snapshot, deadline, &mut clock) + .await?; + let (record, pages) = ReaderEvacuationRecord::new( + maintenance.clone(), + ( + Digest::from_bytes( + *blake3::hash(&snapshot.head().to_bytes().map_err(operation)?).as_bytes(), + ), + snapshot.registry(), + ), + original.retired().clone(), + original.original_digest(), + check.authority().clone(), + policy.map(|policy| policy.revision()), + desired, + minimum.commit_sequence, + (started, clock()?), + check + .replacements() + .iter() + .map(|entry| ReaderReplacementWitness { + node: entry.node, + session: entry.session, + boot_identity: entry.boot_identity, + enrollment_key: entry.enrollment_key, + enrollment_digest: entry.enrollment_digest, + commit_sequence: entry.receipt.commit_sequence, + }) + .collect(), + ) + .map_err(operation)?; + Ok(FleetReaderEvacuationCandidate { + snapshot, + record, + pages, + }) + }) + .await + } +} diff --git a/crates/cellule-host/src/fleet/reader_evacuation/verification.rs b/crates/cellule-host/src/fleet/reader_evacuation/verification.rs new file mode 100644 index 00000000..f014ecd1 --- /dev/null +++ b/crates/cellule-host/src/fleet/reader_evacuation/verification.rs @@ -0,0 +1,379 @@ +use super::*; +use crate::read_replicas::{ + ReaderReplacement, + maintenance::{boot_identity, same_authority, validate_replacement}, +}; +use cellule_runtime::{ + Error, Result, + client::{CellDescription, Receipt}, + control::{Control, authority::CellAuthority}, + fleet::operations::{EnrollmentRole, MaintenancePhase}, + node::{NodeDirectory, NodeMode}, + peer::ReplicaPeerClient, + read_policy::ReadPolicyStore, +}; +use tokio::time::{Instant, timeout_at}; + +/// Canonical readers of current authority, policy, signed boots and native status. +/// It starts no role effect or task; the application authenticates peer routing +/// and accounts bounded metadata and its existing accepted backend work. +#[derive(Clone)] +pub struct FleetReaderEvacuationVerifier { + pub(super) directory: NodeDirectory, + pub(super) authority: CellAuthority, + pub(super) policy: ReadPolicyStore, + pub(super) peer: ReplicaPeerClient, +} +impl FleetReaderEvacuationVerifier { + /// Binds the existing canonical handles and authenticated native peer client. + #[must_use] + pub fn new( + directory: NodeDirectory, + authority: CellAuthority, + policy: ReadPolicyStore, + peer: ReplicaPeerClient, + ) -> Self { + Self { + directory, + authority, + policy, + peer, + } + } + /// Reloads the latest durable manifest and every exact page, then confirms + /// current policy, authority, enrollment, selected boots and native prefixes + /// twice around the full roster barrier. Historical times stay unchanged; + /// this result has a separate fresh interval and grants no node finalization. + pub async fn recheck( + &self, + journal: &dyn FleetReaderEvacuationJournal, + record: &ReaderEvacuationRecord, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + let scope = record.retired().spec().scope; + let started = clock()?; + let mut last = started; + let mut clock = || { + let next = clock()?; + if next < last || started < 0 || next - started > 30_000 { + return Err(Error::Deadline); + } + last = next; + Ok(next) + }; + bounded(deadline, async { + let snapshot = journal.load_snapshot(scope).await.map_err(adapter)?; + if journal + .latest_reader_evacuation( + &snapshot, + record.operation().id(), + record.retired().spec().key().map_err(operation)?, + ) + .await + .map_err(adapter)? + .as_ref() + != Some(record) + { + return Err(Error::Fenced); + } + let mut pages = Vec::with_capacity(record.pages().len()); + for digest in record.pages() { + pages.push( + journal + .load_reader_evacuation_page(scope, *digest) + .await + .map_err(adapter)? + .ok_or(Error::Fenced)?, + ); + } + self.confirm( + journal, record, &pages, &snapshot, deadline, &mut clock, started, + ) + .await + }) + .await + } + /// Confirms a newly constructed native candidate before its first durable + /// publication. The caller still owns its original collection interval. + pub(super) async fn candidate( + &self, + journal: &dyn FleetReaderEvacuationJournal, + record: &ReaderEvacuationRecord, + pages: &[ReaderEvacuationPage], + snapshot: &FleetJournalSnapshot, + deadline: Instant, + clock: &mut impl FnMut() -> Result, + ) -> Result { + let started = clock()?; + bounded( + deadline, + self.confirm(journal, record, pages, snapshot, deadline, clock, started), + ) + .await + } + #[allow(clippy::too_many_arguments)] + async fn confirm( + &self, + journal: &dyn FleetReaderEvacuationJournal, + record: &ReaderEvacuationRecord, + pages: &[ReaderEvacuationPage], + snapshot: &FleetJournalSnapshot, + deadline: Instant, + clock: &mut impl FnMut() -> Result, + started: i64, + ) -> Result { + record.validate_pages(pages).map_err(operation)?; + let operation_record = record.operation(); + let scope = record.retired().spec().scope; + let current = snapshot.head().maintenance().ok_or(Error::Fenced)?; + if self.directory.fleet() != scope.fleet + || current.id() != operation_record.id() + || current.node() != operation_record.node() + || current.session() != operation_record.session() + || current.intent_revision() != operation_record.intent_revision() + || current.deadline_ms() != operation_record.deadline_ms() + || !matches!( + current.phase(), + MaintenancePhase::Evacuating | MaintenancePhase::Closing + ) + || snapshot.registry().bootstrap_revision().is_none() + { + return Err(Error::Fenced); + } + let roster = FleetRoster::collect(journal, snapshot, deadline).await?; + let intent = roster + .intents() + .iter() + .find(|intent| intent.node() == current.node()) + .ok_or(Error::Fenced)?; + if intent.session() != current.session() + || intent.revision() != current.intent_revision() + || intent.mode() != NodeMode::Draining + || !roster + .enrollments() + .iter() + .any(|row| row == record.retired()) + { + return Err(Error::Fenced); + } + let EnrollmentRole::Reader { target, position } = &record.retired().spec().role else { + return Err(Error::Fenced); + }; + let minimum = Receipt { + cell: target.cell_id(), + incarnation: position.incarnation, + commit_sequence: record.minimum_sequence(), + }; + let expected = CellDescription { + cell: minimum.cell, + incarnation: minimum.incarnation, + code: record.authority().code, + schema: record.authority().schema, + }; + let witnesses: Vec<_> = pages.iter().flat_map(|page| page.entries()).collect(); + let mut replacements = Vec::with_capacity(witnesses.len()); + let mut authority = record.authority().clone(); + for _ in 0..2 { + let now = clock()?; + if now >= current.deadline_ms() { + return Err(Error::Deadline); + } + let observed = self + .authority + .load(minimum.cell) + .await? + .ok_or(Error::Fenced)?; + let fresh = observed.value(); + if !same_authority(&authority, fresh) { + return Err(Error::Fenced); + } + let policy = self.policy.load(minimum.cell).await?.map(|row| row.value()); + if policy.map(|policy| policy.revision()) != record.policy_revision() + || policy.map_or(0, |policy| policy.desired_readers()) != record.desired_readers() + || policy.is_some_and(|policy| { + policy.cell() != minimum.cell || policy.incarnation() != minimum.incarnation + }) + { + return Err(Error::Fenced); + } + let selected = self + .directory + .select_readers( + minimum.cell, + fresh.owner.as_ref().ok_or(Error::Fenced)?.session, + fresh.code, + witnesses.len(), + now, + 10_000, + ) + .await?; + if selected.len() != witnesses.len() { + return Err(Error::ReplicaUnavailable); + } + replacements.clear(); + for selected in selected { + let witness = witnesses + .iter() + .find(|entry| { + entry.node == selected.node() && entry.session == selected.session() + }) + .ok_or(Error::Fenced)?; + let node = self + .directory + .load(witness.session, clock()?) + .await? + .ok_or(Error::Fenced)?; + let node = node.advertisement(); + if node.node() != witness.node + || boot_identity(node)? != witness.boot_identity + || validate_replacement(&roster, node, target, minimum)? + != (witness.enrollment_key, witness.enrollment_digest) + { + return Err(Error::Fenced); + } + let (receipt, ready) = self.peer.status(target, node.clone(), expected).await?; + if !ready + || receipt.cell != minimum.cell + || receipt.incarnation != minimum.incarnation + || receipt.commit_sequence + < minimum.commit_sequence.max(witness.commit_sequence) + { + return Err(Error::ReplicaUnavailable); + } + replacements.push(ReaderReplacement { + node: witness.node, + session: witness.session, + boot_identity: witness.boot_identity, + enrollment_key: witness.enrollment_key, + enrollment_digest: witness.enrollment_digest, + receipt, + }); + } + authority = fresh.clone(); + // Probes can suspend; no authority/policy change is hidden behind + // their replies or a stable native topology cursor. + let after = self + .authority + .load(minimum.cell) + .await? + .ok_or(Error::Fenced)?; + if !same_authority(&authority, after.value()) + || self.policy.load(minimum.cell).await?.map(|row| row.value()) != policy + { + return Err(Error::Fenced); + } + authority = after.value().clone(); + let selected = self + .directory + .select_readers( + minimum.cell, + authority.owner.as_ref().ok_or(Error::Fenced)?.session, + authority.code, + witnesses.len(), + clock()?, + 10_000, + ) + .await?; + if selected.len() != witnesses.len() { + return Err(Error::Fenced); + } + for node in selected { + let witness = witnesses + .iter() + .find(|entry| entry.node == node.node() && entry.session == node.session()) + .ok_or(Error::Fenced)?; + let fresh = self + .directory + .load(witness.session, clock()?) + .await? + .ok_or(Error::Fenced)?; + let fresh = fresh.advertisement(); + if fresh.node() != witness.node + || boot_identity(fresh)? != witness.boot_identity + || validate_replacement(&roster, fresh, target, minimum)? + != (witness.enrollment_key, witness.enrollment_digest) + { + return Err(Error::Fenced); + } + } + roster.confirm(journal, deadline).await?; + } + let finished = clock()?; + if finished < started || finished - started > 30_000 || finished >= current.deadline_ms() { + return Err(Error::Deadline); + } + Ok(FleetReaderEvacuationCheck { + snapshot: snapshot.clone(), + record: record.digest().map_err(operation)?, + roster_digest: roster.digest()?, + authority, + replacements, + started_at_ms: started, + finished_at_ms: finished, + }) + } +} + +/// Fresh confirmation of one persisted reader obligation; other roles remain. +pub struct FleetReaderEvacuationCheck { + snapshot: FleetJournalSnapshot, + record: Digest, + roster_digest: Digest, + authority: Control, + replacements: Vec, + started_at_ms: i64, + finished_at_ms: i64, +} +impl FleetReaderEvacuationCheck { + /// Exact final journal barrier; compare it again before dependent actions. + #[must_use] + pub fn snapshot(&self) -> &FleetJournalSnapshot { + &self.snapshot + } + /// Immutable durable history confirmed by this fresh observation. + #[must_use] + pub const fn record_digest(&self) -> Digest { + self.record + } + /// Full roster identity, including original terminal and Pending rows. + #[must_use] + pub const fn roster_digest(&self) -> Digest { + self.roster_digest + } + /// Current serving authority, without ownership rights. + #[must_use] + pub fn authority(&self) -> &Control { + &self.authority + } + /// Actual selected/probed replacements and their nonregressing prefixes. + #[must_use] + pub fn replacements(&self) -> &[ReaderReplacement] { + &self.replacements + } + /// Separate fresh interval; the persisted record keeps its original times. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + (self.started_at_ms, self.finished_at_ms) + } +} +pub(super) fn adapter(source: Box) -> Error { + Error::Facility { + name: "fleet-reader-evacuation-journal", + source, + } +} +pub(super) async fn bounded( + deadline: Instant, + future: impl std::future::Future>, +) -> Result { + if Instant::now() >= deadline { + return Err(Error::Deadline); + } + timeout_at(deadline, future) + .await + .map_err(|source| Error::Facility { + name: "fleet-reader-evacuation-deadline", + source: Box::new(source), + })? +} diff --git a/crates/cellule-host/src/fleet/reconciler/maintenance.rs b/crates/cellule-host/src/fleet/reconciler/maintenance.rs new file mode 100644 index 00000000..81f6525f --- /dev/null +++ b/crates/cellule-host/src/fleet/reconciler/maintenance.rs @@ -0,0 +1,88 @@ +use super::*; +use cellule_runtime::fleet::operations::{FleetOutcome, MaintenanceAction, MaintenanceEvent}; + +impl FleetReconciler { + pub(super) async fn advance_maintenance( + &self, + clock: &PassClock<'_>, + report: &mut FleetReconcileReport, + ) -> Result<()> { + let Some(maintenance) = report.snapshot.head().maintenance().cloned() else { + return Ok(()); + }; + match maintenance.phase() { + MaintenancePhase::Requested => { + // Cordon is monotonic even after the evacuation deadline. A + // missing reply cannot clear the retained physical-node intent. + // Replaying its stable key adopts the original acceptance. + let action = report + .snapshot + .head() + .maintenance_action(MaintenanceAction::Cordon, clock.now()?) + .map_err(operation)?; + report.dispatched += 1; + let completion = call( + clock.deadline, + "fleet-transport", + self.transport.dispatch(&action, clock.deadline), + ) + .await?; + completion + .accepted + .validate_replay(&action, maintenance.node(), maintenance.session()) + .map_err(operation)?; + completion + .accepted + .validate_result(&completion.outcome) + .map_err(operation)?; + report.maintenance_failure = completion.execution_error.clone(); + if !completion.committed { + if let Some(error) = &completion.journal_error { + report.maintenance_failure = Some(Arc::clone(error)); + } + report.blocked(DrainBlocker::PendingPublication); + return Ok(()); + } + match completion.outcome.outcome { + FleetOutcome::Cordoned => { + self.commit( + clock, + report, + JournalTransition::Maintenance(MaintenanceEvent::Cordoned), + ) + .await?; + } + FleetOutcome::Rejected(blocker) | FleetOutcome::Blocked(blocker) => { + report.blocked(blocker); + } + FleetOutcome::Unknown => { + report.blocked(DrainBlocker::OutcomeUnknown); + } + _ => return Err(operation(OperationError::Conflict)), + } + } + MaintenancePhase::Cordoned => { + // Only a previously committed endpoint proof reaches this phase. + // Registry intent already excludes this node from receiving. + self.commit( + clock, + report, + JournalTransition::Maintenance(MaintenanceEvent::BeginEvacuation), + ) + .await?; + } + MaintenancePhase::Evacuating | MaintenancePhase::Closing => { + // Full role inventory and joined facility/withdrawal evidence + // are still required. Empty movement permits cannot prove them. + report.blocked(DrainBlocker::IncompleteObservation); + } + MaintenancePhase::Completed => {} + } + if maintenance.phase() != MaintenancePhase::Completed + && clock.now()? >= maintenance.deadline_ms() + { + report.blocked(DrainBlocker::Deadline); + } + Ok(()) + } +} diff --git a/crates/cellule-host/src/fleet/reconciler/mod.rs b/crates/cellule-host/src/fleet/reconciler/mod.rs new file mode 100644 index 00000000..bf3701e7 --- /dev/null +++ b/crates/cellule-host/src/fleet/reconciler/mod.rs @@ -0,0 +1,431 @@ +//! Bounded application-driven reconciliation. Every effect follows journal CAS. + +use std::{future::Future, sync::Arc, time::Duration}; + +use cellule_runtime::fleet::operations::AttemptId; +use cellule_runtime::fleet::operations::{ + DrainBlocker, FleetAction, FleetInspectionObservation, FleetInspectionRequest, FleetProfile, + FleetScope, JournalTransition, MaintenancePhase, OperationError, +}; +use cellule_runtime::identity::SessionId; +use cellule_runtime::{Error, Result}; +use tokio::time::{Instant, timeout_at}; + +use super::actions::operation; +use super::{ + FleetActionCompletion, FleetAdapterFuture, FleetJournal, FleetJournalSnapshot, FleetRoster, +}; + +mod maintenance; +mod movement; +mod observation; +mod planning; +mod successor; +pub use observation::{FleetObservation, FleetOwnedCell}; + +/// Application-owned complete roster and authenticated paginated observation. +/// +/// Implementations compare membership and registry revisions before and after +/// collecting pages, pin boot identities and signing keys, and include busy and +/// transitioning ownership. `complete` must not be inferred from a filtered live +/// directory. Bounded pages retain their original sample times. HTTP and product +/// authorization remain in the application. +pub trait FleetObserver: Send + Sync + 'static { + /// Captures advisory inputs for the complete retained roster supplied by + /// the reconciler, which rechecks the roster after capture. Complete + /// coverage is required for count balancing; partial pressure observations + /// still require authenticated source and receiver evidence. + fn observe<'a>( + &'a self, + roster: &'a FleetRoster, + now_ms: i64, + deadline: Instant, + ) -> FleetAdapterFuture<'a, FleetObservation>; +} + +/// Authenticated management transport. A timeout never means definite refusal. +pub trait FleetTransport: Send + Sync + 'static { + /// Dispatches to the exact action endpoint. Dropping this waiter does not + /// cancel node-owned work or free its journal permit. + fn dispatch<'a>( + &'a self, + action: &'a FleetAction, + deadline: Instant, + ) -> FleetAdapterFuture<'a, Arc>; + + /// Captures current authority and actor evidence for the entire request. + /// Cached effect receipts cannot satisfy this boundary. + fn inspect<'a>( + &'a self, + request: &'a FleetInspectionRequest, + deadline: Instant, + ) -> FleetAdapterFuture<'a, Arc>; +} + +/// One endpoint failure retained without suppressing healthy sibling progress. +#[derive(Debug)] +pub struct FleetAttemptFailure { + /// Exact still-charged attempt whose endpoint call could not be confirmed. + pub attempt: AttemptId, + /// Original source chain, including transport and waiter timeout errors. + pub error: Arc, +} + +/// Bounded progress from one pass; detailed adapter errors preserve their source. +#[derive(Debug)] +pub struct FleetReconcileReport { + /// Latest committed snapshot observed by this pass. + pub snapshot: FleetJournalSnapshot, + /// New journal-backed permits allocated across all donors. + pub allocated: usize, + /// Effect waiters started; their effects remain owned by target nodes. + pub dispatched: usize, + /// Fresh current-state captures consumed. + pub inspected: usize, + /// Permits atomically retired with immutable history. + pub retired: usize, + /// Clean source releases newly committed during this pass. + pub released: usize, + /// Fresh successor activations newly committed during this pass. + pub activated: usize, + /// Canonical failed-source recovery completions newly committed this pass. + pub recovered: usize, + /// Proven pre-release cancellations newly committed during this pass. + pub cancelled: usize, + /// Bounded distinct conditions preventing further optional work. + pub blockers: Vec, + /// Endpoint failures retained independently of healthy sibling progress. + /// At most one entry per previously charged attempt; permits remain charged. + pub failures: Vec, + /// Original maintenance endpoint error, independent of movement failures. + /// Its durable intent and phase remain retained for a later pass. + pub maintenance_failure: Option>, + /// Suggested logical wake time; an application event may wake sooner. + pub next_wake_at_ms: i64, +} + +impl FleetReconcileReport { + fn blocked(&mut self, reason: DrainBlocker) { + if !self.blockers.contains(&reason) { + self.blockers.push(reason); + } + } +} + +/// One caller-driven controller facade; it starts no scheduler or runtime. +/// +/// The path cordons maintenance nodes, executes settled movement, and invokes +/// explicit busy maintenance release. Node role evacuation and finalization +/// require their host barriers; an unfinished maintenance operation +/// remains visible and cannot be reported complete by this facade. +pub struct FleetReconciler { + scope: FleetScope, + claimant: SessionId, + profile: FleetProfile, + journal: Arc, + observer: Arc, + transport: Arc, +} + +impl FleetReconciler { + /// Validates scope, identity and bounds before any adapter call or task. + pub fn new( + scope: FleetScope, + claimant: SessionId, + profile: FleetProfile, + journal: Arc, + observer: Arc, + transport: Arc, + ) -> Result { + let profile = profile.validate().map_err(operation)?; + cellule_runtime::fleet::operations::FleetHead::new(scope, 0).map_err(operation)?; + if claimant.as_bytes().iter().all(|byte| *byte == 0) { + return Err(operation(OperationError::Invalid( + "zero controller claimant", + ))); + } + Ok(Self { + scope, + claimant, + profile, + journal, + observer, + transport, + }) + } + + /// Reconciles every existing permit before choosing new movement. Each pass + /// performs at most one movement step per previously charged attempt and + /// allocates no more than the shared profile allows. The supplied clock reads + /// the same logical domain as observations and node evidence at each boundary. + /// It must be nonnegative and never regress during a pass. + /// A stopped scheduling policy still permits settling accepted work. + pub async fn reconcile_once( + &self, + now: impl Fn() -> Result + Send + Sync, + deadline: Instant, + ) -> Result { + let clock = PassClock::new(&now, deadline)?; + let now_ms = clock.now()?; + let initial = call( + deadline, + "fleet-journal", + self.journal.load_snapshot(self.scope), + ) + .await?; + if initial.head().scope() != self.scope { + return Err(operation(OperationError::Conflict)); + } + let snapshot = call( + deadline, + "fleet-journal", + self.journal.claim_controller( + self.scope, + initial.head().revision(), + self.claimant, + clock.now()?, + ), + ) + .await?; + let claimed_epoch = self.controller_epoch(&snapshot, clock.now()?)?; + let mut report = FleetReconcileReport { + snapshot, + allocated: 0, + dispatched: 0, + inspected: 0, + retired: 0, + released: 0, + activated: 0, + recovered: 0, + cancelled: 0, + blockers: Vec::new(), + failures: Vec::new(), + maintenance_failure: None, + next_wake_at_ms: now_ms + .checked_add(self.profile.reconcile_interval_ms) + .ok_or(Error::Control("fleet wake time overflow"))?, + }; + let claimed_revision = report.snapshot.head().revision(); + let ids = report + .snapshot + .head() + .attempts() + .iter() + .map(|a| a.spec().id) + .collect::>(); + let count = ids.len(); + for (index, id) in ids.into_iter().enumerate() { + if !report + .snapshot + .head() + .attempts() + .iter() + .any(|attempt| attempt.spec().id == id) + { + continue; + } + let step_clock = clock.partition(count - index + 1); + if let Err(error) = self.advance(id, &step_clock, &mut report).await { + let timed_out = matches!( + &error, + Error::Facility { + name: "fleet-controller-deadline", + .. + } + ); + let endpoint_failed = matches!( + &error, + Error::Facility { + name: "fleet-transport", + .. + } + ); + if !timed_out && !endpoint_failed { + return Err(error); + } + if timed_out { + // A journal CAS may have committed after its waiter expired. + // Re-read at the outer deadline before any dependent action. + let snapshot = call( + clock.deadline, + "fleet-journal", + self.journal.load_snapshot(self.scope), + ) + .await?; + if self.controller_epoch(&snapshot, clock.now()?)? != claimed_epoch { + return Err(operation(OperationError::Fenced)); + } + report.snapshot = snapshot; + } + report.blocked(DrainBlocker::OutcomeUnknown); + report.failures.push(FleetAttemptFailure { + attempt: id, + error: Arc::new(error), + }); + } + } + if let Err(error) = self + .advance_maintenance(&clock.partition(2), &mut report) + .await + { + let timed_out = matches!( + &error, + Error::Facility { + name: "fleet-controller-deadline", + .. + } + ); + let endpoint_failed = matches!( + &error, + Error::Facility { + name: "fleet-transport", + .. + } + ); + if !timed_out && !endpoint_failed { + return Err(error); + } + if timed_out { + // Acceptance or phase publication may have outlived its waiter. + let snapshot = call( + clock.deadline, + "fleet-journal", + self.journal.load_snapshot(self.scope), + ) + .await?; + if self.controller_epoch(&snapshot, clock.now()?)? != claimed_epoch { + return Err(operation(OperationError::Fenced)); + } + report.snapshot = snapshot; + } + report.blocked(DrainBlocker::OutcomeUnknown); + report.maintenance_failure = Some(Arc::new(error)); + } + if report.snapshot.registry().scheduling_enabled() { + self.plan(&clock, &mut report).await?; + } + // Proven cancellation keeps the periodic retry interval, avoiding an + // immediate allocation/refusal loop when a receiver cannot admit work. + if report.snapshot.head().revision() != claimed_revision + && report.cancelled == 0 + && report.failures.is_empty() + && report.maintenance_failure.is_none() + { + // A multi-phase move must not spend one periodic interval between + // every action and expire its admission deadline before release. + report.next_wake_at_ms = clock.now()?; + } + Ok(report) + } + + async fn commit( + &self, + clock: &PassClock<'_>, + report: &mut FleetReconcileReport, + transition: JournalTransition, + ) -> Result<()> { + let now_ms = clock.now()?; + let epoch = self.controller_epoch(&report.snapshot, now_ms)?; + report.snapshot = call( + clock.deadline, + "fleet-journal", + self.journal + .compare_exchange(&report.snapshot, epoch, now_ms, &transition), + ) + .await?; + Ok(()) + } + + fn controller_epoch(&self, snapshot: &FleetJournalSnapshot, now_ms: i64) -> Result { + if snapshot.head().scope() != self.scope { + return Err(operation(OperationError::Conflict)); + } + let lease = snapshot + .head() + .controller() + .ok_or_else(|| operation(OperationError::Fenced))?; + if lease.claimant != self.claimant || now_ms >= lease.expires_at_ms { + return Err(operation(OperationError::Fenced)); + } + Ok(lease.epoch) + } +} + +struct PassClock<'a> { + read: &'a (dyn Fn() -> Result + Send + Sync), + deadline: Instant, + last_ms: Arc, +} +impl<'a> PassClock<'a> { + fn new(read: &'a (dyn Fn() -> Result + Send + Sync), deadline: Instant) -> Result { + let now_ms = read()?; + if now_ms < 0 { + return Err(operation(OperationError::Invalid( + "negative controller time", + ))); + } + if deadline <= Instant::now() { + return Err(operation(OperationError::Deadline)); + } + Ok(Self { + read, + deadline, + last_ms: Arc::new(std::sync::atomic::AtomicI64::new(now_ms)), + }) + } + fn now(&self) -> Result { + let now_ms = (self.read)()?; + let previous = self + .last_ms + .fetch_max(now_ms, std::sync::atomic::Ordering::SeqCst); + if now_ms < previous { + return Err(Error::Control("fleet controller clock regressed")); + } + Ok(now_ms) + } + fn partition(&self, units: usize) -> Self { + // FleetHead bounds attempts to two, and one share remains for planning. + let now = Instant::now(); + let share = self.deadline.saturating_duration_since(now) / units as u32; + Self { + read: self.read, + last_ms: Arc::clone(&self.last_ms), + deadline: now + share, + } + } + fn capture_deadline_ms(&self) -> Result { + // No inspection may outlive this pass or its controller lease. + let remaining = self + .deadline + .saturating_duration_since(Instant::now()) + .min(Duration::from_secs(30)); + self.now()? + .checked_add(i64::try_from(remaining.as_millis()).map_err(|source| { + Error::Facility { + name: "fleet-controller-clock", + source: Box::new(source), + } + })?) + .ok_or(Error::Control("fleet inspection deadline overflow")) + } +} + +async fn call( + deadline: Instant, + name: &'static str, + future: impl Future>>, +) -> Result { + timeout_at(deadline, future) + .await + .map_err(|source| Error::Facility { + name: "fleet-controller-deadline", + source: Box::new(source), + })? + .map_err(|source| Error::Facility { name, source }) +} + +fn nonce() -> cellule_runtime::identity::Digest { + cellule_runtime::identity::Digest::from_bytes( + *blake3::hash(uuid::Uuid::now_v7().as_bytes()).as_bytes(), + ) +} diff --git a/crates/cellule-host/src/fleet/reconciler/movement.rs b/crates/cellule-host/src/fleet/reconciler/movement.rs new file mode 100644 index 00000000..2cfcee57 --- /dev/null +++ b/crates/cellule-host/src/fleet/reconciler/movement.rs @@ -0,0 +1,558 @@ +use super::*; +use cellule_runtime::fleet::operations::{ + AttemptEvent, AttemptId, AttemptPhase, FleetActionOutcome, FleetOutcome, MoveAttempt, + MovementAction, +}; +use cellule_runtime::identity::{NodeId, SessionId}; + +impl FleetReconciler { + pub(super) async fn advance( + &self, + id: AttemptId, + clock: &PassClock<'_>, + report: &mut FleetReconcileReport, + ) -> Result<()> { + let attempt = current(report, id)?.clone(); + let mut effect = attempt.next_action(); + if effect == MovementAction::Retire { + if matches!( + attempt.phase(), + AttemptPhase::Activated | AttemptPhase::Recovered + ) { + let fresh = self.inspect_attempt(&attempt, clock, report).await?; + match &fresh.outcome { + FleetOutcome::Activated(evidence) => { + attempt.validate_activation(evidence).map_err(operation)? + } + FleetOutcome::Recovered(evidence) + if attempt.recovered().is_some_and(|original| { + evidence.recovery == original.recovery + && evidence.serving.position.epoch + >= original.serving.position.epoch + }) => {} + _ => { + report.blocked(DrainBlocker::OutcomeUnknown); + return Ok(()); + } + } + } + let progress = report + .snapshot + .head() + .retirement_page(&[id]) + .map_err(operation)?; + self.commit(clock, report, JournalTransition::Retire { progress }) + .await?; + report.retired += 1; + return Ok(()); + } + let now = clock.now()?; + let transition = match attempt.phase() { + AttemptPhase::Planned if now >= attempt.spec().deadline_ms => { + effect = MovementAction::Cancel; + Some(AttemptEvent::BeginCancel) + } + AttemptPhase::Planned => Some(AttemptEvent::BeginPrepare), + AttemptPhase::Reserved + if attempt.blocker().is_some() + || now >= attempt.spec().deadline_ms + || attempt.reservation().is_none_or(|r| r.expires_at_ms <= now) => + { + effect = MovementAction::Cancel; + Some(AttemptEvent::BeginCancel) + } + AttemptPhase::Reserved + if report.snapshot.head().maintenance().is_some_and(|m| { + m.id() == id.operation && m.node() == attempt.spec().source_node + }) => + { + effect = MovementAction::ReleaseMaintenance; + Some(AttemptEvent::BeginMaintenanceRelease) + } + AttemptPhase::Reserved => Some(AttemptEvent::BeginRelease), + AttemptPhase::Released + if !attempt.receiver_resources_settled() + && attempt + .reservation() + .is_some_and(|r| r.expires_at_ms <= now) => + { + // Expired unused credit must join before first activation. + // The exact release and fleet permits remain charged. + effect = MovementAction::Cancel; + Some(AttemptEvent::BeginCancel) + } + AttemptPhase::Released => Some(AttemptEvent::BeginActivate), + _ => None, + }; + if let Some(event) = transition { + self.commit_event(id, event, clock, report).await?; + } else if effect == MovementAction::Inspect { + let outcome = self.inspect_attempt(&attempt, clock, report).await?; + if attempt.phase() == AttemptPhase::Preparing + && clock.now()? >= attempt.spec().deadline_ms + { + self.commit_event(id, AttemptEvent::BeginCancel, clock, report) + .await?; + return self + .dispatch_effect(id, MovementAction::Cancel, clock, report) + .await; + } + if !matches!( + outcome.outcome, + FleetOutcome::Unknown | FleetOutcome::Blocked(_) | FleetOutcome::Rejected(_) + ) { + return self.consume(id, &outcome, true, clock, report).await; + } + // The exact acceptance decides whether replay can inspect owned + // work. A lookup alone cannot fence a concurrent delayed acceptance. + effect = phase_effect(attempt.phase()) + .ok_or(Error::Control("fleet inspection phase has no effect"))?; + let original = self.retained_action(&attempt, effect, clock).await?; + if original.is_none() { + self.commit( + clock, + report, + JournalTransition::ResolveUnaccepted { id, effect }, + ) + .await?; + let resolved = current(report, id)?; + if resolved.phase() == AttemptPhase::Reserved { + report.blocked(DrainBlocker::Deadline); + return Ok(()); + } + effect = phase_effect(resolved.phase()) + .ok_or(Error::Control("resolved absence has no effect"))?; + return self.dispatch_effect(id, effect, clock, report).await; + } + if let Some(super::super::FleetActionAcceptance::Existing { + result: Some(result), + .. + }) = &original + { + // Current serving still needs fresh capture; retained source + // release/refusal/cleanup is historical and may advance directly. + if !matches!( + result.outcome, + FleetOutcome::Activated(_) | FleetOutcome::Recovered(_) | FleetOutcome::Unknown + ) { + return self.consume(id, result, false, clock, report).await; + } + if matches!( + result.outcome, + FleetOutcome::Activated(_) | FleetOutcome::Recovered(_) + ) { + report.blocked(DrainBlocker::OutcomeUnknown); + return Ok(()); + } + } + if effect.is_source_release() && original.is_some() { + // A source effect without retained release evidence cannot be + // repeated or declared refused. Canonical recovery is separate. + report.blocked(DrainBlocker::OutcomeUnknown); + return Ok(()); + } + if matches!( + outcome.outcome, + FleetOutcome::Blocked(DrainBlocker::Deadline) + ) && matches!( + attempt.phase(), + AttemptPhase::Preparing | AttemptPhase::Activating + ) { + self.commit_event(id, AttemptEvent::BeginCancel, clock, report) + .await?; + effect = MovementAction::Cancel; + } + } + self.dispatch_effect(id, effect, clock, report).await + } + + async fn dispatch_effect( + &self, + id: AttemptId, + effect: MovementAction, + clock: &PassClock<'_>, + report: &mut FleetReconcileReport, + ) -> Result<()> { + let attempt = current(report, id)?.clone(); + // No preparation was accepted for an expired Planned attempt. Its + // cancellation is journal-local; a missing receipt alone would not + // justify this shortcut for Preparing or any later phase. + if effect == MovementAction::Cancel + && attempt.reservation().is_none() + && !matches!(attempt.phase(), AttemptPhase::CleaningReceiver) + { + let accepted = self + .retained_action(&attempt, MovementAction::Prepare, clock) + .await?; + if accepted.is_none() { + return self + .commit_event(id, AttemptEvent::Cancelled, clock, report) + .await; + } + } + let action = if attempt.blocker() == Some(DrainBlocker::OutcomeUnknown) { + // Unknown does not authorize a new effect. Replay only the exact + // original acceptance; the node may inspect/resume its owned work. + let Some(super::super::FleetActionAcceptance::Existing { accepted, .. }) = + self.retained_action(&attempt, effect, clock).await? + else { + report.blocked(DrainBlocker::OutcomeUnknown); + return Ok(()); + }; + accepted.action().clone() + } else { + report + .snapshot + .head() + .movement_action(id, effect, clock.now()?) + .map_err(operation)? + }; + report.dispatched += 1; + let completion = call( + clock.deadline, + "fleet-transport", + self.transport.dispatch(&action, clock.deadline), + ) + .await?; + let (node, session) = endpoint(&attempt, effect); + completion + .accepted + .validate_replay(&action, node, session) + .map_err(operation)?; + completion + .accepted + .validate_result(&completion.outcome) + .map_err(operation)?; + if !completion.committed { + if let Some(error) = &completion.journal_error { + tracing::warn!(error = ?error, "fleet result publication remains unresolved"); + } + report.blocked(DrainBlocker::PendingPublication); + return Ok(()); + } + if matches!( + completion.outcome.outcome, + FleetOutcome::Activated(_) | FleetOutcome::Recovered(_) + ) { + let fresh = self.inspect_attempt(&attempt, clock, report).await?; + self.consume(id, &fresh, true, clock, report).await + } else { + self.consume(id, &completion.outcome, false, clock, report) + .await + } + } + + async fn inspect_attempt( + &self, + attempt: &MoveAttempt, + clock: &PassClock<'_>, + report: &mut FleetReconcileReport, + ) -> Result { + let (mut node, mut session) = endpoint( + attempt, + if matches!( + attempt.phase(), + AttemptPhase::Releasing | AttemptPhase::MaintenanceReleasing + ) { + if attempt.phase() == AttemptPhase::MaintenanceReleasing { + MovementAction::ReleaseMaintenance + } else { + MovementAction::Release + } + } else { + MovementAction::Activate + }, + ); + let discover = attempt.released().is_some() + && (attempt.phase() == AttemptPhase::Activating + || (attempt.phase() == AttemptPhase::Activated + && attempt.receiver_resources_settled())); + if discover && let Some(serving) = attempt.activated() { + (node, session) = (serving.node, serving.session); + } + // Keep the ordinary direct check. Only unresolved serving needs a + // fleet traversal, with time reserved for its capture and native read. + let preferred_clock = clock.partition(2); + let original = self + .capture_attempt( + attempt, + node, + session, + if discover { &preferred_clock } else { clock }, + report, + ) + .await; + if !discover + || matches!(&original, Ok(outcome) if matches!(outcome.outcome, FleetOutcome::Activated(_))) + { + return original; + } + let fallback = async { + let Some(successor) = self.successor_endpoint(attempt, clock, report).await? else { + return Ok(None); + }; + if successor == (node, session) { + return Ok(None); + } + let fresh = self + .capture_attempt(attempt, successor.0, successor.1, clock, report) + .await?; + Ok::<_, Error>(matches!(fresh.outcome, FleetOutcome::Activated(_)).then_some(fresh)) + } + .await; + match fallback { + Ok(Some(fresh)) => { + if let Err(error) = original { + // Preserve the failed original endpoint even when another + // boot proves serving. No failure can free receiver credit. + report.failures.push(super::FleetAttemptFailure { + attempt: attempt.spec().id, + error: Arc::new(error), + }); + } + Ok(fresh) + } + Ok(None) => original, + Err(error) => match original { + Err(original) => { + tracing::warn!(error = ?error, "fleet successor fallback failed after original inspection failure"); + Err(original) + } + Ok(_) => Err(error), + }, + } + } + + async fn capture_attempt( + &self, + attempt: &MoveAttempt, + node: NodeId, + session: SessionId, + clock: &PassClock<'_>, + report: &mut FleetReconcileReport, + ) -> Result { + // Collecting a routing hint creates no effect acceptance. The request + // still binds this pass's exact head, registry and capture interval. + // Cleanup and its retained lookups continue using the original receiver. + let action = report + .snapshot + .head() + .movement_action(attempt.spec().id, MovementAction::Inspect, clock.now()?) + .map_err(operation)?; + let lease = report + .snapshot + .head() + .controller() + .ok_or_else(|| operation(OperationError::Fenced))?; + let request = FleetInspectionRequest::new( + action, + report.snapshot.registry(), + nonce(), + node, + session, + clock.capture_deadline_ms()?.min(lease.expires_at_ms), + ) + .map_err(operation)?; + let observation = call( + clock.deadline, + "fleet-transport", + self.transport.inspect(&request, clock.deadline), + ) + .await?; + observation + .validate_for(&request, clock.now()?, 30_000) + .map_err(operation)?; + report.inspected += 1; + Ok(observation.outcome().clone()) + } + + async fn retained_action( + &self, + attempt: &MoveAttempt, + effect: MovementAction, + clock: &PassClock<'_>, + ) -> Result> { + let (node, session) = endpoint(attempt, effect); + let original = call( + clock.deadline, + "fleet-journal", + self.journal + .load_movement_action(self.scope, attempt.spec().id, effect, node, session), + ) + .await?; + match &original { + None => {} + Some(super::super::FleetActionAcceptance::New(_)) => { + return Err(operation(OperationError::Conflict)); + } + Some(super::super::FleetActionAcceptance::Existing { accepted, result }) => { + accepted + .validate_replay(accepted.action(), node, session) + .map_err(operation)?; + let cellule_runtime::fleet::operations::FleetActionKind::Movement { + action, + attempt: input, + } = accepted.action().kind() + else { + return Err(operation(OperationError::Conflict)); + }; + if accepted.action().scope() != self.scope + || *action != effect + || input.spec() != attempt.spec() + { + return Err(operation(OperationError::Conflict)); + } + if let Some(result) = result { + accepted.validate_result(result).map_err(operation)?; + } + } + } + Ok(original) + } + + async fn commit_event( + &self, + id: AttemptId, + event: AttemptEvent, + clock: &PassClock<'_>, + report: &mut FleetReconcileReport, + ) -> Result<()> { + let confirmed = match &event { + AttemptEvent::Released(_) => Some(AttemptPhase::Released), + AttemptEvent::Activated(_) => Some(AttemptPhase::Activated), + AttemptEvent::Recovered(_) => Some(AttemptPhase::Recovered), + AttemptEvent::Cancelled => Some(AttemptPhase::Cancelled), + _ => None, + }; + let revision = report.snapshot.head().revision(); + self.commit(clock, report, JournalTransition::Attempt { id, event }) + .await?; + if report.snapshot.head().revision() != revision { + match confirmed { + Some(AttemptPhase::Released) => report.released += 1, + Some(AttemptPhase::Activated) => report.activated += 1, + Some(AttemptPhase::Recovered) => report.recovered += 1, + Some(AttemptPhase::Cancelled) => report.cancelled += 1, + _ => {} + } + } + Ok(()) + } + + async fn consume( + &self, + id: AttemptId, + result: &FleetActionOutcome, + fresh: bool, + clock: &PassClock<'_>, + report: &mut FleetReconcileReport, + ) -> Result<()> { + let attempt = current(report, id)?.clone(); + if let FleetOutcome::Rejected(blocker) | FleetOutcome::Blocked(blocker) = &result.outcome { + report.blocked(*blocker); + } + let event = match &result.outcome { + FleetOutcome::Reserved(r) if attempt.phase() == AttemptPhase::Preparing => { + if r.expires_at_ms <= clock.now()? { + self.commit_event(id, AttemptEvent::BeginCancel, clock, report) + .await?; + return Ok(()); + } + AttemptEvent::Reserved(*r) + } + FleetOutcome::Released(p) + if matches!( + attempt.phase(), + AttemptPhase::Releasing | AttemptPhase::MaintenanceReleasing + ) => + { + AttemptEvent::Released(p.clone()) + } + FleetOutcome::Activated(e) + if fresh + && matches!( + attempt.phase(), + AttemptPhase::Activating | AttemptPhase::CleaningReceiver + ) => + { + AttemptEvent::Activated(e.clone()) + } + FleetOutcome::Recovered(e) if fresh && attempt.phase() == AttemptPhase::Recovering => { + AttemptEvent::Recovered(e.clone()) + } + FleetOutcome::ReceiverCleaned if attempt.phase() == AttemptPhase::Cancelling => { + AttemptEvent::Cancelled + } + FleetOutcome::ReceiverCleaned + if matches!( + attempt.phase(), + AttemptPhase::Activated + | AttemptPhase::Recovered + | AttemptPhase::CleaningReceiver + ) => + { + AttemptEvent::ReceiverCleaned + } + FleetOutcome::Rejected(blocker) + if matches!( + attempt.phase(), + AttemptPhase::Releasing | AttemptPhase::MaintenanceReleasing + ) => + { + AttemptEvent::ReleaseRefused(*blocker) + } + FleetOutcome::Rejected(_) if attempt.phase() == AttemptPhase::Preparing => { + // Retained definite preparation refusal proves no receiver work + // was installed, and this phase cannot yet have released source. + self.commit_event(id, AttemptEvent::BeginCancel, clock, report) + .await?; + AttemptEvent::Cancelled + } + FleetOutcome::Unknown => { + report.blocked(DrainBlocker::OutcomeUnknown); + return Ok(()); + } + FleetOutcome::Blocked(blocker) | FleetOutcome::Rejected(blocker) => { + report.blocked(*blocker); + return Ok(()); + } + _ => { + report.blocked(DrainBlocker::OutcomeUnknown); + return Ok(()); + } + }; + self.commit_event(id, event, clock, report).await + } +} + +fn current(report: &FleetReconcileReport, id: AttemptId) -> Result<&MoveAttempt> { + report + .snapshot + .head() + .attempts() + .iter() + .find(|a| a.spec().id == id) + .ok_or_else(|| operation(OperationError::NotFound)) +} +fn phase_effect(phase: AttemptPhase) -> Option { + match phase { + AttemptPhase::Preparing => Some(MovementAction::Prepare), + AttemptPhase::Releasing => Some(MovementAction::Release), + AttemptPhase::MaintenanceReleasing => Some(MovementAction::ReleaseMaintenance), + AttemptPhase::Activating => Some(MovementAction::Activate), + AttemptPhase::Recovering => Some(MovementAction::Recover), + AttemptPhase::Cancelling + | AttemptPhase::CleaningReceiver + | AttemptPhase::Activated + | AttemptPhase::Recovered => Some(MovementAction::Cancel), + _ => None, + } +} +fn endpoint(attempt: &MoveAttempt, effect: MovementAction) -> (NodeId, SessionId) { + let spec = attempt.spec(); + if effect.is_source_release() { + (spec.source_node, spec.source) + } else { + (spec.destination_node, spec.destination) + } +} diff --git a/crates/cellule-host/src/fleet/reconciler/observation/mod.rs b/crates/cellule-host/src/fleet/reconciler/observation/mod.rs new file mode 100644 index 00000000..4f79c6a6 --- /dev/null +++ b/crates/cellule-host/src/fleet/reconciler/observation/mod.rs @@ -0,0 +1,312 @@ +use std::collections::{HashMap, HashSet}; + +use crate::fleet::{FleetRoleCoverage, FleetRoster}; +use cellule_runtime::cell::actor::OwnedCellObservation; +use cellule_runtime::fleet::operations::{FleetScope, RegistryVersion}; +use cellule_runtime::fleet::placement::PlacementObservation; +use cellule_runtime::identity::{Digest, NodeId, SessionId}; +use cellule_runtime::node::NodeAdvertisement; +use cellule_runtime::{Error, Result}; + +/// Generation-bound actor observation from an authenticated exact node boot. +#[derive(Clone, Debug)] +pub struct FleetOwnedCell { + /// Physical origin authenticated by the observation adapter. + pub node: NodeId, + /// Exact boot whose actor supplied this page entry. + pub session: SessionId, + /// Existing actor inventory, including measured costs and blockers. + pub observation: OwnedCellObservation, +} + +/// Bounded aggregate of authenticated paginated observations for one barrier. +/// +/// This is an in-process adapter value, not a wire or persisted format. Each +/// transport page remains bounded to 128 rows and one MiB. Aggregation is capped +/// at the planner's 10,000 nodes/Cells; applications account collector buffers. +/// The reconciler traverses the durable roster. The adapter proves native-role +/// coverage, unexpected-live discovery and signing-key enrollment independently +/// of self-signature verification. A digest identifies inputs, not atomicity. +pub struct FleetObservation { + pub(super) scope: FleetScope, + pub(super) registry: RegistryVersion, + pub(super) membership_revision: u64, + pub(super) capture_started_at_ms: i64, + pub(super) capture_finished_at_ms: i64, + pub(super) complete: bool, + pub(super) nodes: Vec, + pub(super) cells: Vec, + roster: Option, + role_coverage: Option, +} + +impl FleetObservation { + /// Retains the original collection interval. `complete` asserts a stable, + /// fully scanned native-role and ownership inventory, including busy/transitional + /// entries in the signed counts. The reconciler separately validates durable + /// boot coverage, Pending enrollment and matching signed writer counts. + #[allow(clippy::too_many_arguments)] + pub fn new( + scope: FleetScope, + registry: RegistryVersion, + membership_revision: u64, + capture_started_at_ms: i64, + capture_finished_at_ms: i64, + complete: bool, + nodes: Vec, + cells: Vec, + ) -> Result { + let observation = Self { + scope, + registry, + membership_revision, + capture_started_at_ms, + capture_finished_at_ms, + complete, + nodes, + cells, + roster: None, + role_coverage: None, + }; + observation.placements(capture_finished_at_ms)?; + Ok(observation) + } + + /// Retains the checked native/foreign graph inside this original capture. + /// This cannot upgrade `complete`: authentication, membership discovery, + /// current Cell authority, policy and failed-process evidence remain the + /// adapter's duties. The reconciler compares the exact full roster again. + pub fn with_role_coverage(mut self, coverage: FleetRoleCoverage) -> Result { + if self.role_coverage.is_some() { + return Err(Error::Control("fleet role coverage already retained")); + } + self.role_coverage = Some(coverage); + self.validate_role_coverage()?; + Ok(self) + } + + /// Original role graph retained in the planner inputs; never restamped. + #[must_use] + pub fn role_coverage(&self) -> Option<&FleetRoleCoverage> { + self.role_coverage.as_ref() + } + + fn validate_role_coverage(&self) -> Result<()> { + if let Some(coverage) = &self.role_coverage { + let (started, finished) = coverage.interval(); + if coverage.snapshot().head().scope() != self.scope + || coverage.snapshot().registry() != self.registry + || started < self.capture_started_at_ms + || finished > self.capture_finished_at_ms + { + return Err(Error::Node("fleet role coverage barrier differs")); + } + if let Some(roster) = &self.roster + && (coverage.snapshot() != roster.snapshot() + || roster.digest()? != coverage.roster_digest()) + { + return Err(Error::Node("fleet role coverage roster differs")); + } + } + Ok(()) + } + + /// Retains a fully traversed durable roster in these planner inputs. The + /// adapter must recheck it after native capture. Attaching rows alone cannot + /// establish native-role coverage or upgrade an incomplete observation. + pub(super) fn with_roster(mut self, roster: FleetRoster) -> Result { + if roster.snapshot().registry() != self.registry + || roster.snapshot().head().scope() != self.scope + { + return Err(Error::Node("fleet observation roster barrier differs")); + } + self.roster = Some(roster); + self.validate_role_coverage()?; + Ok(self) + } + + /// Returns the original roster retained by the reconciler after capture. + #[must_use] + pub fn roster(&self) -> Option<&FleetRoster> { + self.roster.as_ref() + } + + pub(super) fn counts_match(&self, placements: &[PlacementObservation]) -> bool { + let mut counts = HashMap::new(); + for owned in &self.cells { + *counts.entry(owned.session).or_insert(0_u32) += 1; + } + // A transitioning actor remains in the signed count. Omitting its row + // cannot turn a partial inventory into a complete count barrier. + placements + .iter() + .all(|node| counts.get(&node.session).copied().unwrap_or(0) == node.active_cells) + } + + pub(super) fn placements(&self, now_ms: i64) -> Result> { + self.validate_role_coverage()?; + if self.registry.scope() != self.scope + || self.membership_revision == 0 + || self.capture_started_at_ms < 0 + || self.capture_finished_at_ms < self.capture_started_at_ms + || self.capture_finished_at_ms > now_ms + || now_ms - self.capture_started_at_ms > 30_000 + || self.nodes.len() > 10_000 + || self.cells.len() > 10_000 + { + return Err(Error::Node("invalid fleet observation barrier or bounds")); + } + let mut nodes = HashSet::new(); + let mut sessions = HashSet::new(); + let mut placements = Vec::with_capacity(self.nodes.len()); + for node in &self.nodes { + if node.fleet() != self.scope.fleet + || !nodes.insert(node.node()) + || !sessions.insert(node.session()) + { + return Err(Error::Node("fleet observation duplicates or crosses scope")); + } + placements.push(PlacementObservation::from_signed_advertisement( + node, now_ms, false, + )?); + } + let mut cells = HashSet::new(); + for owned in &self.cells { + let row = &owned.observation; + if !cells.insert(row.target.cell_id()) + || row.target.application() != self.scope.application + || !placements + .iter() + .any(|node| node.node == owned.node && node.session == owned.session) + || row.generation == 0 + || row.resident_since_ms < 0 + || row.resident_since_ms > now_ms + { + return Err(Error::Node("fleet ownership observation identity mismatch")); + } + } + placements.sort_by_key(|node| (*node.node.as_bytes(), *node.session.as_bytes())); + Ok(placements) + } + + /// Identifies the canonical planner inputs after signature/shape validation. + /// Full envelope/role transport codecs remain distinct from this local value. + pub(super) fn digest(&self, now_ms: i64) -> Result { + let nodes = self.placements(now_ms)?; + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-planner-inputs.v5\0"); + hash.update(self.scope.fleet.as_bytes()); + hash.update(self.scope.application.as_bytes()); + hash.update(&self.registry.to_bytes().map_err(super::operation)?); + for time in [ + self.membership_revision, + self.capture_started_at_ms as u64, + self.capture_finished_at_ms as u64, + ] { + hash.update(&time.to_be_bytes()); + } + hash.update(&[u8::from(self.complete)]); + hash.update(&[u8::from(self.roster.is_some())]); + if let Some(roster) = &self.roster { + hash.update(roster.digest()?.as_bytes()); + } + hash.update(&[u8::from(self.role_coverage.is_some())]); + if let Some(coverage) = &self.role_coverage { + hash.update(coverage.digest().as_bytes()); + } + hash.update(&(nodes.len() as u64).to_be_bytes()); + for node in nodes { + hash.update(node.node.as_bytes()); + hash.update(node.session.as_bytes()); + for n in [ + node.observed_at_ms as u64, + node.memory_capacity_bytes, + node.free_memory_bytes, + node.disk_capacity_bytes, + node.free_disk_bytes, + u64::from(node.active_cells), + u64::from(node.max_active_cells), + u64::from(node.running_jobs), + u64::from(node.job_capacity), + u64::from(node.publication_backlog), + u64::from(node.hydration_backlog), + u64::from(node.primitive_backlog), + ] { + hash.update(&n.to_be_bytes()); + } + hash.update(&[node.pressure as u8, u8::from(node.draining)]); + } + let mut cells = self.cells.iter().collect::>(); + cells.sort_by_key(|row| *row.observation.target.cell_id().as_bytes()); + hash.update(&(cells.len() as u64).to_be_bytes()); + for owned in cells { + let row = &owned.observation; + hash.update(owned.node.as_bytes()); + hash.update(owned.session.as_bytes()); + hash.update(row.target.cell_id().as_bytes()); + hash.update(row.incarnation.as_bytes()); + hash.update(row.code.as_bytes()); + hash.update(&row.schema.to_be_bytes()); + // Role now controls busy-maintenance eligibility. Use explicit tags + // in this producer domain rather than enum declaration order. + use cellule_runtime::cell::catalog::CatalogRole; + hash.update(&[match row.role { + CatalogRole::Application => 1, + CatalogRole::Sql => 2, + CatalogRole::Kv => 3, + CatalogRole::Queue => 4, + CatalogRole::Workflow => 5, + CatalogRole::Blob => 6, + CatalogRole::Cron => 7, + }]); + hash.update(&row.generation.to_be_bytes()); + hash.update(&row.resident_since_ms.to_be_bytes()); + hash.update(&row.sampled_at_ms.unwrap_or(-1).to_be_bytes()); + hash.update(&[row.stable_observations]); + hash.update(&[ + u8::from(row.quiescing), + u8::from(row.maintenance_work.is_some()), + ]); + if let Some(inventory) = row.maintenance_work { + use cellule_runtime::primitives::maintenance_readiness::MaintenanceWorkBlocker; + for blocker in [ + MaintenanceWorkBlocker::EffectLease, + MaintenanceWorkBlocker::QueueLease, + MaintenanceWorkBlocker::ActivityLease, + MaintenanceWorkBlocker::BlobInventory, + ] { + hash.update(&[u8::from(inventory.has_blocker(blocker))]); + } + } + + hash.update(&[u8::from( + row.blockers.is_empty() && row.work_blocker.is_none(), + )]); + hash.update(&(row.blockers.len() as u64).to_be_bytes()); + for blocker in &row.blockers { + hash.update(&[*blocker as u8]); + } + for cost in [row.cost, row.maintenance_cost] { + hash.update(&[u8::from(cost.is_some())]); + if let Some(cost) = cost { + hash.update(&cost.memory_bytes.to_be_bytes()); + hash.update(&cost.disk_bytes.to_be_bytes()); + hash.update(&cost.file_descriptors.to_be_bytes()); + hash.update(&cost.job_credits.to_be_bytes()); + } + } + hash.update(&[u8::from(row.position.is_some())]); + if let Some(position) = &row.position { + hash.update(&position.epoch.to_be_bytes()); + hash.update(position.root.digest.as_bytes()); + hash.update(&position.root.txid.to_be_bytes()); + hash.update(&position.root.checksum.to_be_bytes()); + hash.update(&position.root.commit_sequence.to_be_bytes()); + } + } + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-host/src/fleet/reconciler/observation/tests.rs b/crates/cellule-host/src/fleet/reconciler/observation/tests.rs new file mode 100644 index 00000000..201179d7 --- /dev/null +++ b/crates/cellule-host/src/fleet/reconciler/observation/tests.rs @@ -0,0 +1,149 @@ +use super::*; +use cellule_runtime::cell::catalog::CatalogRole; +use cellule_runtime::fleet::operations::{DrainBlocker, TransferCost}; +use cellule_runtime::identity::{ApplicationId, CellTarget, IncarnationId, NamespaceId, TenantId}; +use cellule_runtime::node::{ + NodeCapacity, NodeFailureDomain, NodeOperationalSample, NodePlacementCapacity, +}; + +fn observation() -> FleetObservation { + let scope = FleetScope { + fleet: Digest::from_bytes([1; 32]), + application: ApplicationId::from_bytes([2; 16]), + }; + let node = NodeId::from_bytes([3; 16]); + let session = SessionId::from_bytes([4; 16]); + let key = ed25519_dalek::SigningKey::from_bytes(&[5; 32]); + let signed = NodeAdvertisement::sign( + node, + session, + "https://node.internal:8789".into(), + scope.fleet, + Digest::from_bytes([6; 32]), + Digest::from_bytes([7; 32]), + Digest::from_bytes([8; 32]), + &key, + 1, + 100, + 30_100, + vec![Digest::from_bytes([12; 32])], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + free_memory_bytes: 1000, + free_disk_bytes: 1000, + job_credits: 2, + log_protocol: 1, + ..NodeCapacity::default() + }, + ) + .unwrap() + .with_operational_placement( + NodePlacementCapacity { + memory_capacity_bytes: 1000, + disk_capacity_bytes: 1000, + active_cells: 1, + max_active_cells: 2, + job_capacity: 2, + ..NodePlacementCapacity::default() + }, + NodeOperationalSample { + sequence: 1, + observed_at_ms: 100, + mode: cellule_runtime::node::NodeMode::Active, + pressure: cellule_runtime::node::NodePressure::Normal, + }, + &key, + ) + .unwrap(); + FleetObservation::new( + scope, + RegistryVersion::new(scope).unwrap(), + 1, + 100, + 100, + false, + vec![signed], + vec![FleetOwnedCell { + node, + session, + observation: OwnedCellObservation { + target: CellTarget::new( + TenantId::from_bytes([9; 16]), + scope.application, + NamespaceId::from_bytes([10; 16]), + b"digest", + ) + .unwrap(), + generation: 1, + incarnation: IncarnationId::from_bytes([11; 16]), + code: Digest::from_bytes([12; 32]), + schema: 1, + role: CatalogRole::Sql, + resident_since_ms: 50, + last_used_ms: 100, + position: None, + cost: None, + maintenance_cost: Some(TransferCost { + memory_bytes: 100, + disk_bytes: 200, + file_descriptors: 8, + job_credits: 1, + }), + database_bytes: None, + sampled_at_ms: None, + stable_observations: 0, + work_blocker: None, + quiescing: false, + maintenance_work: None, + blockers: vec![DrainBlocker::BusyExecution, DrainBlocker::UnknownInventory], + }, + }], + ) + .unwrap() +} + +#[test] +fn planner_digest_binds_peak_cost_presence_and_every_admission_dimension() { + let baseline = observation().digest(100).unwrap(); + for field in 0..5 { + let mut inputs = observation(); + let cost = &mut inputs.cells[0].observation.maintenance_cost; + if field == 0 { + *cost = None; + } else { + let cost = cost.as_mut().unwrap(); + match field { + 1 => cost.memory_bytes += 1, + 2 => cost.disk_bytes += 1, + 3 => cost.file_descriptors += 1, + _ => cost.job_credits += 1, + } + } + assert_ne!(inputs.digest(100).unwrap(), baseline); + } + assert_eq!(observation().digest(100).unwrap(), baseline); +} + +#[test] +fn planner_digest_binds_role_blocker_and_executable_identity() { + let baseline = observation().digest(100).unwrap(); + for field in 0..4 { + let mut inputs = observation(); + let row = &mut inputs.cells[0].observation; + match field { + 0 => row.role = CatalogRole::Blob, + 1 => row.blockers[1] = DrainBlocker::FollowerObligation, + 2 => row.code = Digest::from_bytes([99; 32]), + _ => row.schema += 1, + } + assert_ne!(inputs.digest(100).unwrap(), baseline); + } +} + +#[test] +fn busy_envelope_cannot_refresh_an_expired_collection_barrier() { + let observation = observation(); + assert!(observation.digest(30_101).is_err()); + assert!(observation.placements(30_101).is_err()); +} diff --git a/crates/cellule-host/src/fleet/reconciler/planning.rs b/crates/cellule-host/src/fleet/reconciler/planning.rs new file mode 100644 index 00000000..8dfb7b1b --- /dev/null +++ b/crates/cellule-host/src/fleet/reconciler/planning.rs @@ -0,0 +1,278 @@ +use super::*; +use cellule_runtime::fleet::operations::{ + AttemptId, FleetHead, MaintenanceOperation, MoveAttemptSpec, OperationId, +}; +use cellule_runtime::fleet::placement::{CellTransferDemand, PlacementPlanner, PlacementPressure}; + +impl FleetReconciler { + pub(super) async fn plan( + &self, + clock: &PassClock<'_>, + report: &mut FleetReconcileReport, + ) -> Result<()> { + if report.snapshot.head().attempts().len() >= self.profile.max_inflight { + report.blocked(DrainBlocker::MovementBudget); + return Ok(()); + } + if report.snapshot.registry().bootstrap_revision().is_none() { + report.blocked(DrainBlocker::IncompleteObservation); + return Ok(()); + } + let roster = crate::fleet::FleetRoster::collect( + self.journal.as_ref(), + &report.snapshot, + clock.deadline, + ) + .await?; + let observation = call( + clock.deadline, + "fleet-observer", + self.observer.observe(&roster, clock.now()?, clock.deadline), + ) + .await?; + if observation.scope != self.scope || observation.registry != report.snapshot.registry() { + return Err(operation(OperationError::Conflict)); + } + roster + .confirm(self.journal.as_ref(), clock.deadline) + .await?; + let now = clock.now()?; + let boot_complete = roster.covers_advertisements(&observation.nodes, now)?; + let enrollment_settled = roster.enrollments().iter().all(|record| { + record.status() != cellule_runtime::fleet::operations::EnrollmentStatus::Pending + }); + let observation = observation.with_roster(roster)?; + let mut placements = observation.placements(now)?; + let inventory_complete = observation.complete + && boot_complete + && enrollment_settled + && observation.counts_match(&placements); + // The most recently retired page is a post-batch sample barrier. Query + // all earlier movement times as well: late retirement must not make + // the count rule forget a more recent completion in an earlier page. + let since = call( + clock.deadline, + "fleet-journal", + self.journal.last_movement_at(&report.snapshot), + ) + .await? + .unwrap_or(-1); + let planner = PlacementPlanner::default(); + let count_fresh = inventory_complete + && report.snapshot.head().attempts().is_empty() + && placements.iter().all(|node| node.observed_at_ms > since); + if !inventory_complete { + report.blocked(DrainBlocker::IncompleteObservation); + } else if !count_fresh { + report.blocked(DrainBlocker::StaleObservation); + } + // Retained intents take precedence over cached signed advertisements. + for intent in observation + .roster() + .ok_or(Error::Control("fleet roster was not retained"))? + .intents() + { + if let Some(node) = placements + .iter_mut() + .find(|node| node.node == intent.node()) + { + if node.session != intent.session() { + return Err(operation(OperationError::Conflict)); + } + node.draining |= intent.mode() != cellule_runtime::node::NodeMode::Active; + } + } + let mut demands = Vec::new(); + for owned in &observation.cells { + let row = &owned.observation; + if report + .snapshot + .head() + .attempts() + .iter() + .any(|attempt| attempt.spec().target.cell_id() == row.target.cell_id()) + || report + .snapshot + .head() + .maintenance() + .is_some_and(|operation| { + operation.node() == owned.node + && (operation.phase() != MaintenancePhase::Evacuating + || now >= operation.deadline_ms()) + }) + { + continue; + } + let maintenance = maintenance_for(report.snapshot.head(), owned, now).is_some(); + if maintenance && row.role == cellule_runtime::cell::catalog::CatalogRole::Blob { + report.blocked(DrainBlocker::UnknownInventory); + continue; + } + let Some(cost) = (if maintenance { + row.maintenance_cost + } else { + row.cost + }) else { + continue; + }; + cost.validate().map_err(operation)?; + let settled = row.blockers.is_empty() + && row.work_blocker.is_none() + && row + .sampled_at_ms + .is_some_and(|at| at >= 0 && at <= now && now - at <= 30_000); + // These local conditions are joined/rechecked by explicit busy + // release after receiver preparation. Other role/fleet blockers + // cannot be discharged by a Cell actor readiness read. + if maintenance + && let Some(blocker) = row.blockers.iter().find(|blocker| { + !matches!( + blocker, + DrainBlocker::BusyExecution + | DrainBlocker::ExternalLease + | DrainBlocker::PendingPublication + | DrainBlocker::UnknownInventory + ) + }) + { + report.blocked(*blocker); + continue; + } + if row.position.as_ref().is_none_or(|position| { + position.incarnation != row.incarnation || position.epoch == 0 + }) || (!maintenance && !settled) + { + continue; + } + let source = placements + .iter() + .find(|node| node.session == owned.session) + .ok_or(Error::Node("fleet donor observation absent"))?; + if !count_fresh && !source.draining && source.pressure < PlacementPressure::Shedding { + continue; + } + let moved = call( + clock.deadline, + "fleet-journal", + self.journal + .last_moved_at(&report.snapshot, row.target.cell_id(), row.incarnation), + ) + .await?; + demands.push(CellTransferDemand { + cell: row.target.cell_id(), + source: owned.session, + generation: row.generation, + memory_bytes: cost.memory_bytes, + disk_bytes: cost.disk_bytes, + job_credits: cost.job_credits, + resident_since_ms: row.resident_since_ms, + last_used_ms: row.last_used_ms, + last_moved_at_ms: moved, + stable_observations: row.stable_observations, + settled, + maintenance, + }); + } + // Unknown receives stay charged independently of lagging advertisements. + // Projection is conservative until permit retirement; pressure relief + // can use the remaining count/byte budget without inventing count balance. + for attempt in report.snapshot.head().attempts() { + if let Some(receiver) = placements + .iter_mut() + .find(|node| node.session == attempt.spec().destination) + { + let cost = attempt.spec().cost; + receiver.free_memory_bytes = + receiver.free_memory_bytes.saturating_sub(cost.memory_bytes); + receiver.free_disk_bytes = receiver.free_disk_bytes.saturating_sub(cost.disk_bytes); + receiver.active_cells = receiver.active_cells.saturating_add(1); + receiver.running_jobs = receiver.running_jobs.saturating_add(cost.job_credits); + } + } + // Recompute balance using the final eligibility inputs after intents. + let balance = if count_fresh { + planner.fleet_balance(now, &placements, since)? + } else { + None + }; + let proposals = planner.plan_transfers(now, &placements, &demands, balance.as_ref())?; + let digest = observation.digest(now)?; + let operation_id = + OperationId::from_bytes(*uuid::Uuid::now_v7().as_bytes()).map_err(operation)?; + for proposal in proposals { + let head = report.snapshot.head(); + if head.attempts().len() >= self.profile.max_inflight + || head + .reserved_restore_bytes() + .checked_add(proposal.disk_bytes) + .is_none_or(|bytes| bytes > self.profile.max_restore_bytes) + { + report.blocked(DrainBlocker::MovementBudget); + break; + } + let owned = observation + .cells + .iter() + .find(|row| { + row.observation.target.cell_id() == proposal.cell + && row.session == proposal.source + && row.observation.generation == proposal.generation + }) + .ok_or(Error::Node("fleet proposal lost its actor observation"))?; + let row = &owned.observation; + let destination = placements + .iter() + .find(|node| node.session == proposal.destination) + .ok_or(Error::Node("fleet proposal receiver absent"))?; + let position = row + .position + .as_ref() + .ok_or(Error::Node("fleet proposal position absent"))?; + let maintenance = maintenance_for(head, owned, now); + let cost = if maintenance.is_some() { + row.maintenance_cost + } else { + row.cost + } + .ok_or(Error::Node("fleet proposal cost absent"))?; + let id = AttemptId { + operation: maintenance.map_or(operation_id, |m| m.id()), + sequence: head.next_sequence(), + }; + let spec = MoveAttemptSpec { + id, + target: row.target.clone(), + incarnation: row.incarnation, + source_node: owned.node, + source: owned.session, + generation: row.generation, + source_epoch: position.epoch, + destination_node: destination.node, + destination: destination.session, + cost, + snapshot_digest: digest, + deadline_ms: now + .checked_add(self.profile.controller_lease_ms) + .ok_or(Error::Control("fleet movement deadline overflow"))? + .min(maintenance.map_or(i64::MAX, |m| m.deadline_ms())), + }; + self.commit(clock, report, JournalTransition::Allocate(spec)) + .await?; + report.allocated += 1; + } + Ok(()) + } +} + +fn maintenance_for<'a>( + head: &'a FleetHead, + owned: &FleetOwnedCell, + now: i64, +) -> Option<&'a MaintenanceOperation> { + head.maintenance().filter(|operation| { + operation.node() == owned.node + && operation.session() == owned.session + && operation.phase() == MaintenancePhase::Evacuating + && now < operation.deadline_ms() + }) +} diff --git a/crates/cellule-host/src/fleet/reconciler/successor.rs b/crates/cellule-host/src/fleet/reconciler/successor.rs new file mode 100644 index 00000000..d3eda77a --- /dev/null +++ b/crates/cellule-host/src/fleet/reconciler/successor.rs @@ -0,0 +1,71 @@ +use super::*; +use cellule_runtime::fleet::operations::{ + ActivationEvidence, EnrollmentRole, EnrollmentStatus, MoveAttempt, +}; +use cellule_runtime::identity::{NodeId, SessionId}; + +impl FleetReconciler { + pub(super) async fn successor_endpoint( + &self, + attempt: &MoveAttempt, + clock: &PassClock<'_>, + report: &FleetReconcileReport, + ) -> Result> { + let roster = + FleetRoster::collect(self.journal.as_ref(), &report.snapshot, clock.deadline).await?; + let observation = call( + clock.deadline, + "fleet-observer", + self.observer.observe(&roster, clock.now()?, clock.deadline), + ) + .await?; + if observation.scope != self.scope || observation.registry != report.snapshot.registry() { + return Err(operation(OperationError::Conflict)); + } + roster + .confirm(self.journal.as_ref(), clock.deadline) + .await?; + observation.placements(clock.now()?)?; + for owned in &observation.cells { + let row = &owned.observation; + if row.target != attempt.spec().target || row.incarnation != attempt.spec().incarnation + { + continue; + } + let Some(position) = &row.position else { + continue; + }; + let hint = ActivationEvidence { + node: owned.node, + session: owned.session, + position: position.clone(), + }; + if attempt.validate_activation(&hint).is_err() { + continue; + } + let enrolled = roster + .intents() + .iter() + .filter(|intent| intent.node() == owned.node && intent.session() == owned.session) + .any(|intent| { + roster.enrollments().iter().any(|record| { + record.status() == EnrollmentStatus::Established + && matches!(record.spec().role, EnrollmentRole::Node { .. }) + && crate::fleet::FleetBootObservation::new( + intent.clone(), + record.clone(), + ) + .is_ok() + }) + }); + if !enrolled { + continue; + } + // This bounded, authenticated row selects a read endpoint only. + // Native authority + actor inspection supplies the serving proof; + // incomplete role coverage cannot establish absence/finalization. + return Ok(Some((owned.node, owned.session))); + } + Ok(None) + } +} diff --git a/crates/cellule-host/src/fleet/recovered/mod.rs b/crates/cellule-host/src/fleet/recovered/mod.rs new file mode 100644 index 00000000..1811a043 --- /dev/null +++ b/crates/cellule-host/src/fleet/recovered/mod.rs @@ -0,0 +1,273 @@ +//! Durable enrollment publication after canonical recovered ensemble retirement. +use super::{FleetJournal, FleetJournalSnapshot, FleetRoster, operation}; +use cellule_runtime::{ + Error, Result, + fleet::operations::EnrollmentRecord, + identity::{Digest, NodeId, SessionId}, + node::{NodeDirectory, SealedNodeLog, log_state::NodeLogPhase}, +}; +use std::{future::Future, sync::Arc}; +use tokio::time::{Instant, timeout_at}; + +mod publication; +mod records; + +/// Exact original enrollment requests bound to canonical recovered retirement. +/// +/// Capture after the existing recovery/member-retirement protocol commits its +/// Retired tombstone. No native retirement or recovery effect runs here. The +/// application authenticates the claimant and accounts the bounded copied rows +/// (at most sixteen records). Failed-boot/process closure and replacement policy +/// remain separate obligations; this value never grants node stop permission. +pub struct FleetRecoveredFollowerRetirement { + snapshot: FleetJournalSnapshot, + leader_node: NodeId, + retired: SealedNodeLog, + members: Vec, + evidence: Vec, + started_at_ms: i64, + finished_at_ms: i64, +} + +impl FleetRecoveredFollowerRetirement { + /// Confirms the complete original roster around fresh canonical retirement. + /// Missing/duplicate/foreign member requests and unretired epochs refuse + /// capture. Terminal rows retain their original acceptance/evidence times. + pub async fn capture( + journal: &dyn FleetJournal, + directory: &NodeDirectory, + roster: &FleetRoster, + sealed: &SealedNodeLog, + claimant: SessionId, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + let started_at_ms = clock()?; + if roster.snapshot().registry().bootstrap_revision().is_none() + || directory.fleet() != roster.snapshot().head().scope().fleet + { + return Err(Error::Fenced); + } + roster.confirm(journal, deadline).await?; + let (leader_node, retired) = + canonical(directory, sealed, claimant, started_at_ms, deadline).await?; + let members = records::select(roster, leader_node, &retired)?; + let evidence = members + .iter() + .map(|row| records::evidence(leader_node, &retired, row)) + .collect::>>()?; + for (row, digest) in members.iter().zip(&evidence) { + records::validate_terminal(row, *digest)?; + } + roster.confirm(journal, deadline).await?; + let finished_at_ms = clock()?; + interval(started_at_ms, finished_at_ms)?; + Ok(Self { + snapshot: roster.snapshot().clone(), + leader_node, + retired, + members, + evidence, + started_at_ms, + finished_at_ms, + }) + } + + /// Original full journal barrier; publication rechecks it before effects. + #[must_use] + pub fn snapshot(&self) -> &FleetJournalSnapshot { + &self.snapshot + } + /// Original physical leader, verified through its canonical tombstone. + #[must_use] + pub const fn leader_node(&self) -> NodeId { + self.leader_node + } + /// Canonical Retired epoch, complete ensemble and pinned manifest. + #[must_use] + pub const fn retired(&self) -> &SealedNodeLog { + &self.retired + } + /// All original requests in canonical member order, including Pending rows. + #[must_use] + pub fn members(&self) -> &[EnrollmentRecord] { + &self.members + } + /// Capture times; publication cannot renew or restamp these observations. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + (self.started_at_ms, self.finished_at_ms) + } +} + +/// One joined durable publication, retaining its original source error. +pub struct FleetRecoveredFollowerMember { + original: EnrollmentRecord, + result: std::result::Result>, +} +impl FleetRecoveredFollowerMember { + /// Immutable original request and first acceptance. + #[must_use] + pub const fn original(&self) -> &EnrollmentRecord { + &self.original + } + /// Returned original retirement record or independently retained failure. + pub fn result(&self) -> std::result::Result<&EnrollmentRecord, Arc> { + self.result.as_ref().map_err(Arc::clone) + } +} + +/// Joined member publications plus the final complete-roster/canonical recheck. +/// An error in either stage prevents closure without discarding sibling results. +pub struct FleetRecoveredFollowerPublication { + members: Vec, + closure: std::result::Result>, +} +impl FleetRecoveredFollowerPublication { + /// Every original member's response; lost replies never imply settlement. + #[must_use] + pub fn members(&self) -> &[FleetRecoveredFollowerMember] { + &self.members + } + /// Confirms publication against a fresh complete roster and canonical epoch. + /// Boot/process joining, replacement policy and node finalization are separate. + pub fn confirmed(&self) -> Result<&FleetRecoveredFollowerClosure> { + for member in &self.members { + if let Err(source) = &member.result { + return Err(retained(Arc::clone(source))); + } + } + self.closure + .as_ref() + .map_err(|source| retained(Arc::clone(source))) + } + /// Original final-check error, including one caused by a failed member reply. + #[must_use] + pub fn closure_error(&self) -> Option> { + self.closure.as_ref().err().cloned() + } +} + +/// Complete original ensemble enrollment closure at one checked journal barrier. +/// This settles only these follower rows. The failed leader's boot, reader roles, +/// affected Cells and native process lifetimes still require their own proofs. +pub struct FleetRecoveredFollowerClosure { + snapshot: FleetJournalSnapshot, + leader_node: NodeId, + retired: SealedNodeLog, + members: Vec, + digest: Digest, + started_at_ms: i64, + finished_at_ms: i64, +} +impl FleetRecoveredFollowerClosure { + /// Full post-publication head and registry confirmed around canonical authority. + #[must_use] + pub fn snapshot(&self) -> &FleetJournalSnapshot { + &self.snapshot + } + /// Original canonical physical leader. + #[must_use] + pub const fn leader_node(&self) -> NodeId { + self.leader_node + } + /// Canonical terminal epoch and retained manifest. + #[must_use] + pub const fn retired(&self) -> &SealedNodeLog { + &self.retired + } + /// Original retired member rows, preserving accepted and establishment history. + #[must_use] + pub fn members(&self) -> &[EnrollmentRecord] { + &self.members + } + /// Identifies terminal authority and full original member records, without a clock. + #[must_use] + pub const fn digest(&self) -> Digest { + self.digest + } + /// Original capture through final confirmation; this is interval evidence. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + (self.started_at_ms, self.finished_at_ms) + } +} + +async fn canonical( + directory: &NodeDirectory, + sealed: &SealedNodeLog, + claimant: SessionId, + now: i64, + deadline: Instant, +) -> Result<(NodeId, SealedNodeLog)> { + let retired = bounded( + deadline, + directory.retired_recovered_log(sealed, claimant, now), + ) + .await? + .ok_or(Error::Control( + "recovered ensemble is not canonically retired", + ))?; + let member = *retired.log().members().first().ok_or(Error::Fenced)?; + let authorization = bounded( + deadline, + directory.authorize_recovered_log_retire( + claimant, + member, + retired.session(), + retired.log().epoch(), + retired.log().recovery_manifest(), + now, + ), + ) + .await?; + if authorization.sealed() != &retired || retired.log().phase() != NodeLogPhase::Retired { + return Err(Error::Fenced); + } + Ok((authorization.leader_node(), retired)) +} + +fn interval(start: i64, end: i64) -> Result<()> { + if start < 0 || end < start || end - start > 30_000 { + return Err(Error::Deadline); + } + Ok(()) +} + +async fn bounded(deadline: Instant, future: impl Future>) -> Result { + if Instant::now() >= deadline { + return Err(Error::Deadline); + } + timeout_at(deadline, future) + .await + .map_err(|source| Error::Facility { + name: "fleet-recovered-follower-deadline", + source: Box::new(source), + })? +} + +fn journal_error(source: Box) -> Error { + Error::Facility { + name: "fleet-recovered-follower-journal", + source, + } +} + +#[derive(Debug)] +struct RetainedError(Arc); +impl std::fmt::Display for RetainedError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + self.0.fmt(f) + } +} +impl std::error::Error for RetainedError { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + Some(self.0.as_ref()) + } +} +fn retained(source: Arc) -> Error { + Error::Facility { + name: "fleet-recovered-follower-publication", + source: Box::new(RetainedError(source)), + } +} diff --git a/crates/cellule-host/src/fleet/recovered/publication.rs b/crates/cellule-host/src/fleet/recovered/publication.rs new file mode 100644 index 00000000..9ab189b3 --- /dev/null +++ b/crates/cellule-host/src/fleet/recovered/publication.rs @@ -0,0 +1,158 @@ +use super::*; +use cellule_runtime::fleet::operations::EnrollmentEvent; +use futures_util::future::join_all; + +impl FleetRecoveredFollowerRetirement { + /// Publishes each original member's canonical retirement through the existing + /// journal contract. All dispatched waiters settle, retaining failures separately. + /// The complete post-publication roster and authority must confirm before a + /// closure is returned. A lost/cancelled waiter supplies no closure: recapture + /// a fresh roster and replay the same events. Exact duplicates preserve time. + /// The adapter retains and joins backend work after a waiter deadline/drop. + /// Own this future in the application's accepted finite work; it starts no + /// background task, second action bank, recovery or native retirement path. + pub async fn publish( + &self, + journal: &dyn FleetJournal, + directory: &NodeDirectory, + claimant: SessionId, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + let now = clock()?; + interval(self.started_at_ms, now)?; + if now < self.finished_at_ms { + return Err(Error::Deadline); + } + let snapshot = bounded(deadline, async { + journal + .load_snapshot(self.snapshot.head().scope()) + .await + .map_err(journal_error) + }) + .await?; + if snapshot != self.snapshot { + return Err(Error::Fenced); + } + self.confirm_authority(directory, claimant, now, deadline) + .await?; + let members = join_all(self.members.iter().zip(&self.evidence).map( + |(original, digest)| async move { + let result = bounded(deadline, async { + let result = journal + .publish_enrollment_result(original, EnrollmentEvent::Retired(*digest), now) + .await + .map_err(journal_error)?; + records::validate_result(original, &result, *digest)?; + Ok(result) + }) + .await + .map_err(Arc::new); + FleetRecoveredFollowerMember { + original: original.clone(), + result, + } + }, + )) + .await; + // Keep every original response even when a sibling or the final complete + // barrier fails. Neither a successful write nor an absence read upgrades + // an ambiguous publication to confirmed ensemble closure. + let mut last_ms = now; + let mut advancing_clock = || { + let next = clock()?; + if next < last_ms { + return Err(Error::Deadline); + } + last_ms = next; + Ok(next) + }; + let closure = self + .finish( + journal, + directory, + claimant, + deadline, + &mut advancing_clock, + &members, + ) + .await + .map_err(Arc::new); + Ok(FleetRecoveredFollowerPublication { members, closure }) + } + + async fn confirm_authority( + &self, + directory: &NodeDirectory, + claimant: SessionId, + now: i64, + deadline: Instant, + ) -> Result<()> { + if directory.fleet() != self.snapshot.head().scope().fleet { + return Err(Error::Fenced); + } + let (node, retired) = canonical(directory, &self.retired, claimant, now, deadline).await?; + if node != self.leader_node || retired != self.retired { + return Err(Error::Fenced); + } + Ok(()) + } + + async fn finish( + &self, + journal: &dyn FleetJournal, + directory: &NodeDirectory, + claimant: SessionId, + deadline: Instant, + clock: &mut impl FnMut() -> Result, + members: &[FleetRecoveredFollowerMember], + ) -> Result { + for member in members { + if let Err(source) = &member.result { + return Err(retained(Arc::clone(source))); + } + } + let snapshot = bounded(deadline, async { + journal + .load_snapshot(self.snapshot.head().scope()) + .await + .map_err(journal_error) + }) + .await?; + let roster = FleetRoster::collect(journal, &snapshot, deadline).await?; + let rows = records::select(&roster, self.leader_node, &self.retired)?; + if rows.len() != members.len() { + return Err(Error::Fenced); + } + for ((row, member), digest) in rows.iter().zip(members).zip(&self.evidence) { + if member + .result + .as_ref() + .map_err(|source| retained(Arc::clone(source)))? + != row + { + return Err(Error::Control("recovered follower publication changed")); + } + records::validate_result(&member.original, row, *digest)?; + } + let now = clock()?; + interval(self.started_at_ms, now)?; + self.confirm_authority(directory, claimant, now, deadline) + .await?; + roster.confirm(journal, deadline).await?; + let finished_at_ms = clock()?; + interval(self.started_at_ms, finished_at_ms)?; + if finished_at_ms < now { + return Err(Error::Deadline); + } + Ok(FleetRecoveredFollowerClosure { + snapshot, + leader_node: self.leader_node, + retired: self.retired.clone(), + digest: records::closure_digest(self.leader_node, &self.retired, &rows)?, + members: rows, + started_at_ms: self.started_at_ms, + finished_at_ms, + }) + } +} diff --git a/crates/cellule-host/src/fleet/recovered/records.rs b/crates/cellule-host/src/fleet/recovered/records.rs new file mode 100644 index 00000000..aa68d2af --- /dev/null +++ b/crates/cellule-host/src/fleet/recovered/records.rs @@ -0,0 +1,119 @@ +use super::*; +use cellule_runtime::fleet::operations::{EnrollmentRole, EnrollmentStatus}; + +pub(super) fn select( + roster: &FleetRoster, + leader_node: NodeId, + retired: &SealedNodeLog, +) -> Result> { + let rows = roster + .enrollments() + .iter() + .filter(|row| { + row.spec().source.is_some_and(|source| source.session == retired.session()) + && matches!(row.spec().role, EnrollmentRole::Follower { log_epoch } if log_epoch == retired.log().epoch()) + && row.status() != EnrollmentStatus::Refused + }) + .collect::>(); + if rows.len() != retired.log().members().len() || rows.len() > 16 { + return Err(Error::Control( + "recovered follower enrollment set is incomplete", + )); + } + let mut members = Vec::with_capacity(rows.len()); + let mut source = None; + for member in retired.log().members() { + let mut matches = rows.iter().filter(|row| row.spec().target.node == *member); + let row = matches.next().ok_or(Error::Fenced)?; + let endpoint = row.spec().source.ok_or(Error::Fenced)?; + if matches.next().is_some() + || endpoint.node != leader_node + || source.is_some_and(|original| original != endpoint) + || row.spec().scope != roster.snapshot().head().scope() + { + return Err(Error::Fenced); + } + source = Some(endpoint); + row.to_bytes().map_err(operation)?; + members.push((*row).clone()); + } + Ok(members) +} + +fn authority(hash: &mut blake3::Hasher, leader_node: NodeId, retired: &SealedNodeLog) { + hash.update(leader_node.as_bytes()); + hash.update(retired.session().as_bytes()); + hash.update(&retired.log().epoch().to_be_bytes()); + hash.update(&[u8::from(retired.log().active())]); + hash.update(&retired.log().tiered_through().to_be_bytes()); + hash.update(&(retired.log().members().len() as u64).to_be_bytes()); + for member in retired.log().members() { + hash.update(member.as_bytes()); + } + hash.update(&[u8::from(retired.log().recovery_manifest().is_some())]); + if let Some(manifest) = retired.log().recovery_manifest() { + hash.update(manifest.as_bytes()); + } +} + +pub(super) fn evidence( + leader_node: NodeId, + retired: &SealedNodeLog, + row: &EnrollmentRecord, +) -> Result { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-recovered-follower-retirement.v1\0"); + authority(&mut hash, leader_node, retired); + let spec = row.spec().to_bytes().map_err(operation)?; + hash.update(&(spec.len() as u64).to_be_bytes()); + hash.update(&spec); + hash.update(&row.accepted_at_ms().to_be_bytes()); + // Establishment can race the closing transaction. Its mutable evidence and + // current timestamps cannot change this original request's retirement event. + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) +} + +pub(super) fn validate_terminal(row: &EnrollmentRecord, evidence: Digest) -> Result<()> { + if row.status() == EnrollmentStatus::Retired && row.settlement_evidence() != Some(evidence) { + return Err(Error::Control( + "recovered follower settlement evidence differs", + )); + } + Ok(()) +} + +pub(super) fn validate_result( + original: &EnrollmentRecord, + result: &EnrollmentRecord, + evidence: Digest, +) -> Result<()> { + result.validate_replay(original.spec()).map_err(operation)?; + result.to_bytes().map_err(operation)?; + if result.status() != EnrollmentStatus::Retired + || result.accepted_at_ms() != original.accepted_at_ms() + || result.updated_at_ms() < original.updated_at_ms() + || result.settlement_evidence() != Some(evidence) + || original + .established_evidence() + .is_some_and(|old| result.established_evidence() != Some(old)) + { + return Err(Error::Fenced); + } + Ok(()) +} + +pub(super) fn closure_digest( + leader_node: NodeId, + retired: &SealedNodeLog, + rows: &[EnrollmentRecord], +) -> Result { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-recovered-follower-closure.v1\0"); + authority(&mut hash, leader_node, retired); + for row in rows { + let encoded = row.to_bytes().map_err(operation)?; + hash.update(&(encoded.len() as u64).to_be_bytes()); + hash.update(&encoded); + } + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) +} diff --git a/crates/cellule-host/src/fleet/references/mod.rs b/crates/cellule-host/src/fleet/references/mod.rs new file mode 100644 index 00000000..3934f055 --- /dev/null +++ b/crates/cellule-host/src/fleet/references/mod.rs @@ -0,0 +1,203 @@ +//! Complete current directory references beneath the application's observer. +use super::{FleetJournalSnapshot, FleetRoster}; +use cellule_runtime::{ + Error, Result, + fleet::operations::{EnrollmentRole, EnrollmentStatus}, + identity::{Digest, NodeId}, + node::{FollowerLogObservation, NodeDirectory, log_state::NodeLogPhase}, +}; +use tokio::time::{Instant, timeout_at}; + +mod traversal; + +/// All authoritative log references to one physical follower, including failed +/// leader sessions. Native lanes, original producer work, replacement policy +/// and canonical recovery/retirement remain separate obligations. +/// +/// Applications account the bounded copied buffer (at most 10,000 entries). +/// A complete listing is interval evidence, not an atomic directory snapshot or +/// permission to shut down. Recheck it after native collection and reconfirm +/// the complete roster before using it in a fleet observation. +pub struct FleetFollowerReferences { + member: NodeId, + roster: Digest, + snapshot: FleetJournalSnapshot, + topology: Digest, + started_at_ms: i64, + finished_at_ms: i64, + collected_at_ms: i64, + rechecked: Option<(i64, i64)>, + entries: Vec, +} + +impl FleetFollowerReferences { + /// Traverses native continuation pages against the original full roster. + /// The supplied clock records actual capture times; it must not renew or + /// restamp original observations. Source errors survive deadline wrapping. + /// This performs no enrollment, rotation, recovery or retirement effect. + pub async fn collect( + directory: &NodeDirectory, + roster: &FleetRoster, + member: NodeId, + page_limit: usize, + deadline: Instant, + mut clock: impl FnMut() -> Result, + ) -> Result { + if roster.snapshot().registry().bootstrap_revision().is_none() + || directory.fleet() != roster.snapshot().head().scope().fleet + || !roster + .intents() + .iter() + .any(|intent| intent.node() == member) + || !(1..=128).contains(&page_limit) + { + return Err(Error::Fenced); + } + let mut scan = traversal::Scan::default(); + loop { + if Instant::now() >= deadline { + return Err(Error::Deadline); + } + let now = clock()?; + let page = timeout_at( + deadline, + directory.follower_logs_page(member, scan.next, page_limit, now), + ) + .await + .map_err(|source| Error::Facility { + name: "fleet-log-inventory-deadline", + source: Box::new(source), + })??; + let done = scan.accept(member, page, now, clock()?)?; + if done { + break; + } + } + Ok(Self { + member, + roster: roster.digest()?, + snapshot: roster.snapshot().clone(), + topology: scan.topology.ok_or(Error::Fenced)?, + started_at_ms: scan.started_at_ms.ok_or(Error::Fenced)?, + finished_at_ms: scan.finished_at_ms, + collected_at_ms: scan.finished_at_ms, + rechecked: None, + entries: scan.entries, + }) + } + + /// Physical follower node; no current boot substitutes an earlier lane. + #[must_use] + pub const fn member(&self) -> NodeId { + self.member + } + + /// Original-to-latest checked capture times, never refreshed row timestamps. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + (self.started_at_ms, self.finished_at_ms) + } + pub(crate) fn coverage_checkpoint(&self) -> (i64, Option<(i64, i64)>) { + (self.collected_at_ms, self.rechecked) + } + pub(crate) fn coverage_digest(&self) -> Digest { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-foreign-coverage.v1\0"); + hash.update(self.member.as_bytes()); + hash.update(self.roster.as_bytes()); + hash.update(self.topology.as_bytes()); + for row in &self.entries { + hash.update(row.leader.as_bytes()); + hash.update(row.leader_node.as_bytes()); + hash.update(&[ + row.leader_state as u8, + row.log.phase() as u8, + u8::from(row.log.active()), + ]); + hash.update(&row.log.epoch().to_be_bytes()); + hash.update(&row.log.tiered_through().to_be_bytes()); + hash.update(&(row.log.members().len() as u64).to_be_bytes()); + for member in row.log.members() { + hash.update(member.as_bytes()); + } + hash.update(&[u8::from(row.log.recovery().is_some())]); + if let Some(claim) = row.log.recovery() { + hash.update(claim.claimant().as_bytes()); + hash.update(&claim.generation().to_be_bytes()); + hash.update(&claim.expires_at_ms().to_be_bytes()); + } + hash.update(&[u8::from(row.log.recovery_manifest().is_some())]); + if let Some(manifest) = row.log.recovery_manifest() { + hash.update(manifest.as_bytes()); + } + } + Digest::from_bytes(*hash.finalize().as_bytes()) + } + + /// Exact current authority observations in strict leader-session order. + #[must_use] + pub fn entries(&self) -> &[FollowerLogObservation] { + &self.entries + } + + /// Matches discovered references to retained original enrollment requests. + /// Pending requests without a reference remain obligations in the roster; + /// absence here cannot establish their nonexecution or failed-process join. + pub fn validate_enrollments(&self, roster: &FleetRoster) -> Result<()> { + if roster.snapshot() != &self.snapshot || roster.digest()? != self.roster { + return Err(Error::Fenced); + } + for reference in &self.entries { + if !roster.enrollments().iter().any(|row| { + let spec = row.spec(); + spec.target.node == self.member + && (row.unresolved() + || (row.status() == EnrollmentStatus::Retired + && reference.log.phase() == NodeLogPhase::Retired)) + && spec.source.is_some_and(|source| { + source.node == reference.leader_node && source.session == reference.leader + }) + && matches!(spec.role, EnrollmentRole::Follower { log_epoch } + if log_epoch == reference.log.epoch()) + }) { + return Err(Error::Control( + "authoritative follower reference is unregistered", + )); + } + } + Ok(()) + } + + /// Fully traverses again and compares exact rows, including volatile + /// coverage and leader liveness which native topology cursors omit. A first + /// page fingerprint alone cannot prove those authority fields unchanged. + /// On failure the original observations and interval remain intact. + pub async fn recheck( + &mut self, + directory: &NodeDirectory, + roster: &FleetRoster, + page_limit: usize, + deadline: Instant, + clock: impl FnMut() -> Result, + ) -> Result<()> { + self.rechecked = None; + if roster.snapshot() != &self.snapshot || roster.digest()? != self.roster { + return Err(Error::Fenced); + } + let fresh = + Self::collect(directory, roster, self.member, page_limit, deadline, clock).await?; + if fresh.started_at_ms < self.finished_at_ms + || fresh.finished_at_ms - self.started_at_ms > 30_000 + || fresh.topology != self.topology + || fresh.entries != self.entries + { + return Err(Error::Node("authoritative follower inventory changed")); + } + self.finished_at_ms = fresh.finished_at_ms; + self.rechecked = Some((fresh.started_at_ms, fresh.finished_at_ms)); + Ok(()) + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-host/src/fleet/references/tests.rs b/crates/cellule-host/src/fleet/references/tests.rs new file mode 100644 index 00000000..6aa0dad4 --- /dev/null +++ b/crates/cellule-host/src/fleet/references/tests.rs @@ -0,0 +1,133 @@ +//! Directory pagination contracts; managed native roles use the public example. +use super::traversal::Scan; +use super::*; +use cellule_runtime::{ + identity::SessionId, + ltx::CellStorageLayout, + node::{NodeAdvertisement, NodeCapacity, NodeFailureDomain}, +}; +use cellule_store::Store; +use ed25519_dalek::SigningKey; +use object_store::{memory::InMemory, path::Path}; +use std::sync::Arc; + +const NOW: i64 = 1_000_000; +fn node(index: u8) -> NodeId { + NodeId::from_bytes([index; 16]) +} +fn session(index: u8) -> SessionId { + SessionId::from_bytes([index; 16]) +} + +async fn directory() -> NodeDirectory { + let directory = NodeDirectory::new( + CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("references"), + [3; 16], + ), + Digest::from_bytes([2; 32]), + Digest::from_bytes([4; 32]), + Digest::from_bytes([5; 32]), + ); + for index in 1..=3 { + directory + .create( + NodeAdvertisement::sign( + node(index), + session(index), + format!("https://node-{index}.example"), + directory.fleet(), + Digest::from_bytes([3; 32]), + Digest::from_bytes([4; 32]), + Digest::from_bytes([5; 32]), + &SigningKey::from_bytes(&[index; 32]), + 1, + NOW, + NOW + 30_000, + vec![Digest::from_bytes([6; 32])], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + follower_free_bytes: 1 << 20, + free_memory_bytes: 1 << 20, + free_disk_bytes: 1 << 20, + job_credits: 4, + log_protocol: 1, + ..Default::default() + }, + ) + .unwrap(), + NOW, + ) + .await + .unwrap(); + } + for index in [2, 1] { + let owner = directory.load(session(index), NOW).await.unwrap().unwrap(); + let prepared = directory + .prepare_log_enrollment(&owner, 1, 1, 3, NOW) + .await + .unwrap() + .unwrap(); + let attempt = directory + .prepare_log_enrollment_attempt(&prepared, NOW) + .await + .unwrap(); + directory + .commit_log_enrollment(&attempt, NOW) + .await + .unwrap(); + } + directory +} + +#[tokio::test] +async fn canonical_continuations_collect_every_leader_in_order() { + let directory = directory().await; + let mut scan = Scan::default(); + let first = directory + .follower_logs_page(node(3), None, 1, NOW) + .await + .unwrap(); + assert_eq!(first.total_logs(), 2); + assert_eq!(first.entries()[0].leader, session(1)); + assert!(!scan.accept(node(3), first, NOW, NOW + 1).unwrap()); + let second = directory + .follower_logs_page(node(3), scan.next, 1, NOW + 2) + .await + .unwrap(); + assert!(scan.accept(node(3), second, NOW + 2, NOW + 3).unwrap()); + assert_eq!( + scan.entries + .iter() + .map(|row| row.leader) + .collect::>(), + vec![session(1), session(2)] + ); + assert_eq!(scan.started_at_ms, Some(NOW)); + assert_eq!(scan.finished_at_ms, NOW + 3); +} + +#[tokio::test] +async fn foreign_member_and_regressed_capture_cannot_complete_a_traversal() { + let directory = directory().await; + let first = directory + .follower_logs_page(node(3), None, 1, NOW) + .await + .unwrap(); + assert!(Scan::default().accept(node(2), first, NOW, NOW).is_err()); + let mut scan = Scan::default(); + let first = directory + .follower_logs_page(node(3), None, 1, NOW) + .await + .unwrap(); + assert!(!scan.accept(node(3), first, NOW, NOW + 2).unwrap()); + let next = directory + .follower_logs_page(node(3), scan.next, 1, NOW + 1) + .await + .unwrap(); + assert!(scan.accept(node(3), next, NOW + 1, NOW + 3).is_err()); + assert_eq!(scan.entries.len(), 1); + assert!(scan.next.is_some()); +} diff --git a/crates/cellule-host/src/fleet/references/traversal.rs b/crates/cellule-host/src/fleet/references/traversal.rs new file mode 100644 index 00000000..6d180535 --- /dev/null +++ b/crates/cellule-host/src/fleet/references/traversal.rs @@ -0,0 +1,69 @@ +use super::*; +use cellule_runtime::node::{LogInventoryCursor, LogInventoryPage}; + +#[derive(Default)] +pub(super) struct Scan { + pub next: Option, + pub topology: Option, + pub started_at_ms: Option, + pub finished_at_ms: i64, + pub entries: Vec, + total: Option, +} + +impl Scan { + pub fn accept( + &mut self, + member: NodeId, + page: LogInventoryPage, + started: i64, + finished: i64, + ) -> Result { + if page.member() != member + || page.observed_at_ms() != started + || started < 0 + || started < self.finished_at_ms + || finished < started + || finished - self.started_at_ms.unwrap_or(started) > 30_000 + || self.topology.is_some_and(|old| old != page.topology()) + || self.total.is_some_and(|old| old != page.total_logs()) + || page.total_logs() > 10_000 + { + return Err(Error::Node("authoritative follower page differs")); + } + let total = self + .entries + .len() + .checked_add(page.entries().len()) + .filter(|total| *total <= page.total_logs() && *total <= 10_000) + .ok_or(Error::Capacity("authoritative follower row bound"))?; + let mut after = self.entries.last().map(|row| *row.leader.as_bytes()); + for row in page.entries() { + if !row.log.members().contains(&member) + || after.is_some_and(|last| last >= *row.leader.as_bytes()) + { + return Err(Error::Fenced); + } + after = Some(*row.leader.as_bytes()); + } + if let Some(next) = page.next() { + let encoded = next.to_bytes(); + if page.entries().is_empty() + || total >= page.total_logs() + || encoded[..32] != *page.topology().as_bytes() + || after.is_none_or(|last| encoded[32..] != last) + { + return Err(Error::Node("authoritative follower continuation differs")); + } + } else if total != page.total_logs() { + return Err(Error::Node("authoritative follower traversal incomplete")); + } + self.topology = Some(page.topology()); + self.total = Some(page.total_logs()); + self.started_at_ms.get_or_insert(started); + self.finished_at_ms = finished; + self.entries.extend_from_slice(page.entries()); + self.next = page.next(); + Ok(self.next.is_none()) + } +} diff --git a/crates/cellule-host/src/fleet/roster/mod.rs b/crates/cellule-host/src/fleet/roster/mod.rs new file mode 100644 index 00000000..b91d6be5 --- /dev/null +++ b/crates/cellule-host/src/fleet/roster/mod.rs @@ -0,0 +1,326 @@ +//! Complete durable roster traversal in the journal's shared transaction domain. + +use std::{ + collections::{HashMap, HashSet}, + future::Future, +}; + +use cellule_runtime::cell::actor::{CellRuntime, NodeByteReservation}; +use cellule_runtime::fleet::operations::{ + EnrollmentRecord, EnrollmentRole, EnrollmentStatus, MAX_PAGE_BYTES, MAX_RECORD_BYTES, + NodeIntent, +}; +use cellule_runtime::identity::{Digest, NodeId, SessionId}; +use cellule_runtime::node::NodeAdvertisement; +use cellule_runtime::{Error, Result}; +use tokio::time::{Instant, timeout_at}; + +use super::{FleetAdapterFuture, FleetJournal, FleetJournalSnapshot, operation}; + +mod scan; +use scan::Scan; + +const MAX_ROSTER_ENTRIES: usize = 10_000; +const PAGE_ENTRIES: usize = 128; + +/// Exact physical boot required by an unresolved registry responsibility. +/// Multiple boots on the same physical node remain distinct after replacement. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct FleetRosterBoot { + /// Physical node, including an earlier failed boot's identity. + pub node: NodeId, + /// Exact participating process session; never substituted with its successor. + pub session: SessionId, +} + +/// Fully traversed retained intents and enrollments at one journal snapshot. +/// +/// Construction rechecks the entire head and registry, not merely a live node +/// listing. Pages use the canonical 128-row/one-MiB codecs, and each collection +/// is capped at 10,000 rows. Applications account their retained collector +/// buffers. This value proves traversal under the adapter's consistent-page +/// contract; bootstrap, authentication, native-role matching and fresh authority +/// are separate requirements. In particular, it is not a finalization proof. +pub struct FleetRoster { + snapshot: FleetJournalSnapshot, + intents: Vec, + enrollments: Vec, + // Data drops before its retained-byte permits. Application-owned public + // collection keeps this empty; native collection uses the shared ledger. + _memory: Vec, +} + +impl FleetRoster { + /// Reads every page, including Pending, failed-boot and terminal records. + /// An expired deadline fails before issuing another journal request. A lost + /// read reply preserves its source error and never returns a partial roster. + pub async fn collect( + journal: &dyn FleetJournal, + expected: &FleetJournalSnapshot, + deadline: Instant, + ) -> Result { + Self::collect_inner(journal, expected, deadline, None).await + } + + pub(crate) async fn collect_admitted( + journal: &dyn FleetJournal, + expected: &FleetJournalSnapshot, + deadline: Instant, + runtime: &CellRuntime, + ) -> Result { + Self::collect_inner(journal, expected, deadline, Some(runtime)).await + } + + async fn collect_inner( + journal: &dyn FleetJournal, + expected: &FleetJournalSnapshot, + deadline: Instant, + runtime: Option<&CellRuntime>, + ) -> Result { + let header = runtime + .map(|runtime| { + runtime.try_reserve_node_metadata_bytes(2 * MAX_RECORD_BYTES as usize + 4096) + }) + .transpose()?; + confirm(journal, expected, deadline).await?; + let version = expected.registry(); + let mut scan = Scan::new(version); + scan.runtime = runtime; + scan.memory.extend(header); + let mut after = None; + loop { + let _page = reserve_page(runtime)?; + let page = call(deadline, || { + journal.intents_page(version, after, PAGE_ENTRIES) + }) + .await?; + after = scan.intents(page, after)?; + if after.is_none() { + break; + } + } + let mut after = None; + loop { + let _page = reserve_page(runtime)?; + let page = call(deadline, || { + journal.enrollments_page(version, after, PAGE_ENTRIES) + }) + .await?; + after = scan.enrollments(page, after)?; + if after.is_none() { + break; + } + } + confirm(journal, expected, deadline).await?; + Ok(Self { + snapshot: expected.clone(), + intents: scan.intents, + enrollments: scan.enrollments, + _memory: scan.memory, + }) + } + + /// Rechecks the original full snapshot after collecting native observations. + /// A later finalization transaction must compare these versions again. + pub async fn confirm(&self, journal: &dyn FleetJournal, deadline: Instant) -> Result<()> { + confirm(journal, &self.snapshot, deadline).await + } + + /// Returns the immutable head and registry barrier used by every page. + #[must_use] + pub const fn snapshot(&self) -> &FleetJournalSnapshot { + &self.snapshot + } + + /// Returns all physical intents, in strictly ascending node order. + #[must_use] + pub fn intents(&self) -> &[NodeIntent] { + &self.intents + } + + /// Returns all enrollment records in strictly ascending stable-key order. + /// Terminal exclusion and retirement records are retained without renewal. + #[must_use] + pub fn enrollments(&self) -> &[EnrollmentRecord] { + &self.enrollments + } + + pub(crate) fn boot( + &self, + node: NodeId, + session: SessionId, + ) -> Result { + let intent = self + .intents + .iter() + .find(|intent| intent.node() == node) + .ok_or(Error::Fenced)?; + let mut boots = self.enrollments.iter().filter(|row| { + row.unresolved() + && matches!(row.spec().role, EnrollmentRole::Node { .. }) + && row.spec().target.node == node + && row.spec().target.session == session + }); + let boot = boots.next().ok_or(Error::Fenced)?; + if boots.next().is_some() { + return Err(Error::Fenced); + } + super::FleetBootObservation::new(intent.clone(), boot.clone()).map_err(operation) + } + + /// Returns both endpoints of every unresolved responsibility, without a + /// liveness filter. Pending enrollment still requires observation even when + /// its native effect or acceptance reply is unknown. Terminal rows do not + /// introduce required boots, but remain part of the roster and its digest. + #[must_use] + pub fn required_boots(&self) -> Vec { + let mut boots = Vec::new(); + for record in self.enrollments.iter().filter(|record| record.unresolved()) { + let spec = record.spec(); + for endpoint in spec.source.into_iter().chain(std::iter::once(spec.target)) { + boots.push(FleetRosterBoot { + node: endpoint.node, + session: endpoint.session, + }); + } + } + boots.sort_unstable_by_key(|boot| (*boot.node.as_bytes(), *boot.session.as_bytes())); + boots.dedup(); + boots + } + + /// Checks signed advertisement coverage against established durable boots. + /// Missing/unknown/duplicate boots, unresolved boot acceptance, old boots + /// after physical-node replacement, or absent bootstrap yield incomplete + /// coverage. Both endpoints of unresolved reader/follower responsibilities + /// remain required. Applications additionally pin signing keys, prove that + /// unexpected live records were discovered, and match every native role; + /// success here supplies only the registry-to-advertisement part of that + /// proof. It never establishes finalization or replacement redundancy. + pub fn covers_advertisements(&self, nodes: &[NodeAdvertisement], now_ms: i64) -> Result { + if self.snapshot.registry().bootstrap_revision().is_none() + || nodes.len() > MAX_ROSTER_ENTRIES + { + return Ok(false); + } + let mut established = HashSet::new(); + let intents = self + .intents + .iter() + .map(|intent| (intent.node(), intent)) + .collect::>(); + for record in &self.enrollments { + if !matches!(record.spec().role, EnrollmentRole::Node { .. }) || !record.unresolved() { + continue; + } + let endpoint = record.spec().target; + if record.status() != EnrollmentStatus::Established + || !established.insert((endpoint.node, endpoint.session)) + || intents.get(&endpoint.node).is_none_or(|intent| { + super::FleetBootObservation::new((**intent).clone(), record.clone()).is_err() + }) + { + return Ok(false); + } + } + let mut advertised = HashSet::new(); + let mut physical = HashSet::new(); + for node in nodes { + cellule_runtime::fleet::placement::PlacementObservation::from_signed_advertisement( + node, now_ms, false, + )?; + let boot = (node.node(), node.session()); + if node.fleet() != self.snapshot.head().scope().fleet + || !advertised.insert(boot) + || !physical.insert(node.node()) + || !established.contains(&boot) + || intents + .get(&node.node()) + .is_none_or(|intent| intent.session() != node.session()) + { + return Ok(false); + } + } + Ok(advertised == established + && self + .required_boots() + .iter() + .all(|boot| advertised.contains(&(boot.node, boot.session)))) + } + + /// Identifies every original row, including evidence, status and timestamps. + /// This does not prove that native observations are atomic or authenticated. + pub fn digest(&self) -> Result { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-retained-roster.v1\0"); + let mut field = |bytes: &[u8]| { + hash.update(&(bytes.len() as u64).to_be_bytes()); + hash.update(bytes); + }; + field(&self.snapshot.head().to_bytes().map_err(operation)?); + field(&self.snapshot.registry().to_bytes().map_err(operation)?); + field(&(self.intents.len() as u64).to_be_bytes()); + for intent in &self.intents { + field(&intent.to_bytes().map_err(operation)?); + } + field(&(self.enrollments.len() as u64).to_be_bytes()); + for record in &self.enrollments { + field(&record.to_bytes().map_err(operation)?); + } + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) + } +} + +fn reserve_page(runtime: Option<&CellRuntime>) -> Result> { + // Before journal I/O, cover the bounded source page, canonical encoding, + // validation temporaries and overlap while cloning admitted retained rows. + runtime + .map(|runtime| runtime.try_reserve_node_metadata_bytes(4 * MAX_PAGE_BYTES as usize)) + .transpose() +} + +async fn confirm( + journal: &dyn FleetJournal, + expected: &FleetJournalSnapshot, + deadline: Instant, +) -> Result<()> { + let current = call(deadline, || journal.load_snapshot(expected.head().scope())).await?; + if current != *expected { + return Err(operation( + cellule_runtime::fleet::operations::OperationError::Conflict, + )); + } + Ok(()) +} + +async fn call<'a, T, F>(deadline: Instant, request: F) -> Result +where + F: FnOnce() -> FleetAdapterFuture<'a, T>, +{ + // Construct the adapter future only after checking admission. An adapter may + // retain native read work when a dispatched waiter times out; it still joins + // that work through its ordinary shutdown path. + if Instant::now() >= deadline { + return Err(Error::Node("fleet roster collection deadline elapsed")); + } + wait(deadline, request()).await +} + +async fn wait( + deadline: Instant, + request: impl Future>>, +) -> Result { + timeout_at(deadline, request) + .await + .map_err(|source| Error::Facility { + name: "fleet-roster-deadline", + source: Box::new(source), + })? + .map_err(|source| Error::Facility { + name: "fleet-roster-journal", + source, + }) +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-host/src/fleet/roster/scan.rs b/crates/cellule-host/src/fleet/roster/scan.rs new file mode 100644 index 00000000..48894042 --- /dev/null +++ b/crates/cellule-host/src/fleet/roster/scan.rs @@ -0,0 +1,86 @@ +use super::*; +use cellule_runtime::fleet::operations::{ + EnrollmentPage, IntentPage, OperationError, RegistryVersion, +}; + +pub(super) struct Scan<'a> { + version: RegistryVersion, + pub(super) intents: Vec, + pub(super) enrollments: Vec, + pub(super) runtime: Option<&'a CellRuntime>, + pub(super) memory: Vec, +} + +impl Scan<'_> { + pub(super) fn new(version: RegistryVersion) -> Self { + Self { + version, + intents: Vec::new(), + enrollments: Vec::new(), + runtime: None, + memory: Vec::new(), + } + } + + pub(super) fn intents( + &mut self, + page: IntentPage, + after: Option, + ) -> Result> { + // Encode through the canonical page validator: in-process adapters obey + // the same shape and byte bound as transported pages. + let bytes = page.to_bytes().map_err(operation)?; + if page.version() != self.version + || page.after() != after + || self.intents.last().map(NodeIntent::node) != after + { + return Err(operation(OperationError::Conflict)); + } + if page.entries().len() > MAX_ROSTER_ENTRIES.saturating_sub(self.intents.len()) { + return Err(Error::Capacity("fleet intent roster bound")); + } + self.retain::(page.entries().len(), bytes.len())?; + self.intents.extend_from_slice(page.entries()); + Ok(page.next()) + } + + pub(super) fn enrollments( + &mut self, + page: EnrollmentPage, + after: Option, + ) -> Result> { + let bytes = page.to_bytes().map_err(operation)?; + let previous = self + .enrollments + .last() + .map(|record| record.spec().key()) + .transpose() + .map_err(operation)?; + if page.version() != self.version || page.after() != after || previous != after { + return Err(operation(OperationError::Conflict)); + } + if page.entries().len() > MAX_ROSTER_ENTRIES.saturating_sub(self.enrollments.len()) { + return Err(Error::Capacity("fleet enrollment roster bound")); + } + self.retain::(page.entries().len(), bytes.len())?; + self.enrollments.extend_from_slice(page.entries()); + Ok(page.next()) + } + + fn retain(&mut self, entries: usize, encoded_bytes: usize) -> Result<()> { + if let Some(runtime) = self.runtime + && entries != 0 + { + // Three times fixed row/token storage covers spare Vec capacity and + // old/new allocations during growth. Encoded dynamic fields retain + // a separate conservative charge. Reserve before cloning any row. + let memory = runtime.try_reserve_node_metadata_bytes( + 3 * entries * std::mem::size_of::() + + 2 * encoded_bytes + + 3 * std::mem::size_of::(), + )?; + self.memory.push(memory); + } + Ok(()) + } +} diff --git a/crates/cellule-host/src/fleet/roster/tests.rs b/crates/cellule-host/src/fleet/roster/tests.rs new file mode 100644 index 00000000..d3c2a085 --- /dev/null +++ b/crates/cellule-host/src/fleet/roster/tests.rs @@ -0,0 +1,465 @@ +use super::*; +use cellule_runtime::fleet::operations::{ + EnrollmentEndpoint, EnrollmentPage, EnrollmentSpec, FleetHead, FleetScope, IntentPage, + RegistryVersion, +}; +use cellule_runtime::identity::ApplicationId; +use cellule_runtime::node::NodeMode; + +fn scope() -> FleetScope { + FleetScope { + fleet: Digest::from_bytes([1; 32]), + application: ApplicationId::from_bytes([2; 16]), + } +} +fn intent(n: u8) -> NodeIntent { + NodeIntent::initial( + scope(), + NodeId::from_bytes([n; 16]), + SessionId::from_bytes([n; 16]), + ) + .unwrap() +} +fn endpoint(n: u8) -> EnrollmentEndpoint { + EnrollmentEndpoint { + node: intent(n).node(), + session: intent(n).session(), + intent_revision: 1, + } +} +fn version() -> RegistryVersion { + RegistryVersion::new(scope()).unwrap().advance(0).unwrap() +} +fn record(n: u8) -> EnrollmentRecord { + EnrollmentRecord::pending( + EnrollmentSpec { + scope: scope(), + request: Digest::from_bytes([n; 32]), + role: EnrollmentRole::Follower { log_epoch: 7 }, + source: Some(endpoint(1)), + target: endpoint(2), + }, + Some(&intent(1)), + &intent(2), + 1, + ) + .unwrap() +} + +#[test] +fn scans_require_original_version_and_exact_continuation() { + let version = version(); + let mut scan = Scan::new(version); + let first = IntentPage::new(version, None, vec![intent(1)], Some(intent(1).node())).unwrap(); + let after = scan.intents(first, None).unwrap(); + assert!( + scan.intents( + IntentPage::new(version, None, vec![intent(2)], None).unwrap(), + None + ) + .is_err() + ); + assert!( + scan.intents( + IntentPage::new(version.advance(1).unwrap(), after, vec![intent(2)], None).unwrap(), + after + ) + .is_err() + ); + assert_eq!(scan.intents.len(), 1); + assert_eq!( + scan.intents( + IntentPage::new(version, after, vec![intent(2)], None).unwrap(), + after + ) + .unwrap(), + None + ); + assert_eq!(scan.intents, vec![intent(1), intent(2)]); +} + +#[test] +fn enrollment_scan_rejects_reset_cursor_and_revision_change_without_appending() { + let mut rows = [record(3), record(4)]; + rows.sort_by_key(|record| *record.spec().key().unwrap().as_bytes()); + let mut scan = Scan::new(version()); + let key = rows[0].spec().key().unwrap(); + let after = scan + .enrollments( + EnrollmentPage::new(version(), None, vec![rows[0].clone()], Some(key)).unwrap(), + None, + ) + .unwrap(); + assert!( + scan.enrollments( + EnrollmentPage::new(version(), None, vec![rows[1].clone()], None).unwrap(), + None + ) + .is_err() + ); + assert!( + scan.enrollments( + EnrollmentPage::new( + version().advance(1).unwrap(), + after, + vec![rows[1].clone()], + None + ) + .unwrap(), + after + ) + .is_err() + ); + assert_eq!(scan.enrollments.len(), 1); + scan.enrollments( + EnrollmentPage::new(version(), after, vec![rows[1].clone()], None).unwrap(), + after, + ) + .unwrap(); + assert_eq!(scan.enrollments, rows); +} + +#[test] +fn aggregate_limits_fail_before_appending_a_page() { + let mut scan = Scan::new(version()); + scan.intents = vec![intent(1); MAX_ROSTER_ENTRIES]; + let after = Some(intent(1).node()); + assert!(matches!( + scan.intents( + IntentPage::new(version(), after, vec![intent(2)], None).unwrap(), + after + ), + Err(Error::Capacity(_)) + )); + assert_eq!(scan.intents.len(), MAX_ROSTER_ENTRIES); + let mut rows = [record(3), record(4)]; + rows.sort_by_key(|record| *record.spec().key().unwrap().as_bytes()); + scan.enrollments = vec![rows[0].clone(); MAX_ROSTER_ENTRIES]; + let after = Some(rows[0].spec().key().unwrap()); + assert!(matches!( + scan.enrollments( + EnrollmentPage::new(version(), after, vec![rows[1].clone()], None).unwrap(), + after + ), + Err(Error::Capacity(_)) + )); + assert_eq!(scan.enrollments.len(), MAX_ROSTER_ENTRIES); +} + +#[test] +fn original_failed_boot_remains_required_after_intent_replacement() { + let old = record(3); + let replaced = intent(1) + .rebind_active(SessionId::from_bytes([9; 16]), 2) + .unwrap(); + let mut roster = FleetRoster { + snapshot: FleetJournalSnapshot::new(FleetHead::new(scope(), 0).unwrap(), version()) + .unwrap(), + intents: vec![replaced, intent(2)], + enrollments: vec![old.clone()], + _memory: Vec::new(), + }; + assert_eq!( + roster.required_boots(), + vec![ + FleetRosterBoot { + node: endpoint(1).node, + session: endpoint(1).session + }, + FleetRosterBoot { + node: endpoint(2).node, + session: endpoint(2).session + } + ] + ); + let pending_digest = roster.digest().unwrap(); + roster.enrollments[0] = old.refuse(Digest::from_bytes([8; 32]), 2).unwrap(); + assert!(roster.required_boots().is_empty()); + assert_ne!(roster.digest().unwrap(), pending_digest); + assert_eq!(roster.enrollments().len(), 1); + assert!(!roster.covers_advertisements(&[], 100).unwrap()); + // Even a known initial boot must not become a coverage claim before bootstrap. + let boot = EnrollmentSpec { + scope: scope(), + request: Digest::from_bytes([10; 32]), + role: EnrollmentRole::Node { + mode: NodeMode::Active, + }, + source: None, + target: endpoint(2), + }; + roster + .enrollments + .push(EnrollmentRecord::pending(boot, None, &intent(2), 1).unwrap()); + assert!(!roster.covers_advertisements(&[], 100).unwrap()); +} + +#[tokio::test] +async fn expired_deadline_does_not_construct_or_dispatch_a_journal_request() { + let mut called = false; + let result = call::<(), _>(Instant::now(), || { + called = true; + Box::pin(async { Ok(()) }) + }) + .await; + assert!(result.is_err()); + assert!(!called); +} + +#[tokio::test] +async fn journal_source_errors_remain_available_without_reclassification() { + let result = call::<(), _>(Instant::now() + std::time::Duration::from_secs(1), || { + Box::pin(async { + Err(std::io::Error::new( + std::io::ErrorKind::PermissionDenied, + "original journal failure", + ) + .into()) + }) + }) + .await; + let Err(Error::Facility { name, source }) = result else { + panic!("missing source error") + }; + assert_eq!(name, "fleet-roster-journal"); + assert_eq!( + source.downcast_ref::().unwrap().kind(), + std::io::ErrorKind::PermissionDenied + ); + assert_eq!(source.to_string(), "original journal failure"); +} + +fn signed(n: u8) -> NodeAdvertisement { + use cellule_runtime::node::{ + NodeCapacity, NodeFailureDomain, NodeOperationalSample, NodePlacementCapacity, NodePressure, + }; + let key = ed25519_dalek::SigningKey::from_bytes(&[n; 32]); + NodeAdvertisement::sign( + endpoint(n).node, + endpoint(n).session, + format!("https://node-{n}.internal:8789"), + scope().fleet, + Digest::from_bytes([5; 32]), + Digest::from_bytes([6; 32]), + Digest::from_bytes([7; 32]), + &key, + 1, + 100, + 30_100, + vec![Digest::from_bytes([8; 32])], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + free_memory_bytes: 1000, + free_disk_bytes: 1000, + job_credits: 2, + log_protocol: 1, + ..NodeCapacity::default() + }, + ) + .unwrap() + .with_operational_placement( + NodePlacementCapacity { + memory_capacity_bytes: 1000, + disk_capacity_bytes: 1000, + max_active_cells: 2, + job_capacity: 2, + ..NodePlacementCapacity::default() + }, + NodeOperationalSample { + sequence: 1, + observed_at_ms: 100, + mode: NodeMode::Active, + pressure: NodePressure::Normal, + }, + &key, + ) + .unwrap() +} +fn boot(n: u8) -> EnrollmentRecord { + EnrollmentRecord::pending( + EnrollmentSpec { + scope: scope(), + request: Digest::from_bytes([n + 10; 32]), + role: EnrollmentRole::Node { + mode: NodeMode::Active, + }, + source: None, + target: endpoint(n), + }, + None, + &intent(n), + 1, + ) + .unwrap() +} + +#[test] +fn boot_coverage_requires_exact_established_roster_and_fresh_signed_samples() { + let record = boot(1).establish(Digest::from_bytes([22; 32]), 2).unwrap(); + let mut roster = FleetRoster { + snapshot: FleetJournalSnapshot::new( + FleetHead::new(scope(), 0).unwrap(), + version().bootstrap(1).unwrap(), + ) + .unwrap(), + intents: vec![intent(1), intent(2)], + enrollments: vec![record.clone()], + _memory: Vec::new(), + }; + assert!(roster.covers_advertisements(&[signed(1)], 100).unwrap()); + assert!(!roster.covers_advertisements(&[], 100).unwrap()); + assert!( + !roster + .covers_advertisements(&[signed(1), signed(1)], 100) + .unwrap() + ); + assert!( + !roster + .covers_advertisements(&[signed(1), signed(2)], 100) + .unwrap() + ); + assert!(roster.covers_advertisements(&[signed(1)], 30_101).is_err()); + roster.enrollments[0] = boot(1); + assert!(!roster.covers_advertisements(&[signed(1)], 100).unwrap()); + roster.enrollments[0] = record.retire(Digest::from_bytes([23; 32]), 3).unwrap(); + assert!(!roster.covers_advertisements(&[signed(1)], 100).unwrap()); + roster.enrollments[0] = record.clone(); + roster.enrollments.push(record); + assert!(!roster.covers_advertisements(&[signed(1)], 100).unwrap()); +} + +#[test] +fn unresolved_role_requires_both_original_boots_even_without_a_boot_row() { + let mut roster = FleetRoster { + snapshot: FleetJournalSnapshot::new( + FleetHead::new(scope(), 0).unwrap(), + version().bootstrap(1).unwrap(), + ) + .unwrap(), + intents: vec![intent(1), intent(2)], + enrollments: vec![ + boot(1).establish(Digest::from_bytes([22; 32]), 2).unwrap(), + record(3), + ], + _memory: Vec::new(), + }; + assert!(!roster.covers_advertisements(&[signed(1)], 100).unwrap()); + roster + .enrollments + .push(boot(2).establish(Digest::from_bytes([23; 32]), 2).unwrap()); + assert!( + roster + .covers_advertisements(&[signed(1), signed(2)], 100) + .unwrap() + ); + roster.intents[0] = intent(1) + .rebind_active(SessionId::from_bytes([9; 16]), 2) + .unwrap(); + assert!( + !roster + .covers_advertisements(&[signed(1), signed(2)], 100) + .unwrap() + ); + assert_eq!(roster.required_boots()[0].session, endpoint(1).session); +} + +#[test] +fn selected_boot_requires_one_current_established_original_session() { + let established = boot(1).establish(Digest::from_bytes([22; 32]), 2).unwrap(); + let mut roster = FleetRoster { + snapshot: FleetJournalSnapshot::new(FleetHead::new(scope(), 0).unwrap(), version()) + .unwrap(), + intents: vec![intent(1)], + enrollments: vec![established.clone()], + _memory: Vec::new(), + }; + let node = endpoint(1).node; + let session = endpoint(1).session; + assert_eq!( + roster.boot(node, session).unwrap().enrollment(), + &established + ); + assert!(roster.boot(node, SessionId::from_bytes([9; 16])).is_err()); + roster.enrollments.push(established.clone()); + assert!(roster.boot(node, session).is_err()); + roster.enrollments = vec![boot(1)]; + assert!(roster.boot(node, session).is_err()); + roster.enrollments = vec![established.retire(Digest::from_bytes([23; 32]), 3).unwrap()]; + assert!(roster.boot(node, session).is_err()); + roster.enrollments = vec![established]; + roster.intents[0] = intent(1) + .rebind_active(SessionId::from_bytes([9; 16]), 2) + .unwrap(); + assert!(roster.boot(node, session).is_err()); +} + +#[tokio::test] +async fn native_roster_rows_hold_shared_credit_until_the_original_scan_drops() { + let runtime = CellRuntime::new( + cellule_runtime::cell::worker::SqlWorkerPool::new(1, 2).unwrap(), + 8 << 20, + SessionId::from_bytes([90; 16]), + ) + .unwrap(); + let baseline = runtime.stats().retained_bytes(); + let mut scan = Scan::new(version()); + scan.runtime = Some(&runtime); + scan.intents( + IntentPage::new(version(), None, vec![intent(1)], None).unwrap(), + None, + ) + .unwrap(); + let after_intent = runtime.stats().retained_bytes(); + assert!(after_intent > baseline); + scan.enrollments( + EnrollmentPage::new(version(), None, vec![record(3)], None).unwrap(), + None, + ) + .unwrap(); + assert!(runtime.stats().retained_bytes() > after_intent); + let page = reserve_page(Some(&runtime)).unwrap(); + assert!(runtime.stats().retained_bytes() >= baseline + 4 * MAX_PAGE_BYTES as usize); + drop(page); + drop(scan); + assert_eq!(runtime.stats().retained_bytes(), baseline); + runtime.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn native_roster_capacity_refusal_does_not_append_or_leak_retained_rows() { + let runtime = CellRuntime::new( + cellule_runtime::cell::worker::SqlWorkerPool::new(1, 2).unwrap(), + 1 << 20, + SessionId::from_bytes([91; 16]), + ) + .unwrap(); + assert!(matches!( + reserve_page(Some(&runtime)), + Err(Error::Capacity(_)) + )); + let baseline = runtime.stats().retained_bytes(); + let held = runtime + .try_reserve_node_metadata_bytes(runtime.stats().retained_capacity_bytes() - baseline) + .unwrap(); + let mut scan = Scan::new(version()); + scan.runtime = Some(&runtime); + assert!(matches!( + scan.intents( + IntentPage::new(version(), None, vec![intent(1)], None).unwrap(), + None + ), + Err(Error::Capacity(_)) + )); + assert!(scan.intents.is_empty() && scan.memory.is_empty()); + drop(held); + assert_eq!(runtime.stats().retained_bytes(), baseline); + scan.intents( + IntentPage::new(version(), None, vec![intent(1)], None).unwrap(), + None, + ) + .unwrap(); + drop(scan); + assert_eq!(runtime.stats().retained_bytes(), baseline); + runtime.shutdown().await.unwrap(); +} diff --git a/crates/cellule-host/src/fleet/snapshot/capture.rs b/crates/cellule-host/src/fleet/snapshot/capture.rs new file mode 100644 index 00000000..e49028a1 --- /dev/null +++ b/crates/cellule-host/src/fleet/snapshot/capture.rs @@ -0,0 +1,114 @@ +use super::super::actions::{FleetActionExecutor, journal_error, operation, wall_time_ms}; +use super::*; + +impl FleetActionExecutor { + pub(in crate::fleet) async fn execute_snapshot( + &self, + request: FleetSnapshotRequest, + owners: Arc, + ) -> Result, Arc> { + let admitted = wall_time_ms().map_err(Arc::new)?; + self.journal + .authorize_snapshot(&request, admitted) + .await + .map_err(journal_error) + .map_err(Arc::new)?; + let started = wall_time_ms().map_err(Arc::new)?; + if started >= request.deadline_ms() { + return Err(Arc::new(operation( + cellule_runtime::fleet::operations::OperationError::Deadline, + ))); + } + let state_before = owners.state().map_err(Arc::new)?; + let page = match request.subject() { + FleetSnapshotSubject::Host => FleetSnapshotNativePage::Host, + FleetSnapshotSubject::Cells(cursor) => FleetSnapshotNativePage::Cells( + self.runtime + .fleet_cells_page(*cursor, request.limit()) + .await + .map_err(Arc::new)?, + ), + FleetSnapshotSubject::Readers(cursor) => match &owners.readers { + Some(reader) => FleetSnapshotNativePage::Readers( + reader + .fleet_readers_page(*cursor, request.limit(), started) + .await + .map_err(Arc::new)?, + ), + None => FleetSnapshotNativePage::Unbound, + }, + FleetSnapshotSubject::ReaderEnrollments(cursor) => match &owners.readers { + Some(reader) => match reader + .fleet_reader_enrollments_page(*cursor, request.limit(), started) + .map_err(Arc::new)? + { + Some(page) => FleetSnapshotNativePage::ReaderEnrollments(page), + None => FleetSnapshotNativePage::Unbound, + }, + None => FleetSnapshotNativePage::Unbound, + }, + FleetSnapshotSubject::FollowerLanes(cursor) => match &owners.followers { + Some(store) => FleetSnapshotNativePage::FollowerLanes( + store + .fleet_lanes_page(*cursor, request.limit(), started) + .await + .map_err(Arc::new)?, + ), + None => FleetSnapshotNativePage::Unbound, + }, + FleetSnapshotSubject::FollowerEnrollments(cursor) => match &owners.producer { + Some(producer) => FleetSnapshotNativePage::FollowerEnrollments( + producer + .page(*cursor, request.limit(), started) + .map_err(Arc::new)?, + ), + None => FleetSnapshotNativePage::Unbound, + }, + FleetSnapshotSubject::DurabilitySupervisor => match &owners.supervisor { + Some(owner) => FleetSnapshotNativePage::DurabilitySupervisor( + owner.observe(started).map_err(Arc::new)?, + ), + None => FleetSnapshotNativePage::Unbound, + }, + }; + // Keep the native read owned even if its RPC waiter or deadline expires. + // Post-authorization refuses changed head/registry instead of restamping + // the original page under a newer barrier. It can start no native effect. + let node_log = self + .runtime + .try_node_durability() + .map_err(Arc::new)? + .map(|(application, durability)| { + if application != self.scope.application { + return Err(Error::Fenced); + } + durability.identity() + }) + .transpose() + .map_err(Arc::new)?; + let captured = wall_time_ms().map_err(Arc::new)?; + self.journal + .authorize_snapshot(&request, captured) + .await + .map_err(journal_error) + .map_err(Arc::new)?; + let state_after = owners.state().map_err(Arc::new)?; + let mode = self.runtime.node_admission().mode().map_err(Arc::new)?; + let finished = wall_time_ms().map_err(Arc::new)?; + let result = FleetNodeSnapshot { + request, + started_at_ms: started, + finished_at_ms: finished, + state_before, + state_after, + mode, + bindings: owners.bindings(), + node_log, + page, + }; + result + .validate(result.request(), finished) + .map_err(Arc::new)?; + Ok(Arc::new(result)) + } +} diff --git a/crates/cellule-host/src/fleet/snapshot/mod.rs b/crates/cellule-host/src/fleet/snapshot/mod.rs new file mode 100644 index 00000000..5f4f4348 --- /dev/null +++ b/crates/cellule-host/src/fleet/snapshot/mod.rs @@ -0,0 +1,226 @@ +//! Native page collection through the existing retained fleet job owner. +use super::FleetJournalSnapshot; +use crate::NodeState; +use crate::read_replicas::ReadReplicaManager; +use cellule_runtime::{ + Error, + identity::{Digest, NodeId, SessionId}, + node::NodeMode, +}; +use std::sync::{Arc, Mutex}; + +mod capture; +mod request; +pub use request::{FleetSnapshotRequest, FleetSnapshotSubject}; + +/// Authenticated fresh native-page transport through the node's existing finite +/// fleet executor. Preserve exact requests and original errors; cached effect +/// results or restamped responses cannot satisfy this read-only boundary. +pub trait FleetSnapshotTransport: Send + Sync + 'static { + /// Captures this exact boot, category, continuation, nonce and journal barrier. + /// Dropping the waiter leaves any accepted native capture owned by the node. + fn capture<'a>( + &'a self, + request: &'a FleetSnapshotRequest, + deadline: tokio::time::Instant, + ) -> super::FleetAdapterFuture<'a, Arc>; +} + +/// Exact local installed owner bindings. Unbound owners supply no role coverage. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct FleetSnapshotBindings { + /// A matching managed startup intent was confirmed before capture admission. + pub managed_startup: bool, + /// The canonical reader manager is installed. + pub readers: bool, + /// The canonical inbound follower store is installed. + pub follower_store: bool, + /// The retained original supervisor owner is installed. + pub durability_supervisor: bool, + /// Its managed follower producer is installed. + pub follower_producer: bool, +} + +/// One original native page, retaining its existing allocation token and errors. +pub enum FleetSnapshotNativePage { + /// Fixed host metadata lives in the enclosing snapshot. + Host, + /// Original generation-bound actor inventory. + Cells(cellule_runtime::cell::actor::CellInventoryPage), + /// Original managed reader inventory. + Readers(crate::read_replicas::ReaderInventoryPage), + /// Original reader producer progress, including accepted preparation jobs. + ReaderEnrollments(crate::read_replicas::ReaderEnrollmentInventoryPage), + /// Original persisted inbound lane inventory. + FollowerLanes(cellule_runtime::follower::FollowerInventoryPage), + /// Original managed leader producer inventory. + FollowerEnrollments(crate::FollowerEnrollmentInventoryPage), + /// Original supervisor join state and bounded request bank. + DurabilitySupervisor(crate::NodeDurabilitySupervisorObservation), + /// The requested owner/binding is missing. This is never an empty page or absence proof. + Unbound, +} + +/// Request-bound local interval evidence. Authentication, stable full traversal, +/// current remote authority and replacement-policy proof remain adapter duties. +/// Native pages retain their original bounded charges until this response drops. +pub struct FleetNodeSnapshot { + request: FleetSnapshotRequest, + started_at_ms: i64, + finished_at_ms: i64, + state_before: NodeState, + state_after: NodeState, + mode: NodeMode, + bindings: FleetSnapshotBindings, + node_log: Option<(SessionId, NodeId, u64)>, + page: FleetSnapshotNativePage, +} +impl FleetNodeSnapshot { + /// Returns the exact original request and journal barrier. + #[must_use] + pub fn request(&self) -> &FleetSnapshotRequest { + &self.request + } + /// Returns the actual capture start after pre-authorization. + #[must_use] + pub const fn started_at_ms(&self) -> i64 { + self.started_at_ms + } + /// Returns the actual completion after post-authorization. + #[must_use] + pub const fn finished_at_ms(&self) -> i64 { + self.finished_at_ms + } + /// Returns lifecycle before the original native read. + #[must_use] + pub const fn state_before(&self) -> NodeState { + self.state_before + } + /// Returns lifecycle after the original native read. + #[must_use] + pub const fn state_after(&self) -> NodeState { + self.state_after + } + /// Returns shared admission mode at completion, independently of counts. + #[must_use] + pub const fn mode(&self) -> NodeMode { + self.mode + } + /// Returns installed owner bindings; no missing owner supplies role coverage. + #[must_use] + pub const fn bindings(&self) -> FleetSnapshotBindings { + self.bindings + } + /// Returns the current local log binding, without remote membership/closure authority. + #[must_use] + pub const fn node_log(&self) -> Option<(SessionId, NodeId, u64)> { + self.node_log + } + /// Returns the original native page, never a cached movement result. + #[must_use] + pub fn page(&self) -> &FleetSnapshotNativePage { + &self.page + } + /// Validates complete request equality and the original non-restamped interval. + /// This supplies no transport authentication or finalization permission. + pub fn validate( + &self, + request: &FleetSnapshotRequest, + now_ms: i64, + ) -> cellule_runtime::Result<()> { + if &self.request != request + || self.started_at_ms < request.issued_at_ms() + || self.finished_at_ms < self.started_at_ms + || self.finished_at_ms >= request.deadline_ms() + || now_ms < self.finished_at_ms + || now_ms >= request.deadline_ms() + { + return Err(Error::Node("native snapshot request or interval differs")); + } + let scope = request.expected().head().scope(); + let (tag, observed, identity) = match &self.page { + FleetSnapshotNativePage::Host => (1, self.started_at_ms, true), + FleetSnapshotNativePage::Cells(page) => ( + 2, + page.observed_at_ms(), + page.session() == request.session() && page.entries().len() <= request.limit(), + ), + FleetSnapshotNativePage::Readers(page) => ( + 3, + page.observed_at_ms(), + page.session() == request.session() && page.entries().len() <= request.limit(), + ), + FleetSnapshotNativePage::ReaderEnrollments(page) => ( + 4, + page.observed_at_ms(), + page.session() == request.session() + && page.node() == request.node() + && page.scope() == scope + && page.entries().len() <= request.limit(), + ), + FleetSnapshotNativePage::FollowerLanes(page) => ( + 5, + page.observed_at_ms(), + page.entries().len() <= request.limit(), + ), + FleetSnapshotNativePage::FollowerEnrollments(page) => ( + 6, + page.observed_at_ms(), + page.session() == request.session() + && page.node() == request.node() + && page.scope() == scope + && page.entries().len() <= request.limit(), + ), + FleetSnapshotNativePage::DurabilitySupervisor(observed) => ( + 7, + observed.observed_at_ms, + observed.session == request.session() && observed.application == scope.application, + ), + FleetSnapshotNativePage::Unbound => (request.subject().tag(), self.started_at_ms, true), + }; + if tag != request.subject().tag() + || !identity + || observed < self.started_at_ms + || observed > self.finished_at_ms + { + return Err(Error::Node( + "native snapshot page identity or capture differs", + )); + } + if self.node_log.is_some_and(|(session, node, epoch)| { + session != request.session() || node != request.node() || epoch == 0 + }) { + return Err(Error::Fenced); + } + Ok(()) + } +} + +pub(crate) struct SnapshotOwners { + pub(crate) state: Arc>, + pub(crate) managed_startup: bool, + pub(crate) readers: Option>, + pub(crate) followers: Option>, + pub(crate) supervisor: Option>, + pub(crate) producer: Option>, +} +impl SnapshotOwners { + fn state(&self) -> cellule_runtime::Result { + self.state + .lock() + .map(|s| *s) + .map_err(|_| Error::Control("native snapshot lifecycle lock poisoned")) + } + fn bindings(&self) -> FleetSnapshotBindings { + FleetSnapshotBindings { + managed_startup: self.managed_startup, + readers: self.readers.is_some(), + follower_store: self.followers.is_some(), + durability_supervisor: self.supervisor.is_some(), + follower_producer: self.producer.is_some(), + } + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-host/src/fleet/snapshot/request.rs b/crates/cellule-host/src/fleet/snapshot/request.rs new file mode 100644 index 00000000..bbdf7d23 --- /dev/null +++ b/crates/cellule-host/src/fleet/snapshot/request.rs @@ -0,0 +1,207 @@ +use super::*; +use cellule_runtime::fleet::operations::{NodeIntent, OperationError}; + +/// One bounded native inventory through its existing continuation contract. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum FleetSnapshotSubject { + /// Fixed host binding/lifecycle metadata; this is not role-absence proof. + Host, + /// All live and transitioning actors, including unavailable owner metadata. + Cells(Option), + /// Installed managed views; accepted producer/native work is separate. + Readers(Option), + /// Original reader requests and retained protocol jobs. + ReaderEnrollments(Option), + /// Persisted local lanes, including cold and retired lanes. + FollowerLanes(Option), + /// Original selected members and managed leader protocol progress. + FollowerEnrollments(Option), + /// The original durability supervisor and its bounded rotation bank. + DurabilitySupervisor, +} + +impl FleetSnapshotSubject { + pub(super) fn tag(&self) -> u8 { + match self { + Self::Host => 1, + Self::Cells(_) => 2, + Self::Readers(_) => 3, + Self::ReaderEnrollments(_) => 4, + Self::FollowerLanes(_) => 5, + Self::FollowerEnrollments(_) => 6, + Self::DurabilitySupervisor => 7, + } + } + fn cursor_bytes(&self) -> Option> { + match self { + Self::Cells(cursor) => cursor.map(|c| c.to_bytes().to_vec()), + Self::Readers(cursor) => cursor.map(|c| c.to_bytes().to_vec()), + Self::ReaderEnrollments(cursor) => cursor.map(|c| c.to_bytes().to_vec()), + Self::FollowerLanes(cursor) => cursor.map(|c| c.to_bytes().to_vec()), + Self::FollowerEnrollments(cursor) => cursor.map(|c| c.to_bytes().to_vec()), + Self::Host | Self::DurabilitySupervisor => None, + } + } +} + +/// Exact read request binding a native page to the original journal barrier. +/// Applications authenticate the transport and retain this request unchanged +/// across retries. It is an in-process contract, not a persisted/wire codec. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct FleetSnapshotRequest { + expected: FleetJournalSnapshot, + nonce: Digest, + node: NodeId, + session: SessionId, + subject: FleetSnapshotSubject, + limit: usize, + issued_at_ms: i64, + deadline_ms: i64, +} + +impl FleetSnapshotRequest { + /// Creates a read-only page request; it can authorize no acquisition or close. + #[allow(clippy::too_many_arguments)] + pub fn new( + expected: FleetJournalSnapshot, + nonce: Digest, + node: NodeId, + session: SessionId, + subject: FleetSnapshotSubject, + limit: usize, + issued_at_ms: i64, + deadline_ms: i64, + ) -> Result { + let request = Self { + expected, + nonce, + node, + session, + subject, + limit, + issued_at_ms, + deadline_ms, + }; + request.validate()?; + Ok(request) + } + /// Returns the exact original head and enrollment registry version. + #[must_use] + pub fn expected(&self) -> &FleetJournalSnapshot { + &self.expected + } + /// Returns the caller's never-restamped capture nonce. + #[must_use] + pub const fn nonce(&self) -> Digest { + self.nonce + } + /// Returns the exact physical endpoint. + #[must_use] + pub const fn node(&self) -> NodeId { + self.node + } + /// Returns the exact original boot. + #[must_use] + pub const fn session(&self) -> SessionId { + self.session + } + /// Returns the original category and native continuation. + #[must_use] + pub fn subject(&self) -> &FleetSnapshotSubject { + &self.subject + } + /// Returns the original per-page bound. + #[must_use] + pub const fn limit(&self) -> usize { + self.limit + } + /// Returns the original request time. + #[must_use] + pub const fn issued_at_ms(&self) -> i64 { + self.issued_at_ms + } + /// Returns the exclusive capture deadline, bounded by the controller lease. + #[must_use] + pub const fn deadline_ms(&self) -> i64 { + self.deadline_ms + } + /// Identifies every execution input; a key never replaces full replay equality. + pub fn key(&self) -> Result { + self.validate()?; + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-native-snapshot-request.v1\0"); + for bytes in [ + self.expected.head().to_bytes()?, + self.expected.registry().to_bytes()?, + ] { + hash.update(&(bytes.len() as u64).to_be_bytes()); + hash.update(&bytes); + } + hash.update(self.nonce.as_bytes()); + hash.update(self.node.as_bytes()); + hash.update(self.session.as_bytes()); + hash.update(&[self.subject.tag()]); + let cursor = self.subject.cursor_bytes(); + hash.update(&[u8::from(cursor.is_some())]); + if let Some(cursor) = cursor { + hash.update(&cursor); + } + hash.update(&(self.limit as u64).to_be_bytes()); + hash.update(&self.issued_at_ms.to_be_bytes()); + hash.update(&self.deadline_ms.to_be_bytes()); + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) + } + /// Checks fresh full head/registry and the exact current endpoint intent. + /// Call inside the same journal transaction; a separate cached read is insufficient. + pub fn authorize_against( + &self, + current: &FleetJournalSnapshot, + intent: &NodeIntent, + now_ms: i64, + ) -> Result<(), OperationError> { + self.validate()?; + if now_ms < self.issued_at_ms || now_ms >= self.deadline_ms { + return Err(OperationError::Deadline); + } + if current != &self.expected + || intent.scope() != current.head().scope() + || intent.node() != self.node + || intent.session() != self.session + { + return Err(OperationError::Conflict); + } + Ok(()) + } + fn validate(&self) -> Result<(), OperationError> { + self.expected.head().to_bytes()?; + self.expected.registry().to_bytes()?; + let maximum = if matches!(self.subject, FleetSnapshotSubject::FollowerEnrollments(_)) { + 32 + } else { + 128 + }; + if [ + self.nonce.as_bytes().as_slice(), + self.node.as_bytes().as_slice(), + self.session.as_bytes().as_slice(), + ] + .iter() + .any(|id| id.iter().all(|b| *b == 0)) + || !(1..=maximum).contains(&self.limit) + || self.issued_at_ms < 0 + || self.deadline_ms <= self.issued_at_ms + || self.deadline_ms - self.issued_at_ms > 30_000 + { + return Err(OperationError::Invalid("invalid native snapshot request")); + } + let lease = self + .expected + .head() + .controller() + .ok_or(OperationError::Fenced)?; + if self.deadline_ms > lease.expires_at_ms { + return Err(OperationError::Deadline); + } + Ok(()) + } +} diff --git a/crates/cellule-host/src/fleet/snapshot/tests.rs b/crates/cellule-host/src/fleet/snapshot/tests.rs new file mode 100644 index 00000000..6409d1d6 --- /dev/null +++ b/crates/cellule-host/src/fleet/snapshot/tests.rs @@ -0,0 +1,269 @@ +use super::*; +use cellule_runtime::fleet::operations::{ + FleetHead, FleetProfile, FleetScope, NodeIntent, OperationError, RegistryVersion, +}; + +fn barrier() -> (FleetJournalSnapshot, NodeIntent) { + let scope = FleetScope { + fleet: Digest::from_bytes([1; 32]), + application: cellule_runtime::identity::ApplicationId::from_bytes([2; 16]), + }; + let head = FleetHead::new(scope, 100) + .unwrap() + .claim( + FleetProfile::default(), + 0, + SessionId::from_bytes([3; 16]), + 100, + ) + .unwrap(); + ( + FleetJournalSnapshot::new(head, RegistryVersion::new(scope).unwrap()).unwrap(), + NodeIntent::initial( + scope, + NodeId::from_bytes([4; 16]), + SessionId::from_bytes([5; 16]), + ) + .unwrap(), + ) +} +fn request(subject: FleetSnapshotSubject) -> FleetSnapshotRequest { + let (snapshot, intent) = barrier(); + FleetSnapshotRequest::new( + snapshot, + Digest::from_bytes([6; 32]), + intent.node(), + intent.session(), + subject, + 2, + 100, + 1_000, + ) + .unwrap() +} + +#[test] +fn every_request_input_binds_the_original_key_and_authorization() { + let original = request(FleetSnapshotSubject::Cells(None)); + let key = original.key().unwrap(); + let (snapshot, intent) = barrier(); + original.authorize_against(&snapshot, &intent, 100).unwrap(); + let subjects = [ + FleetSnapshotSubject::Host, + FleetSnapshotSubject::Readers(None), + FleetSnapshotSubject::ReaderEnrollments(None), + FleetSnapshotSubject::FollowerLanes(None), + FleetSnapshotSubject::FollowerEnrollments(None), + FleetSnapshotSubject::DurabilitySupervisor, + ]; + for subject in subjects { + assert_ne!(key, request(subject).key().unwrap()); + } + for (nonce, node, session, limit, issued, deadline) in [ + ( + Digest::from_bytes([7; 32]), + original.node(), + original.session(), + 2, + 100, + 1_000, + ), + ( + original.nonce(), + NodeId::from_bytes([7; 16]), + original.session(), + 2, + 100, + 1_000, + ), + ( + original.nonce(), + original.node(), + SessionId::from_bytes([7; 16]), + 2, + 100, + 1_000, + ), + ( + original.nonce(), + original.node(), + original.session(), + 1, + 100, + 1_000, + ), + ( + original.nonce(), + original.node(), + original.session(), + 2, + 101, + 1_000, + ), + ( + original.nonce(), + original.node(), + original.session(), + 2, + 100, + 1_001, + ), + ] { + let changed = FleetSnapshotRequest::new( + snapshot.clone(), + nonce, + node, + session, + original.subject().clone(), + limit, + issued, + deadline, + ) + .unwrap(); + assert_ne!(key, changed.key().unwrap()); + } + let changed = FleetJournalSnapshot::new( + snapshot.head().clone(), + snapshot + .registry() + .advance(snapshot.registry().revision()) + .unwrap(), + ) + .unwrap(); + assert!(matches!( + original.authorize_against(&changed, &intent, 100), + Err(OperationError::Conflict) + )); + let changed_request = FleetSnapshotRequest::new( + changed, + original.nonce(), + original.node(), + original.session(), + original.subject().clone(), + 2, + 100, + 1_000, + ) + .unwrap(); + assert_ne!(key, changed_request.key().unwrap()); + let replaced = NodeIntent::initial( + intent.scope(), + intent.node(), + SessionId::from_bytes([8; 16]), + ) + .unwrap(); + assert!(matches!( + original.authorize_against(&snapshot, &replaced, 100), + Err(OperationError::Conflict) + )); + assert!(matches!( + original.authorize_against(&snapshot, &intent, 99), + Err(OperationError::Deadline) + )); + assert!(matches!( + original.authorize_against(&snapshot, &intent, 1_000), + Err(OperationError::Deadline) + )); +} + +#[test] +fn native_continuations_bounds_and_controller_deadline_are_checked() { + let original = request(FleetSnapshotSubject::Cells(None)); + let cursor = cellule_runtime::cell::actor::CellInventoryCursor::from_bytes(&[9; 64]).unwrap(); + let continued = request(FleetSnapshotSubject::Cells(Some(cursor))); + assert_ne!(original.key().unwrap(), continued.key().unwrap()); + for (subject, limit, nonce, issued, deadline) in [ + ( + FleetSnapshotSubject::Cells(None), + 0, + original.nonce(), + 100, + 1_000, + ), + ( + FleetSnapshotSubject::Cells(None), + 129, + original.nonce(), + 100, + 1_000, + ), + ( + FleetSnapshotSubject::FollowerEnrollments(None), + 33, + original.nonce(), + 100, + 1_000, + ), + ( + FleetSnapshotSubject::Host, + 1, + Digest::from_bytes([0; 32]), + 100, + 1_000, + ), + (FleetSnapshotSubject::Host, 1, original.nonce(), -1, 1_000), + (FleetSnapshotSubject::Host, 1, original.nonce(), 100, 100), + (FleetSnapshotSubject::Host, 1, original.nonce(), 100, 30_101), + ] { + assert!( + FleetSnapshotRequest::new( + original.expected().clone(), + nonce, + original.node(), + original.session(), + subject, + limit, + issued, + deadline + ) + .is_err() + ); + } + let lease = original.expected().head().controller().unwrap(); + assert!( + FleetSnapshotRequest::new( + original.expected().clone(), + original.nonce(), + original.node(), + original.session(), + FleetSnapshotSubject::Host, + 1, + lease.expires_at_ms - 1, + lease.expires_at_ms + 1 + ) + .is_err() + ); +} + +#[test] +fn response_replay_cannot_restamp_its_original_interval_or_claim_unbound_absence() { + let original = request(FleetSnapshotSubject::Readers(None)); + let response = FleetNodeSnapshot { + request: original.clone(), + started_at_ms: 101, + finished_at_ms: 102, + state_before: NodeState::Ready, + state_after: NodeState::Draining, + mode: NodeMode::Draining, + bindings: FleetSnapshotBindings { + managed_startup: false, + readers: false, + follower_store: false, + durability_supervisor: false, + follower_producer: false, + }, + node_log: None, + page: FleetSnapshotNativePage::Unbound, + }; + response.validate(&original, 102).unwrap(); + assert!(matches!(response.page(), FleetSnapshotNativePage::Unbound)); + assert_eq!(response.state_before(), NodeState::Ready); + assert_eq!(response.state_after(), NodeState::Draining); + assert!(response.validate(&original, 101).is_err()); + assert!(response.validate(&original, 1_000).is_err()); + assert!( + response + .validate(&request(FleetSnapshotSubject::Host), 102) + .is_err() + ); +} diff --git a/crates/cellule-host/src/fleet/withdrawal.rs b/crates/cellule-host/src/fleet/withdrawal.rs new file mode 100644 index 00000000..06fc1fb6 --- /dev/null +++ b/crates/cellule-host/src/fleet/withdrawal.rs @@ -0,0 +1,120 @@ +//! Exact boot withdrawal after the canonical host resource sequence joins. + +use std::sync::Arc; + +use cellule_runtime::fleet::operations::{ + EnrollmentEvent, EnrollmentRecord, EnrollmentRole, EnrollmentStatus, +}; +use cellule_runtime::identity::Digest; +use cellule_runtime::node::{NodeDirectory, VersionedNodeAdvertisement}; +use cellule_runtime::{Error, Result}; + +use super::{FleetEnrollmentJournal, actions::wall_time_ms, operation}; + +/// One immutable boot binding retained by the original host drain owner. +pub(crate) struct FleetBootWithdrawal { + directory: NodeDirectory, + observed: VersionedNodeAdvertisement, + original: EnrollmentRecord, + journal: Arc, + evidence: Digest, +} + +impl FleetBootWithdrawal { + pub(crate) fn new( + directory: NodeDirectory, + observed: VersionedNodeAdvertisement, + original: EnrollmentRecord, + journal: Arc, + ) -> Result { + original.to_bytes().map_err(operation)?; + let spec = original.spec(); + let ad = observed.advertisement(); + // VersionedNodeAdvertisement is constructed only by the canonical + // directory's authenticated read/CAS paths; it cannot be fabricated by + // a remote action. Bind that original observation to the durable boot. + if !matches!(spec.role, EnrollmentRole::Node { .. }) + || spec.source.is_some() + || original.status() != EnrollmentStatus::Established + || spec.target.node != ad.node() + || spec.target.session != ad.session() + || spec.scope.fleet != ad.fleet() + { + return Err(Error::Fenced); + } + let established = original + .established_evidence() + .ok_or(Error::Control("fleet boot lacks establishment evidence"))?; + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-boot-withdrawal.v1\0"); + hash.update(&spec.to_bytes().map_err(operation)?); + hash.update(established.as_bytes()); + hash.update(ad.certificate().as_bytes()); + hash.update(ad.image().as_bytes()); + hash.update(ad.release().as_bytes()); + hash.update(&ad.verifying_key()?.to_bytes()); + hash.update(&(ad.endpoint().len() as u64).to_be_bytes()); + hash.update(ad.endpoint().as_bytes()); + Ok(Self { + directory, + observed, + original, + journal, + evidence: Digest::from_bytes(*hash.finalize().as_bytes()), + }) + } + + pub(crate) async fn withdraw(&self) -> Result<()> { + let spec = self.original.spec(); + let current = self + .journal + .load_enrollment(spec.scope, spec.key().map_err(operation)?) + .await + .map_err(journal_error)? + .ok_or(Error::Control("fleet boot withdrawal record is absent"))?; + current.validate_replay(spec).map_err(operation)?; + if current.accepted_at_ms() != self.original.accepted_at_ms() + || current.established_evidence() != self.original.established_evidence() + || !matches!( + current.status(), + EnrollmentStatus::Established | EnrollmentStatus::Retired + ) + { + return Err(Error::Fenced); + } + // Always cross canonical conditional withdrawal, including replay. A + // missing record or a recovery claimant cannot stand in for this boot's + // checked shutdown. The original token reconciles a late heartbeat CAS. + self.directory + .withdraw_after_drain(&self.observed, wall_time_ms()?) + .await?; + if !self.directory.is_withdrawn(spec.target.session).await? { + return Err(Error::Control("fleet boot withdrawal lacks its tombstone")); + } + let retired = self + .journal + .publish_enrollment_result( + &self.original, + EnrollmentEvent::Retired(self.evidence), + wall_time_ms()?, + ) + .await + .map_err(journal_error)?; + retired.validate_replay(spec).map_err(operation)?; + if retired.accepted_at_ms() != self.original.accepted_at_ms() + || retired.established_evidence() != self.original.established_evidence() + || retired.status() != EnrollmentStatus::Retired + || retired.settlement_evidence() != Some(self.evidence) + { + return Err(Error::Control("fleet boot retirement result differs")); + } + Ok(()) + } +} + +fn journal_error(source: Box) -> Error { + Error::Facility { + name: "fleet-boot-journal", + source, + } +} diff --git a/crates/cellule-host/src/lib.rs b/crates/cellule-host/src/lib.rs index abb6b343..8e0abdb0 100644 --- a/crates/cellule-host/src/lib.rs +++ b/crates/cellule-host/src/lib.rs @@ -20,11 +20,18 @@ pub use builder::CellNodeBuilder; pub use durability::{ - FacilityResult, NodeDurabilityProvider, NodeDurabilityRotation, NodeDurabilitySupervisorConfig, + FacilityResult, FleetNodeDurabilityProvider, FleetNodeLogRecruitment, + FollowerEnrollmentCompletion, FollowerEnrollmentInventoryCursor, + FollowerEnrollmentInventoryPage, FollowerEnrollmentMember, FollowerEnrollmentProgress, + FollowerEvacuation, NodeDurabilityProvider, NodeDurabilityRotation, + NodeDurabilitySupervisorConfig, NodeDurabilitySupervisorObservation, + NodeDurabilitySupervisorState, NodeLogRotationCompletion, NodeLogRotationEntry, + NodeLogRotationInventory, NodeLogRotationObservation, NodeLogRotationPhase, + NodeLogRotationRequest, }; pub use facility::CellNodeFacility; pub use node::CellNode; -pub use status::{NodeState, NodeStatus, ScaleDownStatus}; +pub use status::{NodeDrainObservation, NodeDrainPhase, NodeState, NodeStatus, ScaleDownStatus}; use std::{ any::Any, collections::HashSet, @@ -41,6 +48,7 @@ pub use tasks::CellNodeTaskGroup; mod builder; mod durability; mod facility; +pub mod fleet; mod node; pub mod read_replicas; mod status; @@ -69,3 +77,4 @@ const MAX_NODE_TASKS: usize = 256; pub const FOLLOWER_STORE_COMPONENT: &str = "follower-store"; /// Stable host-owned component name for the node-log enrollment provider. pub const NODE_DURABILITY_PROVIDER_COMPONENT: &str = "node-durability-provider"; +pub(crate) const NODE_DURABILITY_SUPERVISOR_COMPONENT: &str = "node-durability-supervisor"; diff --git a/crates/cellule-host/src/node/components.rs b/crates/cellule-host/src/node/components.rs index c5d3ec18..53a83996 100644 --- a/crates/cellule-host/src/node/components.rs +++ b/crates/cellule-host/src/node/components.rs @@ -103,6 +103,18 @@ impl CellNode { where P: NodeDurabilityProvider, { + if self + .fleet_startup + .lock() + .map_err(|_| Error::Control("CellNode fleet startup lock poisoned"))? + .is_some() + && std::any::TypeId::of::

() + != std::any::TypeId::of::() + { + return Err(Error::Control( + "configured fleet durability requires managed follower enrollment", + )); + } let task_group = self .task_group .lock() @@ -111,18 +123,97 @@ impl CellNode { .ok_or(Error::Control( "CellNode durability provider requires an installed task group", ))?; - self.install_owned_component(NODE_DURABILITY_PROVIDER_COMPONENT, Arc::clone(&provider))?; - let runtime = self.runtime.clone(); - let cancellation = task_group.cancellation.clone(); - let result = task_group.spawn_boxed(async move { - run_node_durability_supervisor(provider, runtime, configuration, cancellation).await - }); + task_group.ensure_accepting_tasks()?; + let supervisor = Arc::new(DurabilitySupervisor::new( + provider.clone(), + self.runtime.clone(), + configuration, + self.session, + task_group.cancellation.clone(), + )?); + let drained = Arc::clone(&supervisor); + let drained_provider = Arc::clone(&provider); + self.install_facilities([ + CellNodeFacility::owned(NODE_DURABILITY_PROVIDER_COMPONENT, provider, move || { + Arc::clone(&drained_provider).drain() + })?, + CellNodeFacility::owned( + NODE_DURABILITY_SUPERVISOR_COMPONENT, + Arc::clone(&supervisor), + move || { + let drained = Arc::clone(&drained); + async move { drained.drain().await } + }, + )?, + ])?; + // The task group owns supervision and health; the facility retains the + // actual join. Retain this watcher across deadlines too, so its forced + // cancellation cannot replace the facility's original result. + let result = task_group.spawn_retained(async move { supervisor.join().await }); if result.is_err() { + self.remove_facility(NODE_DURABILITY_SUPERVISOR_COMPONENT)?; self.remove_facility(NODE_DURABILITY_PROVIDER_COMPONENT)?; } result } + /// Requests confirmed retirement of one exact epoch through the existing + /// supervisor, bypassing normal frame thresholds. The embedding application + /// authorizes this call and journals acceptance/results before finalization. + /// Duplicate epoch requests share retained progress; dropping a handle does + /// not cancel work. An automatic rotation already in flight is refused. + pub fn request_node_log_rotation( + &self, + log_epoch: u64, + ) -> cellule_runtime::Result { + let supervisor = self + .owned_component::(NODE_DURABILITY_SUPERVISOR_COMPONENT) + .ok_or(Error::Control( + "CellNode durability supervisor is not installed", + ))?; + if !matches!( + self.state(), + NodeState::Ready | NodeState::ScalingDown | NodeState::Maintenance + ) { + return Err(Error::CellDraining); + } + supervisor + .requests + .request(&self.runtime, log_epoch, &supervisor.cancellation) + } + + /// Looks up one retained epoch request, including interrupted work during + /// drain. Missing local progress never proves role absence or completion. + pub fn node_log_rotation_request( + &self, + log_epoch: u64, + ) -> cellule_runtime::Result> { + match self.owned_component::(NODE_DURABILITY_SUPERVISOR_COMPONENT) { + Some(supervisor) => supervisor.requests.lookup(log_epoch), + None => Ok(None), + } + } + + /// Captures the retained supervisor without awaiting provider/native work. + /// Returned or cancelled work is not joined; joined failure is not role + /// absence. None supplies no coverage. Authentication, current authority, + /// producer/native inventories and durable revision checks remain required. + /// The installed owner charges fixed metadata; capture remains available + /// during drain when new runtime byte admission has already closed. + pub fn fleet_durability_supervisor( + &self, + now_ms: i64, + ) -> cellule_runtime::Result> { + if now_ms < 0 { + return Err(Error::Node( + "invalid durability supervisor observation time", + )); + } + self.try_owned_component::(NODE_DURABILITY_SUPERVISOR_COMPONENT)? + .map(|supervisor| supervisor.observe(now_ms)) + .transpose() + } + /// Owns read-snapshot refresh, eviction, and terminal close for this node. /// /// Install during startup after the task group. The product supplies an @@ -153,12 +244,23 @@ impl CellNode { root, limits, ); + if self + .fleet_startup + .lock() + .map_err(|_| Error::Control("CellNode fleet startup lock poisoned"))? + .is_some() + { + self.require_owned_components(["fleet-reader-enrollment"])?; + manager.require_enrollment(); + } let drained = manager.clone(); self.install_owned_component_with_drain(COMPONENT, Arc::new(manager.clone()), move || { let drained = drained.clone(); async move { - drained.shutdown().await; - Ok(()) + drained + .shutdown() + .await + .map_err(|error| Box::new(error) as Box) } })?; let supervised = manager.clone(); @@ -307,7 +409,8 @@ impl CellNode { let Some((root, limits, disk)) = configuration else { return Ok(()); }; - let store = FollowerStore::open(root, limits, disk)?; + let store = FollowerStore::open(root, limits, disk)? + .with_node_admission(self.runtime.node_admission()); self.install_owned_component(FOLLOWER_STORE_COMPONENT, Arc::new(store)) } @@ -317,11 +420,25 @@ impl CellNode { where T: Send + Sync + 'static, { - self.facilities + self.try_owned_component(name).ok().flatten() + } + + /// Looks up an optional typed component, preserving lock and type failures. + /// Fleet observations must distinguish missing facilities from failed capture. + pub fn try_owned_component(&self, name: &str) -> cellule_runtime::Result>> + where + T: Send + Sync + 'static, + { + let facilities = self + .facilities .lock() - .ok()? - .iter() - .find(|facility| facility.name == name) - .and_then(CellNodeFacility::owner) + .map_err(|_| Error::Control("CellNode facility lock poisoned"))?; + let Some(facility) = facilities.iter().find(|facility| facility.name == name) else { + return Ok(None); + }; + facility + .owner() + .map(Some) + .ok_or(Error::Control("CellNode component type differs")) } } diff --git a/crates/cellule-host/src/node/drain/mod.rs b/crates/cellule-host/src/node/drain/mod.rs new file mode 100644 index 00000000..e30a7d70 --- /dev/null +++ b/crates/cellule-host/src/node/drain/mod.rs @@ -0,0 +1,211 @@ +//! One canonical resource sequence, retained by the original host closing task. +use super::*; +mod owner; +pub(crate) use owner::DrainOwner; + +pub(super) struct DrainResources { + runtime: CellRuntime, + state: Arc>, + facilities: Arc>>, + task_group: Arc>>>, + runtime_drain: tokio::sync::Mutex, + boot_withdrawal: Mutex>>, +} + +#[derive(Default)] +enum RuntimeDrain { + #[default] + Idle, + Running(JoinHandle>), + Finished(Result<(), Arc>), +} + +#[derive(Debug)] +struct RetainedDrainFailure(Arc); + +impl std::fmt::Display for RetainedDrainFailure { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + self.0.fmt(formatter) + } +} + +impl std::error::Error for RetainedDrainFailure { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + Some(self.0.as_ref()) + } +} + +impl DrainResources { + pub(super) async fn run(&self, deadline: Option) -> cellule_runtime::Result<()> { + { + let mut state = self + .state + .lock() + .map_err(|_| Error::Control("CellNode lifecycle lock poisoned"))?; + if *state == NodeState::Stopped { + return Ok(()); + } + *state = NodeState::Draining; + } + let task_group = self + .task_group + .lock() + .map(|task_group| task_group.clone()) + .map_err(|_| Error::Control("CellNode task group lock poisoned")); + let facilities = self + .facilities + .lock() + .map(|facilities| { + facilities + .iter() + .rev() + .map(|facility| (facility.name, Arc::clone(&facility.drain))) + .collect::>() + }) + .map_err(|_| Error::Control("CellNode facility lock poisoned")); + let mut first_error = None; + if let Some(error) = task_group.as_ref().err().map(|error| match error { + Error::Control(message) => Error::Control(message), + _ => Error::Control("CellNode task group unavailable during drain"), + }) { + first_error = Some(error); + } + if let Ok(Some(task_group)) = task_group.as_ref() { + task_group.cancel_work(); + } + match facilities { + Err(error) if first_error.is_none() => first_error = Some(error), + Err(_) => {} + Ok(facilities) => { + for (name, drain) in facilities { + let result = if name == "cell-coordination-tasks" { + // The task group is a retained owner, so its join must share the + // node deadline; an unbounded callback could strand shutdown. + match task_group.as_ref() { + Ok(Some(task_group)) => task_group.drain_work_until(deadline).await, + _ => drain().await, + } + } else { + match deadline { + Some(deadline) => { + match tokio::time::timeout_at(deadline.into(), drain()).await { + Ok(result) => result, + Err(_) => Err(Box::new(std::io::Error::new( + std::io::ErrorKind::TimedOut, + "CellNode facility drain deadline exceeded", + )) + as Box), + } + } + None => drain().await, + } + }; + if let Err(source) = result + && first_error.is_none() + { + first_error = Some(Error::Facility { name, source }); + } + } + } + } + let runtime_result = match deadline { + Some(deadline) => { + match tokio::time::timeout_at(deadline.into(), self.join_runtime_drain()).await { + Ok(result) => result, + Err(_) => Err(Error::Control("CellNode runtime drain deadline exceeded")), + } + } + None => self.join_runtime_drain().await, + }; + let runtime_closed = runtime_result.is_ok(); + if first_error.is_none() { + first_error = runtime_result.err(); + } + // Session withdrawal fences log authority. An owned enrollment may + // still return a committed generation after runtime closure, so retain + // lease maintenance until every required facility join also succeeds. + if runtime_closed + && first_error.is_none() + && let Ok(Some(task_group)) = task_group + && let Err(source) = task_group.drain_until(deadline).await + && first_error.is_none() + { + first_error = Some(Error::Facility { + name: "cell-coordination-tasks", + source, + }); + } + let result = match first_error { + Some(error) => Err(error), + None => self.withdraw_boot(deadline).await, + }; + let result = if result.is_ok() { + match self.facilities.lock() { + Ok(mut facilities) => { + facilities.clear(); + Ok(()) + } + Err(_) => Err(Error::Control("CellNode facility lock poisoned")), + } + } else { + result + }; + if result.is_ok() + && let Ok(mut state) = self.state.lock() + { + *state = NodeState::Stopped; + } + result + } + + async fn withdraw_boot(&self, deadline: Option) -> cellule_runtime::Result<()> { + let withdrawal = self + .boot_withdrawal + .lock() + .map_err(|_| Error::Control("CellNode boot withdrawal lock poisoned"))? + .clone(); + let Some(withdrawal) = withdrawal else { + return Ok(()); + }; + match deadline { + Some(deadline) => tokio::time::timeout_at(deadline.into(), withdrawal.withdraw()) + .await + .map_err(|source| Error::Facility { + name: "fleet-boot-withdrawal-deadline", + source: Box::new(source), + })?, + None => withdrawal.withdraw().await, + } + } + + async fn join_runtime_drain(&self) -> cellule_runtime::Result<()> { + let mut drain = self.runtime_drain.lock().await; + if matches!(*drain, RuntimeDrain::Idle) { + let runtime = self.runtime.clone(); + // The canonical runtime barrier is invoked once. A caller deadline + // drops only this join waiter; the host retains the task and result. + *drain = RuntimeDrain::Running(tokio::spawn(async move { runtime.shutdown().await })); + } + let result = match &mut *drain { + RuntimeDrain::Running(task) => { + let result = match task.await { + Ok(result) => result.map_err(Arc::new), + Err(source) => Err(Arc::new(Error::Facility { + name: "cell-runtime-task", + source: Box::new(source), + })), + }; + *drain = RuntimeDrain::Finished(result.clone()); + result + } + RuntimeDrain::Finished(result) => result.clone(), + RuntimeDrain::Idle => { + return Err(Error::Control("CellNode runtime drain did not start")); + } + }; + result.map_err(|source| Error::Facility { + name: "cell-runtime-drain", + source: Box::new(RetainedDrainFailure(source)), + }) + } +} diff --git a/crates/cellule-host/src/node/drain/owner.rs b/crates/cellule-host/src/node/drain/owner.rs new file mode 100644 index 00000000..af3fa385 --- /dev/null +++ b/crates/cellule-host/src/node/drain/owner.rs @@ -0,0 +1,264 @@ +//! Retain the complete host attempt without moving the shared drain lane. +use super::*; +use tokio::sync::{OwnedMutexGuard, watch}; + +type DrainResult = std::result::Result<(), Arc>; + +pub(crate) struct DrainOwner { + resources: Arc, + bank: Mutex, + history: Arc>, +} + +#[derive(Default)] +struct DrainBank { + serial: u64, + current: Option>, +} + +#[derive(Default)] +struct DrainHistory { + first: Option>, + latest: Option>, +} + +struct DrainAttempt { + serial: u64, + task: tokio::sync::Mutex, + returned: watch::Sender>, + joined: AtomicBool, +} + +enum DrainJoin { + Running(JoinHandle), + Finished { + result: DrainResult, + task_failure: Option>, + }, +} + +impl DrainHistory { + fn record(&mut self, source: Arc) { + self.first.get_or_insert_with(|| Arc::clone(&source)); + self.latest = Some(source); + } +} + +impl DrainOwner { + pub(crate) fn new( + runtime: CellRuntime, + state: Arc>, + facilities: Arc>>, + task_group: Arc>>>, + ) -> Self { + Self { + resources: Arc::new(DrainResources { + runtime, + state, + facilities, + task_group, + runtime_drain: tokio::sync::Mutex::new(RuntimeDrain::Idle), + boot_withdrawal: Mutex::new(None), + }), + bank: Mutex::new(DrainBank::default()), + history: Arc::new(Mutex::new(DrainHistory::default())), + } + } + + pub(crate) fn bind_boot_withdrawal( + &self, + withdrawal: crate::fleet::withdrawal::FleetBootWithdrawal, + ) -> cellule_runtime::Result<()> { + let mut binding = self + .resources + .boot_withdrawal + .lock() + .map_err(|_| Error::Control("CellNode boot withdrawal lock poisoned"))?; + if binding.is_some() { + return Err(Error::Control("CellNode boot withdrawal already installed")); + } + *binding = Some(Arc::new(withdrawal)); + Ok(()) + } + + pub(crate) async fn drain( + &self, + shutdown: OwnedMutexGuard<()>, + deadline: Option, + ) -> cellule_runtime::Result<()> { + let previous = self + .bank + .lock() + .map_err(|_| Error::Control("CellNode drain bank poisoned"))? + .current + .clone(); + if let Some(previous) = previous { + // Acquiring the lane proves the old resource sequence released its + // guard. Join its actual epilogue before replacing this fixed slot. + // A task panic remains fatal; it is not a retryable phase timeout. + previous.join(&self.history).await.map_err(shared_error)?; + } + { + let mut state = self + .resources + .state + .lock() + .map_err(|_| Error::Control("CellNode lifecycle lock poisoned"))?; + if *state == NodeState::Stopped { + return Ok(()); + } + *state = NodeState::Draining; + } + let attempt = { + let mut bank = self + .bank + .lock() + .map_err(|_| Error::Control("CellNode drain bank poisoned"))?; + let serial = bank + .serial + .checked_add(1) + .ok_or(Error::Capacity("CellNode drain serial exhausted"))?; + let resources = Arc::clone(&self.resources); + let history = Arc::clone(&self.history); + let (returned, _) = watch::channel(None); + let result = returned.clone(); + // Capture resources and diagnostic history, never this owner or its + // slot. The retained task cannot form an owner/join-handle cycle. + let task = tokio::spawn(async move { + let _shutdown = shutdown; + let outcome = resources.run(deadline).await.map_err(Arc::new); + if let Err(source) = &outcome { + record_failure(&history, Arc::clone(source)); + } + result.send_replace(Some(outcome.clone())); + outcome + }); + let attempt = Arc::new(DrainAttempt { + serial, + task: tokio::sync::Mutex::new(DrainJoin::Running(task)), + returned, + joined: AtomicBool::new(false), + }); + bank.serial = serial; + bank.current = Some(Arc::clone(&attempt)); + attempt + }; + attempt.join(&self.history).await.map_err(shared_error)?; + attempt.result().await?.map_err(shared_error) + } + + pub(crate) fn observe(&self) -> cellule_runtime::Result> { + let current = self + .bank + .lock() + .map_err(|_| Error::Control("CellNode drain bank poisoned"))? + .current + .clone(); + let Some(current) = current else { + return Ok(None); + }; + let joined = current.joined.load(Ordering::Acquire); + let result = current.returned.borrow().clone(); + let phase = if joined { + NodeDrainPhase::Joined + } else if result.is_some() { + NodeDrainPhase::Returned + } else { + NodeDrainPhase::Running + }; + let history = self + .history + .lock() + .map_err(|_| Error::Control("CellNode drain history poisoned"))?; + Ok(Some(NodeDrainObservation { + serial: current.serial, + phase, + result, + first_failure: history.first.clone(), + latest_failure: history.latest.clone(), + })) + } +} + +impl DrainAttempt { + async fn join(&self, history: &Mutex) -> DrainResult { + let mut joining = self.task.lock().await; + if let DrainJoin::Running(task) = &mut *joining { + let (result, task_failure) = match task.await { + Ok(result) => (result, None), + Err(source) => { + let source = Arc::new(Error::Facility { + name: "cell-node-drain-task", + source: Box::new(source), + }); + record_failure(history, Arc::clone(&source)); + self.returned.send_replace(Some(Err(Arc::clone(&source)))); + (Err(Arc::clone(&source)), Some(source)) + } + }; + // No await after consuming the handle. A cancelled waiter leaves + // the original handle in Running; later callers join it in place. + *joining = DrainJoin::Finished { + result, + task_failure, + }; + self.joined.store(true, Ordering::Release); + } + match &*joining { + DrainJoin::Finished { task_failure, .. } => match task_failure { + Some(source) => Err(Arc::clone(source)), + None => Ok(()), + }, + DrainJoin::Running(_) => { + Err(Arc::new(Error::Control("CellNode drain join incomplete"))) + } + } + } + + async fn result(&self) -> cellule_runtime::Result { + match &*self.task.lock().await { + DrainJoin::Finished { result, .. } => Ok(result.clone()), + DrainJoin::Running(_) => Err(Error::Control("CellNode drain result is unjoined")), + } + } +} + +fn record_failure(history: &Mutex, source: Arc) { + // History contains diagnostics only. Poison recovery preserves the known + // original failure and does not confer authority or declare shutdown safe. + let mut history = match history.lock() { + Ok(history) => history, + Err(poisoned) => poisoned.into_inner(), + }; + history.record(source); +} + +fn shared_error(source: Arc) -> Error { + let name = match source.as_ref() { + Error::Facility { name, .. } => *name, + Error::Control(message) => return Error::Control(message), + _ => "cell-node-drain", + }; + Error::Facility { + name, + source: Box::new(RetainedHostFailure(source)), + } +} + +#[derive(Debug)] +struct RetainedHostFailure(Arc); +impl std::fmt::Display for RetainedHostFailure { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + // The outer error already retains the original facility name. Keep + // its presentation while preserving the complete original source. + match self.0.as_ref() { + Error::Facility { source, .. } => std::fmt::Display::fmt(source, formatter), + source => std::fmt::Display::fmt(source, formatter), + } + } +} +impl std::error::Error for RetainedHostFailure { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + Some(self.0.as_ref()) + } +} diff --git a/crates/cellule-host/src/node/fleet.rs b/crates/cellule-host/src/node/fleet.rs new file mode 100644 index 00000000..a821622c --- /dev/null +++ b/crates/cellule-host/src/node/fleet.rs @@ -0,0 +1,452 @@ +use super::*; +use crate::fleet::{ + FLEET_ACTION_COMPONENT, FleetActionCompletion, FleetActionExecutor, FleetActionJournal, + FleetCellProvider, FleetEnrollmentJournal, +}; +use cellule_runtime::fleet::operations::{FleetAction, FleetScope}; +use cellule_runtime::identity::NodeId; + +impl CellNode { + /// Binds canonical directory withdrawal and durable boot retirement to the + /// existing host drain. Install after managed boot confirmation and before + /// readiness, using the original canonical version and journal record. + /// Facilities, runtime and lease maintenance join first. Stopped is exposed + /// only after checked withdrawal and confirmed journal retirement. A timeout + /// or missing reply retains the exact binding for the next drain attempt. + /// This proves boot closure, not relocation or reader/follower settlement. + pub fn install_fleet_boot_withdrawal( + &self, + directory: cellule_runtime::node::NodeDirectory, + observed: cellule_runtime::node::VersionedNodeAdvertisement, + original: cellule_runtime::fleet::operations::EnrollmentRecord, + journal: Arc, + ) -> cellule_runtime::Result<()> { + let state = self + .state + .lock() + .map_err(|_| Error::Control("CellNode lifecycle lock poisoned"))?; + if *state != NodeState::Starting { + return Err(Error::CellDraining); + } + let startup = self + .fleet_startup + .lock() + .map_err(|_| Error::Control("CellNode fleet startup lock poisoned"))?; + let startup = startup.as_ref().ok_or(Error::Control( + "CellNode boot withdrawal requires managed startup", + ))?; + if startup.boot.as_ref() != Some(&original) + || original.spec().scope != startup.intent.scope() + || original.spec().target.node != startup.intent.node() + || original.spec().target.session != self.session + || observed.advertisement().release() != self.application.registry().release_digest() + { + return Err(Error::Fenced); + } + let withdrawal = crate::fleet::withdrawal::FleetBootWithdrawal::new( + directory, observed, original, journal, + )?; + self.drain_owner.bind_boot_withdrawal(withdrawal) + } + + /// Captures one original native page through the shared finite fleet lane. + /// Authenticate the caller first. Both journal checks use the exact original + /// head/registry and endpoint intent. A lost/canceled waiter retains the + /// original native read and join; no acquisition or cleanup effect starts. + /// Missing owners are explicit Unbound coverage, never proof of empty roles. + pub async fn fleet_snapshot( + &self, + request: crate::fleet::FleetSnapshotRequest, + ) -> Result, Arc> { + if !self.is_management_ready() { + return Err(Arc::new(Error::CellDraining)); + } + let managed_startup = { + let startup = self + .fleet_startup + .lock() + .map_err(|_| Arc::new(Error::Control("CellNode fleet startup lock poisoned")))?; + startup.as_ref().is_some_and(|s| { + s.boot.is_some() + && s.intent.scope() == request.expected().head().scope() + && s.intent.node() == request.node() + && s.intent.session() == request.session() + }) + }; + let owners = crate::fleet::snapshot::SnapshotOwners { + state: Arc::clone(&self.state), + managed_startup, + readers: self + .try_owned_component::("read-replicas") + .map_err(Arc::new)?, + followers: self + .try_owned_component::( + FOLLOWER_STORE_COMPONENT, + ) + .map_err(Arc::new)?, + supervisor: self + .try_owned_component::(NODE_DURABILITY_SUPERVISOR_COMPONENT) + .map_err(Arc::new)?, + producer: self + .try_owned_component::( + NODE_DURABILITY_PROVIDER_COMPONENT, + ) + .map_err(Arc::new)?, + }; + let executor = self + .try_owned_component::(FLEET_ACTION_COMPONENT) + .map_err(Arc::new)? + .ok_or_else(|| Arc::new(Error::Control("fleet action executor is not installed")))?; + executor.snapshot(request, owners).await + } + + /// Installs journal-bound follower production in the existing durability + /// supervisor. Applications supply read-only signed preparation, authenticated + /// transports/authority and one shared atomic fleet journal. Install before + /// startup; every selected member is Pending before its leader CAS. + pub fn install_fleet_node_durability_provider( + &self, + scope: FleetScope, + node: NodeId, + journal: Arc, + provider: Arc

, + configuration: NodeDurabilitySupervisorConfig, + ) -> cellule_runtime::Result<()> { + if configuration.application != scope.application { + return Err(Error::Fenced); + } + if let Some(startup) = self + .fleet_startup + .lock() + .map_err(|_| Error::Control("CellNode fleet startup lock poisoned"))? + .as_ref() + && (startup.intent.scope() != scope || startup.intent.node() != node) + { + return Err(Error::Fenced); + } + let tasks = self + .task_group + .lock() + .map_err(|_| Error::Control("CellNode task group lock poisoned"))? + .clone() + .ok_or(Error::Control( + "fleet follower enrollment requires an installed task group", + ))?; + let producer = Arc::new(crate::durability::enrollment::FleetFollowerEnrollment::new( + scope, + node, + self.session, + provider, + journal, + self.runtime.clone(), + configuration.recruit_interval, + tasks.cancellation_token(), + )?); + self.install_node_durability_provider(producer, configuration) + } + + /// Captures current local follower enrollment, including unknown and paused + /// steps. Durable retirement removes local progress; None proves no absence, + /// completion, withdrawal or fleet finalization. + pub fn follower_enrollment_completion( + &self, + log_epoch: u64, + ) -> cellule_runtime::Result> { + match self.owned_component::( + NODE_DURABILITY_PROVIDER_COMPONENT, + ) { + Some(producer) => producer.completion(log_epoch), + None => Ok(None), + } + } + + /// Captures every retained follower epoch through bounded advisory pages. + /// A busy protocol includes preparation before any request exists. None + /// means no managed producer is installed and supplies no role coverage. + /// Current directory authority, inbound lanes and replacement policy must + /// also be observed before settlement or maintenance finalization. + pub fn fleet_follower_enrollments_page( + &self, + cursor: Option, + limit: usize, + now_ms: i64, + ) -> cellule_runtime::Result> { + if !(1..=crate::durability::enrollment::MAX_FOLLOWER_ENROLLMENT_EPOCHS).contains(&limit) + || now_ms < 0 + { + return Err(Error::Node("invalid follower enrollment inventory bounds")); + } + self.try_owned_component::( + NODE_DURABILITY_PROVIDER_COMPONENT, + )? + .map(|producer| producer.page(cursor, limit, now_ms)) + .transpose() + } + /// Confirms startup against an atomic current-intent/established-boot read. + /// The application first journals Pending, performs canonical directory + /// enrollment and publishes checked evidence. Missing, pending or ambiguous + /// evidence leaves all new roles and readiness closed. This read is safely + /// cancellable. Start removes the hold only after all owned startup probes. + pub async fn confirm_fleet_startup( + &self, + journal: &dyn FleetEnrollmentJournal, + enrollment_key: cellule_runtime::Digest, + ) -> cellule_runtime::Result<()> { + let original = self + .fleet_startup + .lock() + .map_err(|_| Error::Control("CellNode fleet startup lock poisoned"))? + .as_ref() + .map(|startup| startup.intent.clone()) + .ok_or(Error::Control("CellNode fleet startup is not configured"))?; + let observed = journal + .load_boot(original.scope(), original.node(), enrollment_key) + .await + .map_err(|source| Error::Facility { + name: "fleet-enrollment-journal", + source, + })? + .ok_or(Error::Control("CellNode fleet boot enrollment is absent"))?; + self.apply_fleet_boot(&original, enrollment_key, &observed, true, None) + } + + /// Refreshes this running boot's retained intent through one atomic journal read. + /// Call from the application's supervised membership/lease loop before renewal. + /// The original Established boot, acceptance time and evidence must still match. + /// Cordon/drain close the shared role gate without stopping existing Cell work, + /// reopening admission, or starting another scheduler. A read error or deadline + /// preserves the local state and must be retried or handled by the application; + /// it grants no authority to renew the boot's lease. Cancellation drops only + /// this read waiter, and a delayed reply cannot regress intent or reopen shutdown. + pub async fn refresh_fleet_intent( + &self, + journal: &dyn FleetEnrollmentJournal, + deadline: Instant, + ) -> cellule_runtime::Result { + if Instant::now() >= deadline { + return Err(Error::Deadline); + } + let (original, key) = { + let state = self + .state + .lock() + .map_err(|_| Error::Control("CellNode lifecycle lock poisoned"))?; + if !matches!( + *state, + NodeState::Ready | NodeState::ScalingDown | NodeState::Maintenance + ) { + return Err(Error::CellDraining); + } + let startup = self + .fleet_startup + .lock() + .map_err(|_| Error::Control("CellNode fleet startup lock poisoned"))?; + let startup = startup + .as_ref() + .ok_or(Error::Control("CellNode fleet startup is not configured"))?; + let boot = startup.boot.as_ref().ok_or(Error::Control( + "CellNode fleet boot enrollment is unconfirmed", + ))?; + ( + startup.intent.clone(), + boot.spec().key().map_err(crate::fleet::operation)?, + ) + }; + let observed = tokio::time::timeout_at( + deadline.into(), + journal.load_boot(original.scope(), original.node(), key), + ) + .await + .map_err(|_| Error::Deadline)? + .map_err(|source| Error::Facility { + name: "fleet-enrollment-journal", + source, + })? + .ok_or(Error::Control("CellNode fleet boot enrollment is absent"))?; + self.apply_fleet_boot(&original, key, &observed, false, Some(deadline))?; + Ok(observed.intent().clone()) + } + + fn apply_fleet_boot( + &self, + original: &cellule_runtime::fleet::operations::NodeIntent, + enrollment_key: cellule_runtime::Digest, + observed: &crate::fleet::FleetBootObservation, + startup_only: bool, + deadline: Option, + ) -> cellule_runtime::Result<()> { + let intent = observed.intent(); + let enrollment = observed.enrollment(); + if intent.scope() != original.scope() + || intent.node() != original.node() + || intent.session() != self.session + || intent.revision() < original.revision() + || enrollment.spec().key().map_err(crate::fleet::operation)? != enrollment_key + { + return Err(Error::Fenced); + } + // Lifecycle-before-startup matches start() and refresh admission. Shutdown cannot race an + // asynchronous read into reopening the already-draining host. + let state = self + .state + .lock() + .map_err(|_| Error::Control("CellNode lifecycle lock poisoned"))?; + if (startup_only && *state != NodeState::Starting) + || (!startup_only + && !matches!( + *state, + NodeState::Ready | NodeState::ScalingDown | NodeState::Maintenance + )) + { + return Err(Error::CellDraining); + } + let mut startup = self + .fleet_startup + .lock() + .map_err(|_| Error::Control("CellNode fleet startup lock poisoned"))?; + let startup = startup + .as_mut() + .ok_or(Error::Control("CellNode fleet startup is not configured"))?; + if intent.revision() < startup.intent.revision() + || (intent.revision() == startup.intent.revision() && intent != &startup.intent) + || startup.boot.as_ref().is_some_and(|boot| boot != enrollment) + { + return Err(Error::Fenced); + } + if deadline.is_some_and(|deadline| Instant::now() >= deadline) { + return Err(Error::Deadline); + } + match intent.mode() { + cellule_runtime::node::NodeMode::Active => {} + cellule_runtime::node::NodeMode::Cordoned => self.runtime.node_admission().cordon()?, + cellule_runtime::node::NodeMode::Draining => { + self.runtime.node_admission().begin_drain()? + } + } + startup.intent = intent.clone(); + startup.boot = Some(enrollment.clone()); + Ok(()) + } + /// Binds every reader activation and canonical removal to the durable fleet + /// registry. Install after read replicas and before start. Applications own + /// journal storage and authorization; the manager owns accepted finite work. + pub fn install_fleet_reader_enrollment( + &self, + scope: FleetScope, + node: NodeId, + journal: Arc, + ) -> cellule_runtime::Result<()> { + if let Some(startup) = self + .fleet_startup + .lock() + .map_err(|_| Error::Control("CellNode fleet startup lock poisoned"))? + .as_ref() + && (startup.intent.scope() != scope || startup.intent.node() != node) + { + return Err(Error::Fenced); + } + let manager = self + .owned_component::("read-replicas") + .ok_or(Error::Control( + "read replicas must be installed before reader enrollment", + ))?; + let binding = Arc::new(crate::read_replicas::enrollment::ReaderEnrollment::new( + scope, node, journal, + )?); + self.install_owned_component_with_drain( + "fleet-reader-enrollment", + Arc::clone(&binding), + || async { Ok(()) }, + )?; + if let Err(error) = manager.bind_enrollment(binding) { + self.remove_facility("fleet-reader-enrollment")?; + return Err(error); + } + Ok(()) + } + + /// Captures read-only request-bound evidence through the owned fleet lane. + /// Authenticate the caller first. The node checks current journal state and + /// actual authority/actor readiness; cached effect replies are not consulted + /// as proof of current serving. A dropped waiter leaves the finite job owned. + pub async fn inspect_fleet_action( + &self, + request: cellule_runtime::fleet::operations::FleetInspectionRequest, + ) -> Result, Arc> + { + if !self.is_management_ready() { + return Err(Arc::new(Error::CellDraining)); + } + let executor = self + .owned_component::(FLEET_ACTION_COMPONENT) + .ok_or_else(|| Arc::new(Error::Control("fleet action executor is not installed")))?; + executor.observe(request).await + } + + /// Installs the journal-bound fleet executor during startup. + /// + /// The application supplies a stable physical node and authenticated scope, + /// and a journal that atomically checks authorization at first acceptance. + /// The executor uses this node's runtime/session and joins finite accepted + /// work before runtime shutdown. Install before opening readiness. + pub fn install_fleet_actions( + &self, + scope: FleetScope, + node: NodeId, + journal: Arc, + cells: Arc, + ) -> cellule_runtime::Result<()> { + if let Some(startup) = self + .fleet_startup + .lock() + .map_err(|_| Error::Control("CellNode fleet startup lock poisoned"))? + .as_ref() + && (startup.intent.scope() != scope || startup.intent.node() != node) + { + return Err(Error::Fenced); + } + let executor = Arc::new(FleetActionExecutor::new( + self.runtime.clone(), + scope, + node, + self.session, + journal, + cells, + self.application.registry(), + )?); + let drained = Arc::clone(&executor); + self.install_owned_component_with_drain(FLEET_ACTION_COMPONENT, executor, move || { + let drained = Arc::clone(&drained); + async move { + drained + .drain() + .await + .map_err(|error| Box::new(error) as Box) + } + }) + } + + /// Executes a journal-bound movement or cordon independently of its waiter. + /// + /// Applications authenticate the caller before invoking this local boundary. + /// The executor supports settled movement, explicit busy maintenance + /// release, receiver inspection, and cordon through the shared role gate. + /// Role settlement and + /// finalization require their host barriers and are refused. Count a + /// result only when `committed` is true, then inspect current serving evidence. + /// Raw Inspect actions are refused: use `inspect_fleet_action` so a durable + /// historical acknowledgement cannot masquerade as a current observation. + pub async fn apply_fleet_action( + &self, + action: FleetAction, + now_ms: i64, + ) -> Result, Arc> { + if !self.is_management_ready() { + return Err(Arc::new(Error::CellDraining)); + } + let executor = self + .owned_component::(FLEET_ACTION_COMPONENT) + .ok_or_else(|| Arc::new(Error::Control("fleet action executor is not installed")))?; + executor.apply(action, now_ms).await + } +} diff --git a/crates/cellule-host/src/node/lifecycle.rs b/crates/cellule-host/src/node/lifecycle.rs index b4bcac5f..12f1b4d6 100644 --- a/crates/cellule-host/src/node/lifecycle.rs +++ b/crates/cellule-host/src/node/lifecycle.rs @@ -26,10 +26,43 @@ impl CellNode { .map_err(|_| Error::Control("CellNode lifecycle lock poisoned"))?; if *state == NodeState::Starting { self.require_components_present()?; - *state = NodeState::Ready; + let startup = self + .fleet_startup + .lock() + .map_err(|_| Error::Control("CellNode fleet startup lock poisoned"))?; + if let Some(startup) = startup.as_ref() { + if self.runtime.node_durability().is_some() + && self + .owned_component::( + NODE_DURABILITY_PROVIDER_COMPONENT, + ) + .is_none() + { + return Err(Error::Control( + "configured fleet durability requires managed follower enrollment", + )); + } + if startup.boot.is_none() { + return Err(Error::Control( + "CellNode fleet boot enrollment is unconfirmed", + )); + } + self.runtime + .node_admission() + .confirm_startup(startup.intent.mode())?; + *state = if self.runtime.node_admission().mode()? + == cellule_runtime::node::NodeMode::Active + { + NodeState::Ready + } else { + NodeState::Maintenance + }; + } else { + *state = NodeState::Ready; + } return Ok(()); } - if *state == NodeState::Ready { + if matches!(*state, NodeState::Ready | NodeState::Maintenance) { return Ok(()); } Err(Error::Control( @@ -65,129 +98,23 @@ impl CellNode { } /// Stops admission and completes every owned drain phase by `deadline`. pub async fn drain_until(&self, deadline: Option) -> cellule_runtime::Result<()> { - let _shutdown = self.shutdown_lock.lock().await; - self.drain_until_locked(deadline).await + let shutdown = Arc::clone(&self.shutdown_lock).lock_owned().await; + self.drain_until_locked(shutdown, deadline).await } pub(super) async fn drain_until_locked( &self, + shutdown: tokio::sync::OwnedMutexGuard<()>, deadline: Option, ) -> cellule_runtime::Result<()> { - { - let mut state = self - .state - .lock() - .map_err(|_| Error::Control("CellNode lifecycle lock poisoned"))?; - if *state == NodeState::Stopped { - return Ok(()); - } - *state = NodeState::Draining; - } - let task_group = self - .task_group - .lock() - .map(|task_group| task_group.clone()) - .map_err(|_| Error::Control("CellNode task group lock poisoned")); - let facilities = self - .facilities - .lock() - .map(|facilities| { - facilities - .iter() - .rev() - .map(|facility| (facility.name, Arc::clone(&facility.drain))) - .collect::>() - }) - .map_err(|_| Error::Control("CellNode facility lock poisoned")); - let mut first_error = None; - if let Some(error) = task_group.as_ref().err().map(|error| match error { - Error::Control(message) => Error::Control(message), - _ => Error::Control("CellNode task group unavailable during drain"), - }) { - first_error = Some(error); - } - if let Ok(Some(task_group)) = task_group.as_ref() { - task_group.cancel_work(); - } - match facilities { - Err(error) if first_error.is_none() => first_error = Some(error), - Err(_) => {} - Ok(facilities) => { - for (name, drain) in facilities { - let result = if name == "cell-coordination-tasks" { - // The task group is a retained owner, so its join must share the - // node deadline; an unbounded callback could strand shutdown. - match task_group.as_ref() { - Ok(Some(task_group)) => task_group.drain_work_until(deadline).await, - _ => drain().await, - } - } else { - match deadline { - Some(deadline) => { - match tokio::time::timeout_at(deadline.into(), drain()).await { - Ok(result) => result, - Err(_) => Err(Box::new(std::io::Error::new( - std::io::ErrorKind::TimedOut, - "CellNode facility drain deadline exceeded", - )) - as Box), - } - } - None => drain().await, - } - }; - if let Err(source) = result - && first_error.is_none() - { - first_error = Some(Error::Facility { name, source }); - } - } - } - } - let runtime_result = match deadline { - Some(deadline) => { - match tokio::time::timeout_at(deadline.into(), self.runtime.shutdown()).await { - Ok(result) => result, - Err(_) => Err(Error::Control("CellNode runtime drain deadline exceeded")), - } - } - None => self.runtime.shutdown().await, - }; - if first_error.is_none() { - first_error = runtime_result.err(); - } - // Session withdrawal fences the log authority. Keep its heartbeat live - // until runtime publication and the durable log-close barrier finish. - if let Ok(Some(task_group)) = task_group - && let Err(source) = task_group.drain_until(deadline).await - && first_error.is_none() - { - first_error = Some(Error::Facility { - name: "cell-coordination-tasks", - source, - }); - } - let result = match first_error { - Some(error) => Err(error), - None => Ok(()), - }; - let result = if result.is_ok() { - match self.facilities.lock() { - Ok(mut facilities) => { - facilities.clear(); - Ok(()) - } - Err(_) => Err(Error::Control("CellNode facility lock poisoned")), - } - } else { - result - }; - if result.is_ok() - && let Ok(mut state) = self.state.lock() - { - *state = NodeState::Stopped; - } - result + self.drain_owner.drain(shutdown, deadline).await } + + /// Captures the retained original host closing attempt and failure history. + /// This local diagnostic proves neither fleet relocation nor role settlement. + pub fn drain_observation(&self) -> cellule_runtime::Result> { + self.drain_owner.observe() + } + /// Idempotent alias for graceful drain used by process shutdown hooks. pub async fn shutdown(&self) -> cellule_runtime::Result<()> { self.drain().await diff --git a/crates/cellule-host/src/node/mod.rs b/crates/cellule-host/src/node/mod.rs index 68b2472d..d7df7a19 100644 --- a/crates/cellule-host/src/node/mod.rs +++ b/crates/cellule-host/src/node/mod.rs @@ -2,7 +2,12 @@ use super::*; use crate::builder::append_required_components; -use crate::durability::run_node_durability_supervisor; +use crate::durability::DurabilitySupervisor; + +pub(crate) struct FleetStartup { + pub(crate) intent: cellule_runtime::fleet::operations::NodeIntent, + pub(crate) boot: Option, +} /// One started application host with an ordered drain/shutdown boundary. pub struct CellNode { @@ -12,11 +17,46 @@ pub struct CellNode { pub(super) state: Arc>, pub(super) lease_installed: AtomicBool, pub(super) shutdown_lock: Arc>, + pub(super) drain_owner: Arc, pub(super) facilities: Arc>>, pub(super) required_components: Arc>>, pub(super) task_group: Arc>>>, + pub(super) fleet_startup: Mutex>, } impl CellNode { + pub(crate) fn from_runtime( + application: Arc, + runtime: CellRuntime, + session: SessionId, + required_components: Vec<&'static str>, + fleet_startup: Option, + ) -> Self { + let state = Arc::new(Mutex::new(NodeState::Starting)); + let facilities = Arc::new(Mutex::new(Vec::new())); + let task_group = Arc::new(Mutex::new(None)); + let drain_owner = Arc::new(drain::DrainOwner::new( + runtime.clone(), + Arc::clone(&state), + Arc::clone(&facilities), + Arc::clone(&task_group), + )); + Self { + application, + runtime, + session, + state, + lease_installed: AtomicBool::new(false), + shutdown_lock: Arc::new(tokio::sync::Mutex::new(())), + drain_owner, + facilities, + required_components: Arc::new(Mutex::new(required_components)), + task_group, + fleet_startup: Mutex::new( + fleet_startup.map(|intent| FleetStartup { intent, boot: None }), + ), + } + } + /// Returns the compiled application artifact owned by this node. #[must_use] pub fn application(&self) -> &CompiledApplication { @@ -46,6 +86,20 @@ impl CellNode { .and_then(|task_group| task_group.as_ref().map(|group| group.is_healthy())) .unwrap_or(false) } + /// Reports management/recovery availability, including a drained-mode boot + /// whose durable enrollment is confirmed but whose serving gate stays shut. + #[must_use] + pub fn is_management_ready(&self) -> bool { + matches!( + self.state(), + NodeState::Ready | NodeState::ScalingDown | NodeState::Maintenance + ) && self + .task_group + .lock() + .ok() + .and_then(|tasks| tasks.as_ref().map(|group| group.is_healthy())) + .unwrap_or(false) + } /// Returns current shared runtime admission metrics. #[must_use] pub fn stats(&self) -> CellRuntimeStats { @@ -79,6 +133,8 @@ impl CellNode { } mod components; +mod drain; +mod fleet; mod lifecycle; mod qualification; mod scale_down; diff --git a/crates/cellule-host/src/node/scale_down.rs b/crates/cellule-host/src/node/scale_down.rs index 2072957d..0c80e202 100644 --- a/crates/cellule-host/src/node/scale_down.rs +++ b/crates/cellule-host/src/node/scale_down.rs @@ -26,6 +26,29 @@ impl CellNode { .release_idle_cell(cell, source, generation) .await } + /// Releases one exact approved source identity and captures its final position. + /// + /// The application must authorize and durably accept its fleet action before + /// invoking this local mechanism. The runtime rechecks session, generation, + /// incarnation and epoch; its canonical close/release task supplies the root. + /// A subsequent authority observation is never substituted for that result. + /// [`Error::CellReleaseRefused`] proves this request stopped before canonical + /// close began. Other errors require outcome inspection or recovery. + pub async fn release_idle_cell_at( + &self, + cell: CellId, + source: SessionId, + generation: u64, + incarnation: cellule_runtime::identity::IncarnationId, + epoch: u64, + ) -> cellule_runtime::Result { + if !self.is_ready() { + return Err(Error::CellDraining); + } + self.runtime + .release_idle_cell_at(cell, source, generation, incarnation, epoch) + .await + } /// Stops new Cell acquisition while retaining the lease and current owners. pub fn begin_scale_down(&self) -> cellule_runtime::Result<()> { let mut state = self @@ -48,8 +71,11 @@ impl CellNode { &self, deadline: Instant, ) -> cellule_runtime::Result { - let _shutdown = self.shutdown_lock.lock().await; + let shutdown = Arc::clone(&self.shutdown_lock).lock_owned().await; if self.state() == NodeState::Stopped { + // A cancelled shutdown waiter may leave a returned host task whose + // epilogue still needs its original join before scale-down returns. + self.drain_until_locked(shutdown, Some(deadline)).await?; return Ok(ScaleDownStatus { remaining_cells: 0, settled_candidates: 0, @@ -103,7 +129,7 @@ impl CellNode { if Instant::now() >= deadline { return Ok(status); } - self.drain_until_locked(Some(deadline)).await?; + self.drain_until_locked(shutdown, Some(deadline)).await?; return Ok(status); } if Instant::now() >= deadline { diff --git a/crates/cellule-host/src/read_replicas/enrollment/inventory/mod.rs b/crates/cellule-host/src/read_replicas/enrollment/inventory/mod.rs new file mode 100644 index 00000000..f75c2b8f --- /dev/null +++ b/crates/cellule-host/src/read_replicas/enrollment/inventory/mod.rs @@ -0,0 +1,408 @@ +//! Bounded read-only progress while the canonical activation owner awaits I/O. +use super::*; +use cellule_runtime::node::NodeMode; + +const PAGE_BYTES: usize = 1 << 20; +const PAGE_ENTRIES: usize = 128; + +#[cfg(test)] +mod tests; + +/// Continuation bound to this manager and the original producer progress. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ReaderEnrollmentInventoryCursor { + topology: Digest, + after: CellId, +} +impl ReaderEnrollmentInventoryCursor { + /// Encodes a fixed-width application continuation. + #[must_use] + pub fn to_bytes(self) -> [u8; 64] { + let mut bytes = [0; 64]; + bytes[..32].copy_from_slice(self.topology.as_bytes()); + bytes[32..].copy_from_slice(self.after.as_bytes()); + bytes + } + /// Decodes the fixed width; each page rechecks the original progress. + pub fn from_bytes(bytes: &[u8]) -> Result { + let bytes: &[u8; 64] = bytes + .try_into() + .map_err(|_| Error::Node("invalid reader enrollment cursor width"))?; + let mut topology = [0; 32]; + topology.copy_from_slice(&bytes[..32]); + let mut after = [0; 32]; + after.copy_from_slice(&bytes[32..]); + if topology == [0; 32] || after == [0; 32] { + return Err(Error::Node("zero reader enrollment cursor identity")); + } + Ok(Self { + topology: Digest::from_bytes(topology), + after: CellId::from_bytes(after), + }) + } +} + +/// Accepted finite jobs, including preparation before a Pending request exists. +/// These counts never infer native closure from a task handle's disappearance. +#[derive(Clone, Debug)] +pub struct ReaderEnrollmentJobs { + retained: usize, + running: usize, + unobserved: usize, + joining: usize, + draining: bool, + task_failure: Option>, + protocol_failure: Option>, +} +impl ReaderEnrollmentJobs { + /// Counts every retained join handle, including completed protocol jobs. + #[must_use] + pub const fn retained(&self) -> usize { + self.retained + } + /// Counts jobs with no response that are still executing or preparing. + #[must_use] + pub const fn running(&self) -> usize { + self.running + } + /// Counts finished jobs without their canonical protocol response. + #[must_use] + pub const fn unobserved(&self) -> usize { + self.unobserved + } + /// Counts handles currently held by the retained join owner; state is unknown. + #[must_use] + pub const fn joining(&self) -> usize { + self.joining + } + /// Reports permanent closure of this producer's job admission. + #[must_use] + pub const fn draining(&self) -> bool { + self.draining + } + /// Returns the original retained task failure, without replacing its source. + #[must_use] + pub fn task_failure(&self) -> Option<&Arc> { + self.task_failure.as_ref() + } + /// Returns the first original response failure observed in this capture. + #[must_use] + pub fn protocol_failure(&self) -> Option<&Arc> { + self.protocol_failure.as_ref() + } +} + +/// Bounded original responsibilities, including unresolved acceptance/open/close. +/// The page retains one MiB from the shared node byte ledger. It is advisory: +/// current authority, replacement policy and durable settlement remain required. +pub struct ReaderEnrollmentInventoryPage { + scope: FleetScope, + node: NodeId, + session: SessionId, + mode: NodeMode, + observed_at_ms: i64, + topology: Digest, + total_enrollments: usize, + jobs: ReaderEnrollmentJobs, + entries: Vec, + next: Option, + _memory: NodeByteReservation, +} +impl ReaderEnrollmentInventoryPage { + /// Returns the producer's original installed journal scope. + #[must_use] + pub const fn scope(&self) -> FleetScope { + self.scope + } + /// Returns the producer's original receiving physical node. + #[must_use] + pub const fn node(&self) -> NodeId { + self.node + } + /// Returns the exact manager boot session. + #[must_use] + pub const fn session(&self) -> SessionId { + self.session + } + /// Returns current shared admission mode, independent of local role counts. + #[must_use] + pub const fn mode(&self) -> NodeMode { + self.mode + } + /// Returns the supplied original capture time; no proof is refreshed. + #[must_use] + pub const fn observed_at_ms(&self) -> i64 { + self.observed_at_ms + } + /// Returns the manager and captured producer-state fingerprint. + #[must_use] + pub const fn topology(&self) -> Digest { + self.topology + } + /// Counts every retained responsibility, including those outside this page. + #[must_use] + pub const fn total_enrollments(&self) -> usize { + self.total_enrollments + } + /// Returns accepted job state, including pre-journal preparation and failure. + #[must_use] + pub const fn jobs(&self) -> &ReaderEnrollmentJobs { + &self.jobs + } + /// Returns original requests, native states and source errors in Cell order. + #[must_use] + pub fn entries(&self) -> &[ReaderEnrollmentCompletion] { + &self.entries + } + /// Continues only while producer topology/progress still matches. + #[must_use] + pub const fn next(&self) -> Option { + self.next + } +} + +impl ReadReplicaManager { + /// Captures durable reader producer progress without awaiting its peer/journal + /// calls or acquiring the manager's activation lane. Unbound managers return + /// None, which supplies no enrollment coverage. Installed native views must + /// also be scanned through `fleet_readers_page`. A page cannot authorize + /// shutdown, retry unknown opening or erase an accepted job. + pub fn fleet_reader_enrollments_page( + &self, + cursor: Option, + limit: usize, + now_ms: i64, + ) -> Result> { + if !(1..=PAGE_ENTRIES).contains(&limit) || now_ms < 0 { + return Err(Error::Node("invalid reader enrollment inventory bounds")); + } + self.bound_enrollment()? + .map(|binding| binding.page(self, cursor, limit, now_ms)) + .transpose() + } +} + +impl ReaderEnrollment { + fn jobs_observation(&self) -> Result { + let (jobs, draining, failure) = { + let bank = self + .jobs + .lock() + .map_err(|_| Error::Control("reader job bank poisoned"))?; + (bank.jobs.clone(), bank.draining, bank.failure.clone()) + }; + Ok(observe_jobs(jobs, draining, failure)) + } + + fn page( + &self, + manager: &ReadReplicaManager, + cursor: Option, + limit: usize, + now_ms: i64, + ) -> Result { + let memory = manager.runtime.try_reserve_node_bytes(PAGE_BYTES)?; + let mut records = { + let records = self.records()?; + if records.len() > MAX_READ_VIEWS { + return Err(Error::Capacity("reader enrollment inventory bound")); + } + records + .iter() + .map(|(cell, record)| (*cell, record.clone())) + .collect::>() + }; + records.sort_unstable_by_key(|(cell, _)| *cell.as_bytes()); + let jobs = self.jobs_observation()?; + let mode = manager.runtime.node_admission().mode()?; + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.reader-enrollment-inventory.v1\0"); + hash.update(self.scope.fleet.as_bytes()); + hash.update(self.scope.application.as_bytes()); + hash.update(self.node.as_bytes()); + hash.update(manager.session.as_bytes()); + hash.update(&[ + mode as u8, + u8::from(jobs.draining), + u8::from(jobs.task_failure.is_some()), + u8::from(jobs.protocol_failure.is_some()), + ]); + for count in [ + records.len(), + jobs.retained, + jobs.running, + jobs.unobserved, + jobs.joining, + ] { + hash.update(&(count as u64).to_be_bytes()); + } + let mut entries = Vec::with_capacity(limit); + // Account index and row storage before cloning dynamic request/source + // bytes. Shared Arc errors and native ownership are not copied. + let mut bytes = records + .len() + .checked_mul(std::mem::size_of::<(CellId, Record)>()) + .and_then(|v| { + v.checked_add( + limit * std::mem::size_of::() + + 4096 + + 2 * MAX_RECORD_BYTES as usize, + ) + }) + .ok_or(Error::Capacity("reader inventory byte overflow"))?; + let mut after_found = cursor.is_none(); + let mut page_full = false; + for (cell, record) in &records { + let progress = data(record)?; + let spec = progress.spec.to_bytes().map_err(operation)?; + let original = progress + .original + .as_ref() + .map(EnrollmentRecord::to_bytes) + .transpose() + .map_err(operation)?; + hash.update(cell.as_bytes()); + field(&mut hash, &spec); + hash.update(&[u8::from(original.is_some())]); + if let Some(original) = &original { + field(&mut hash, original); + } + hash.update(&[ + u8::from(progress.published), + u8::from(progress.opening_started), + u8::from(progress.opening_joined), + u8::from(progress.execution_error.is_some()), + u8::from(progress.journal_error.is_some()), + ]); + let (tag, digest) = match progress.event { + None => (0, None), + Some(EnrollmentEvent::Established(d)) => (1, Some(d)), + Some(EnrollmentEvent::Refused(d)) => (2, Some(d)), + Some(EnrollmentEvent::Retired(d)) => (3, Some(d)), + }; + hash.update(&[tag]); + if let Some(digest) = digest { + hash.update(digest.as_bytes()); + } + hash.update(progress.source.description().code.as_bytes()); + hash.update(&progress.source.description().schema.to_be_bytes()); + field(&mut hash, progress.source.owner().endpoint.as_bytes()); + if cursor.is_some_and(|cursor| cursor.after == *cell) { + after_found = true; + } + if page_full + || entries.len() >= limit + || cursor.is_some_and(|cursor| cell.as_bytes() <= cursor.after.as_bytes()) + { + continue; + } + let dynamic = spec + .len() + .checked_mul(2) + .and_then(|v| v.checked_add(original.as_ref().map_or(0, Vec::len) * 2)) + .and_then(|v| { + v.checked_add( + progress.source.target().partition().len() + + progress.source.owner().endpoint.len() + + 512, + ) + }) + .ok_or(Error::Capacity("reader inventory byte overflow"))?; + if !admit_entry(&mut bytes, dynamic, entries.is_empty())? { + // Stop copying before the first row that would exceed admission, + // but hash every remaining original row for stable continuation. + page_full = true; + continue; + } + entries.push(progress.completion()); + } + let topology = Digest::from_bytes(*hash.finalize().as_bytes()); + if !after_found || cursor.is_some_and(|cursor| cursor.topology != topology) { + return Err(Error::Node( + "reader enrollment inventory changed; restart scan", + )); + } + let next = entries.last().and_then(|last| { + let cell = last.source.description().cell; + records + .last() + .is_some_and(|(last, _)| last.as_bytes() > cell.as_bytes()) + .then_some(ReaderEnrollmentInventoryCursor { + topology, + after: cell, + }) + }); + Ok(ReaderEnrollmentInventoryPage { + scope: self.scope, + node: self.node, + session: manager.session, + mode, + observed_at_ms: now_ms, + topology, + total_enrollments: records.len(), + jobs, + entries, + next, + _memory: memory, + }) + } +} +fn field(hash: &mut blake3::Hasher, bytes: &[u8]) { + hash.update(&(bytes.len() as u64).to_be_bytes()); + hash.update(bytes); +} + +fn admit_entry(bytes: &mut usize, dynamic: usize, empty: bool) -> Result { + let next = bytes + .checked_add(dynamic) + .ok_or(Error::Capacity("reader inventory byte overflow"))?; + if next > PAGE_BYTES { + if empty { + return Err(Error::Capacity("reader enrollment inventory page bytes")); + } + return Ok(false); + } + *bytes = next; + Ok(true) +} + +fn observe_jobs( + jobs: Vec>>, + draining: bool, + failure: Option>, +) -> ReaderEnrollmentJobs { + let mut observation = ReaderEnrollmentJobs { + retained: jobs.len(), + running: 0, + unobserved: 0, + joining: 0, + draining, + task_failure: failure, + protocol_failure: None, + }; + for job in jobs { + let Ok(job) = job.try_lock() else { + observation.joining += 1; + continue; + }; + let response = job.response.borrow(); + match response.as_ref() { + Some(Err(error)) => { + observation + .protocol_failure + .get_or_insert_with(|| error.clone()); + } + Some(Ok(_)) => {} + None if job.task.as_ref().is_some_and(|task| !task.is_finished()) => { + observation.running += 1 + } + None => observation.unobserved += 1, + } + if let Some(error) = &job.failure { + observation + .task_failure + .get_or_insert_with(|| error.clone()); + } + } + observation +} diff --git a/crates/cellule-host/src/read_replicas/enrollment/inventory/tests.rs b/crates/cellule-host/src/read_replicas/enrollment/inventory/tests.rs new file mode 100644 index 00000000..0baced64 --- /dev/null +++ b/crates/cellule-host/src/read_replicas/enrollment/inventory/tests.rs @@ -0,0 +1,188 @@ +use super::*; +use tokio::sync::watch; + +#[test] +fn variable_row_budget_stops_before_copy_and_cannot_skip_an_oversized_first_row() { + let mut bytes = PAGE_BYTES - 128; + assert!(admit_entry(&mut bytes, 64, true).unwrap()); + assert_eq!(bytes, PAGE_BYTES - 64); + assert!(!admit_entry(&mut bytes, 65, false).unwrap()); + assert_eq!(bytes, PAGE_BYTES - 64); + assert!(admit_entry(&mut bytes, 64, false).unwrap()); + assert_eq!(bytes, PAGE_BYTES); + assert!(!admit_entry(&mut bytes, 1, false).unwrap()); + assert!(matches!( + admit_entry(&mut bytes, 1, true), + Err(Error::Capacity(_)) + )); + assert_eq!(bytes, PAGE_BYTES); + bytes = usize::MAX; + assert!(matches!( + admit_entry(&mut bytes, 1, false), + Err(Error::Capacity(_)) + )); + assert_eq!(bytes, usize::MAX); +} + +#[test] +fn cursor_rejects_width_and_zero_identity() { + for bytes in [&[][..], &[1; 63][..], &[1; 65][..], &[0; 64][..]] { + assert!(ReaderEnrollmentInventoryCursor::from_bytes(bytes).is_err()); + } + for range in [0..32, 32..64] { + let mut bytes = [1; 64]; + bytes[range].fill(0); + assert!(ReaderEnrollmentInventoryCursor::from_bytes(&bytes).is_err()); + } + let bytes = [3; 64]; + assert_eq!( + ReaderEnrollmentInventoryCursor::from_bytes(&bytes) + .unwrap() + .to_bytes(), + bytes + ); +} + +#[tokio::test] +async fn live_job_and_joining_owner_remain_unknown() { + let (sender, response) = watch::channel(None); + let task = tokio::spawn(async move { + let _sender = sender; + std::future::pending::<()>().await; + }); + let job = Arc::new(Mutex::new(ReaderJob { + task: Some(task), + failure: None, + response, + })); + let running = observe_jobs(vec![job.clone()], false, None); + assert_eq!( + ( + running.retained(), + running.running(), + running.unobserved(), + running.joining() + ), + (1, 1, 0, 0) + ); + let mut owner = job.lock().await; + let joining = observe_jobs(vec![job.clone()], true, None); + assert_eq!( + ( + joining.retained(), + joining.running(), + joining.unobserved(), + joining.joining() + ), + (1, 0, 0, 1) + ); + assert!(joining.draining()); + owner.task.as_ref().unwrap().abort(); + assert!(owner.join().await.is_err()); +} + +#[tokio::test] +async fn handle_completion_without_original_response_is_unobserved() { + let (sender, response) = watch::channel(None); + let task = tokio::spawn(async move { + drop(sender); + panic!("original reader owner failed without a protocol response"); + }); + tokio::time::timeout(Duration::from_secs(1), async { + while !task.is_finished() { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + let job = Arc::new(Mutex::new(ReaderJob { + task: Some(task), + failure: None, + response, + })); + let observation = observe_jobs(vec![job.clone()], false, None); + assert_eq!( + ( + observation.retained(), + observation.running(), + observation.unobserved(), + observation.joining() + ), + (1, 0, 1, 0) + ); + let original = { + let mut owner = job.lock().await; + assert!(owner.join().await.is_err()); + let original = owner.failure.clone().unwrap(); + assert!(matches!(original.as_ref(), Error::Facility { source, .. } + if source.downcast_ref::().unwrap().is_panic())); + original + }; + let joined = observe_jobs(vec![job], false, None); + assert_eq!(joined.unobserved(), 1); + assert!(Arc::ptr_eq(joined.task_failure().unwrap(), &original)); +} + +#[tokio::test] +async fn original_success_response_distinguishes_returned_protocol_from_unknown_task() { + let receipt = Receipt { + cell: CellId::from_bytes([1; 32]), + incarnation: IncarnationId::from_bytes([2; 16]), + commit_sequence: 3, + }; + let (_, response) = watch::channel(Some(Ok(receipt))); + let job = Arc::new(Mutex::new(ReaderJob { + task: None, + failure: None, + response, + })); + let observation = observe_jobs(vec![job], false, None); + assert_eq!( + ( + observation.retained(), + observation.running(), + observation.unobserved(), + observation.joining() + ), + (1, 0, 0, 0) + ); + assert!(observation.protocol_failure().is_none()); + assert!(observation.task_failure().is_none()); +} + +#[tokio::test] +async fn protocol_and_task_errors_preserve_original_arcs_separately() { + let protocol = Arc::new(Error::Node("original reader protocol failure")); + let task_error = Arc::new(Error::Node("original reader task failure")); + let (_, response) = watch::channel(Some(Err(protocol.clone()))); + let job = Arc::new(Mutex::new(ReaderJob { + task: None, + failure: Some(task_error.clone()), + response, + })); + let observation = observe_jobs(vec![job], false, None); + assert_eq!( + ( + observation.retained(), + observation.running(), + observation.unobserved(), + observation.joining() + ), + (1, 0, 0, 0) + ); + assert!(Arc::ptr_eq( + observation.protocol_failure().unwrap(), + &protocol + )); + assert!(Arc::ptr_eq( + observation.task_failure().unwrap(), + &task_error + )); + let bank_failure = Arc::new(Error::Node("original bank failure")); + let observation = observe_jobs(vec![], true, Some(bank_failure.clone())); + assert!(Arc::ptr_eq( + observation.task_failure().unwrap(), + &bank_failure + )); + assert!(observation.draining()); +} diff --git a/crates/cellule-host/src/read_replicas/enrollment/jobs.rs b/crates/cellule-host/src/read_replicas/enrollment/jobs.rs new file mode 100644 index 00000000..270a1189 --- /dev/null +++ b/crates/cellule-host/src/read_replicas/enrollment/jobs.rs @@ -0,0 +1,152 @@ +//! Bounded accepted-job ownership, retained joins and original task failures. + +use super::*; +use tokio::sync::watch; + +impl ReaderJob { + pub(super) async fn join(&mut self) -> Result<()> { + if let Some(task) = self.task.as_mut() { + let result = task.await; + self.task = None; + if let Err(source) = result { + self.failure = Some(Arc::new(Error::Facility { + name: "fleet-reader-task", + source: Box::new(source), + })); + } + } + match &self.failure { + Some(error) => Err(retained(Arc::clone(error))), + None => Ok(()), + } + } +} + +impl ReaderEnrollment { + /// Own the entire finite protocol before the first journal call. Reaping + /// finished tasks never drops a still-running accepted opening. + pub(in crate::read_replicas) async fn activate( + self: &Arc, + manager: ReadReplicaManager, + request: ActivationRequest, + ) -> Result { + self.reap().await?; + let (mut response, retained_job) = { + let mut jobs = self + .jobs + .lock() + .map_err(|_| Error::Control("reader job bank poisoned"))?; + if jobs.draining { + return Err(Error::RuntimeClosed); + } + if jobs.jobs.len() >= MAX_JOBS { + return Err(Error::Capacity("reader enrollment job bound")); + } + let reservation = manager + .runtime + .try_reserve_node_bytes(MAX_RECORD_BYTES as usize)?; + let (sender, response) = watch::channel(None); + let task = tokio::spawn(async move { + let _reservation = reservation; + let result = match request { + ActivationRequest::Hint(target, origin) => { + manager.activate_open(target, origin).await + } + ActivationRequest::Source(source) => manager.activate_initial(*source).await, + } + .map_err(Arc::new); + let _ = sender.send(Some(result)); + }); + let job = Arc::new(Mutex::new(ReaderJob { + task: Some(task), + failure: None, + response: response.clone(), + })); + jobs.jobs.push(Arc::clone(&job)); + (response, job) + }; + loop { + let completed = response.borrow().clone(); + if let Some(result) = completed { + // Completion precedes the task's epilogue and byte-token drop. + // Join the exact retained task before returning usable credit; + // cancellation leaves its handle in the same producer bank. + retained_job.lock().await.join().await?; + return result.map_err(retained); + } + if let Err(source) = response.changed().await { + // Prefer the original task failure to its downstream channel + // closure, even if a concurrent reaper removed the bank entry. + retained_job.lock().await.join().await?; + return Err(Error::Facility { + name: "fleet-reader-completion", + source: Box::new(source), + }); + } + } + } + + pub(in crate::read_replicas) async fn reap(&self) -> Result<()> { + let jobs = self + .jobs + .lock() + .map_err(|_| Error::Control("reader job bank poisoned"))? + .jobs + .clone(); + for job in jobs { + let Ok(mut state) = job.try_lock() else { + continue; + }; + if state.task.as_ref().is_some_and(|task| !task.is_finished()) { + continue; + } + let joined = state.join().await; + let mut bank = self + .jobs + .lock() + .map_err(|_| Error::Control("reader job bank poisoned"))?; + if let Err(error) = joined { + bank.failure.get_or_insert(Arc::new(error)); + } + bank.jobs.retain(|retained| !Arc::ptr_eq(retained, &job)); + } + match &self + .jobs + .lock() + .map_err(|_| Error::Control("reader job bank poisoned"))? + .failure + { + Some(error) => Err(retained(Arc::clone(error))), + None => Ok(()), + } + } + + pub(in crate::read_replicas) fn close_admission(&self) -> Result<()> { + self.jobs + .lock() + .map_err(|_| Error::Control("reader job bank poisoned"))? + .draining = true; + Ok(()) + } + + // Await the handle in place. Cancellation drops only its lock guard, so + // the next shutdown waiter joins the same accepted work and original error. + pub(in crate::read_replicas) async fn join(&self) -> Result<()> { + let jobs = self + .jobs + .lock() + .map_err(|_| Error::Control("reader job bank poisoned"))? + .jobs + .clone(); + for job in jobs { + if let Err(error) = job.lock().await.join().await { + self.jobs + .lock() + .map_err(|_| Error::Control("reader job bank poisoned"))? + .failure + .get_or_insert(Arc::new(error)); + } + } + self.reap().await + } +} diff --git a/crates/cellule-host/src/read_replicas/enrollment/maintenance.rs b/crates/cellule-host/src/read_replicas/enrollment/maintenance.rs new file mode 100644 index 00000000..ab6bbfba --- /dev/null +++ b/crates/cellule-host/src/read_replicas/enrollment/maintenance.rs @@ -0,0 +1,77 @@ +//! Original reader/intent barriers for replacement-checked local evacuation. +use super::*; +use crate::fleet::FleetRoster; +use cellule_runtime::fleet::operations::{MaintenanceOperation, MaintenancePhase}; +use cellule_runtime::node::NodeMode; + +impl ReaderEnrollment { + pub(in crate::read_replicas) async fn maintenance_roster( + &self, + original: &EnrollmentRecord, + maintenance: &MaintenanceOperation, + deadline: tokio::time::Instant, + runtime: &CellRuntime, + ) -> Result { + if original.spec().scope != self.scope + || original.spec().target.node != self.node + || maintenance.node() != self.node + || original.spec().target.session != maintenance.session() + { + return Err(Error::Fenced); + } + let snapshot = self + .journal + .load_snapshot(self.scope) + .await + .map_err(journal)?; + if now_ms()? >= maintenance.deadline_ms() { + return Err(Error::Node("reader maintenance operation deadline elapsed")); + } + if snapshot.head().maintenance() != Some(maintenance) + || maintenance.phase() != MaintenancePhase::Evacuating + || snapshot.registry().bootstrap_revision().is_none() + { + return Err(Error::Fenced); + } + let roster = + FleetRoster::collect_admitted(self.journal.as_ref(), &snapshot, deadline, runtime) + .await?; + let intent = roster + .intents() + .iter() + .find(|intent| intent.node() == self.node) + .ok_or(Error::Fenced)?; + if intent.session() != maintenance.session() + || intent.revision() != maintenance.intent_revision() + || intent.mode() != NodeMode::Draining + { + return Err(Error::Fenced); + } + let current = roster + .enrollments() + .iter() + .find(|row| row.spec() == original.spec()) + .ok_or(Error::Control("original reader enrollment is absent"))?; + current + .validate_replay(original.spec()) + .map_err(operation)?; + if current.accepted_at_ms() != original.accepted_at_ms() + || current.established_evidence() != original.established_evidence() + || !matches!( + current.status(), + EnrollmentStatus::Established | EnrollmentStatus::Retired + ) + { + return Err(Error::Fenced); + } + Ok(roster) + } + + pub(in crate::read_replicas) async fn confirm_maintenance_roster( + &self, + roster: &FleetRoster, + deadline: tokio::time::Instant, + ) -> Result<()> { + roster.confirm(self.journal.as_ref(), deadline).await + } +} diff --git a/crates/cellule-host/src/read_replicas/enrollment/mod.rs b/crates/cellule-host/src/read_replicas/enrollment/mod.rs new file mode 100644 index 00000000..379d7694 --- /dev/null +++ b/crates/cellule-host/src/read_replicas/enrollment/mod.rs @@ -0,0 +1,291 @@ +//! Durable ownership around the manager's canonical activation/closure lane. + +use super::*; +use crate::fleet::{FleetEnrollmentAcceptance, FleetJournal}; +use cellule_runtime::cell::actor::NodeByteReservation; +use cellule_runtime::fleet::operations::{ + EnrollmentEndpoint, EnrollmentEvent, EnrollmentRecord, EnrollmentRole, EnrollmentSpec, + EnrollmentStatus, FleetScope, MAX_RECORD_BYTES, PublishedPosition, +}; +use cellule_runtime::identity::NodeId; +use std::sync::Mutex as StdMutex; +use tokio::task::JoinHandle; + +mod inventory; +mod jobs; +mod maintenance; +mod protocol; +pub use inventory::{ + ReaderEnrollmentInventoryCursor, ReaderEnrollmentInventoryPage, ReaderEnrollmentJobs, +}; + +const MAX_JOBS: usize = 32; + +pub(super) enum ActivationRequest { + Hint(CellTarget, SessionId), + Source(Box), +} + +pub(crate) struct ReaderEnrollment { + scope: FleetScope, + node: NodeId, + journal: Arc, + records: StdMutex>, + jobs: StdMutex, +} + +#[derive(Default)] +struct JobBank { + draining: bool, + jobs: Vec>>, + failure: Option>, +} + +struct ReaderJob { + task: Option>, + failure: Option>, + response: tokio::sync::watch::Receiver>>>, +} +/// Retained finite-protocol result for one locally owned reader obligation. +/// Errors preserve their original sources; absence is not retirement evidence. +#[derive(Clone)] +pub struct ReaderEnrollmentCompletion { + /// Exact immutable request, including both physical intents and pinned root. + pub spec: EnrollmentSpec, + /// Original catalog/authority/signed-boot source used by canonical opening. + pub source: ReadReplicaSource, + /// Original acceptance; None means the acceptance reply remains ambiguous. + pub accepted: Option, + /// Checked opening or joined-closure evidence awaiting/confirming publication. + pub event: Option, + /// Whether the journal confirmed this exact event. + pub published: bool, + /// Whether the retained owner dispatched canonical native opening. + pub opening_started: bool, + /// Whether canonical opening returned after joining its accepted native work. + pub opening_joined: bool, + /// Original native opening failure, independent of publication failure. + pub execution_error: Option>, + /// Original unresolved result-publication failure. + pub journal_error: Option>, +} + +type Record = Arc>; + +struct Responsibility { + spec: EnrollmentSpec, + source: ReadReplicaSource, + original: Option, + // A publication failure retains the same evidence, never a new opening. + event: Option, + published: bool, + opening_started: bool, + opening_joined: bool, + execution_error: Option>, + journal_error: Option>, + _retained: NodeByteReservation, +} + +#[derive(Debug)] +struct RetainedFailure(Arc); +impl std::fmt::Display for RetainedFailure { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + self.0.fmt(f) + } +} +impl std::error::Error for RetainedFailure { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + Some(self.0.as_ref()) + } +} +fn retained(error: Arc) -> Error { + Error::Facility { + name: "fleet-reader-enrollment", + source: Box::new(RetainedFailure(error)), + } +} +fn journal(source: Box) -> Error { + Error::Facility { + name: "fleet-enrollment-journal", + source, + } +} +fn operation(error: cellule_runtime::fleet::operations::OperationError) -> Error { + Error::FleetOperation(Box::new(error)) +} + +impl ReaderEnrollment { + pub(crate) fn new( + scope: FleetScope, + node: NodeId, + journal: Arc, + ) -> Result { + cellule_runtime::fleet::operations::FleetHead::new(scope, 0).map_err(operation)?; + if node.as_bytes().iter().all(|byte| *byte == 0) { + return Err(Error::Node("invalid reader enrollment physical node")); + } + Ok(Self { + scope, + node, + journal, + records: StdMutex::new(HashMap::new()), + jobs: StdMutex::new(JobBank::default()), + }) + } + + pub(super) fn matches_boot( + &self, + advertisement: &cellule_runtime::node::NodeAdvertisement, + ) -> bool { + advertisement.node() == self.node && advertisement.fleet() == self.scope.fleet + } + + async fn spec( + &self, + manager: &ReadReplicaManager, + source: &ReadReplicaSource, + ) -> Result { + if source.fleet() != self.scope.fleet + || source.target().application() != self.scope.application + { + return Err(Error::Fenced); + } + let receiver = manager + .directory + .load_if_live(manager.session, now_ms()?) + .await? + .ok_or(Error::Fenced)?; + if !self.matches_boot(receiver.advertisement()) { + return Err(Error::Fenced); + } + let version = self + .journal + .load_snapshot(self.scope) + .await + .map_err(journal)? + .registry(); + let mut after = None; + let mut endpoints = HashMap::new(); + let mut count = 0; + loop { + let page = self + .journal + .intents_page(version, after, 128) + .await + .map_err(journal)?; + if page.version() != version || page.after() != after { + return Err(Error::Fenced); + } + count += page.entries().len(); + if count > MAX_LIVE_NODES { + return Err(Error::Capacity("reader intent inventory bound")); + } + for intent in page.entries() { + if intent.node() == source.node() || intent.node() == self.node { + endpoints.insert( + intent.node(), + EnrollmentEndpoint { + node: intent.node(), + session: intent.session(), + intent_revision: intent.revision(), + }, + ); + } + } + if endpoints.len() == 2 || page.next().is_none() { + break; + } + if page.next().map(|id| *id.as_bytes()) <= after.map(|id| *id.as_bytes()) { + return Err(Error::Fenced); + } + after = page.next(); + } + let origin = endpoints + .get(&source.node()) + .copied() + .ok_or(Error::Fenced)?; + let target = endpoints.get(&self.node).copied().ok_or(Error::Fenced)?; + if origin.session != source.owner().session || target.session != manager.session { + return Err(Error::Fenced); + } + Ok(EnrollmentSpec { + scope: self.scope, + request: Digest::from_bytes(*blake3::hash(Uuid::now_v7().as_bytes()).as_bytes()), + source: Some(origin), + target, + role: EnrollmentRole::Reader { + target: source.target().clone(), + position: PublishedPosition { + incarnation: source.description().incarnation, + epoch: source.epoch(), + root: source.root().clone(), + }, + }, + }) + } + + fn records(&self) -> Result>> { + self.records + .lock() + .map_err(|_| Error::Control("reader enrollment index poisoned")) + } + + pub(super) fn completion(&self, cell: CellId) -> Result> { + let record = self.records()?.get(&cell).cloned(); + record + .map(|record| Ok(data(&record)?.completion())) + .transpose() + } + + pub(super) fn unresolved_cells(&self) -> Result> { + Ok(self.records()?.keys().copied().collect()) + } + + pub(in crate::read_replicas) fn reconciliation_cells( + &self, + runtime: &CellRuntime, + views: impl ExactSizeIterator, + ) -> Result<(Vec, NodeByteReservation)> { + let records = self.records()?; + if views.len() > MAX_READ_VIEWS || records.len() > MAX_READ_VIEWS { + return Err(Error::Capacity("read reconciliation responsibility bound")); + } + // The caller holds its view-index read guard; this short record guard + // pins the other count through charge and copy. Allocate exact observed + // capacity, so a small node need not admit a maximum-size empty scan. + let count = views.len() + records.len(); + let memory = runtime + .try_reserve_node_metadata_bytes(count * std::mem::size_of::() + 4096)?; + let mut cells = Vec::with_capacity(count); + cells.extend(views); + cells.extend(records.keys().copied()); + Ok((cells, memory)) + } +} + +fn data(record: &Record) -> Result> { + record + .lock() + .map_err(|_| Error::Control("reader enrollment progress poisoned")) +} + +impl Responsibility { + fn completion(&self) -> ReaderEnrollmentCompletion { + ReaderEnrollmentCompletion { + spec: self.spec.clone(), + source: self.source.clone(), + accepted: self.original.clone(), + event: self.event, + published: self.published, + opening_started: self.opening_started, + opening_joined: self.opening_joined, + execution_error: self.execution_error.clone(), + journal_error: self.journal_error.clone(), + } + } + fn remember_journal(&mut self, source: Box) -> Arc { + let error = Arc::new(journal(source)); + self.journal_error.get_or_insert_with(|| error.clone()); + error + } +} diff --git a/crates/cellule-host/src/read_replicas/enrollment/protocol.rs b/crates/cellule-host/src/read_replicas/enrollment/protocol.rs new file mode 100644 index 00000000..d8441804 --- /dev/null +++ b/crates/cellule-host/src/read_replicas/enrollment/protocol.rs @@ -0,0 +1,287 @@ +//! Canonical activation-lane protocol with observable progress across every await. +use super::*; + +impl ReaderEnrollment { + pub(in crate::read_replicas) async fn open( + &self, + manager: &ReadReplicaManager, + source: ReadReplicaSource, + path: PathBuf, + ) -> Result { + let cell = source.description().cell; + let existing = self.records()?.get(&cell).cloned(); + if let Some(record) = existing { + self.publish(&record).await?; + return Err(Error::Control( + "reader enrollment remains unresolved; close before a new opening", + )); + } + if self.records()?.len() >= MAX_READ_VIEWS { + return Err(Error::Capacity("reader enrollment inventory bound")); + } + let spec = self.spec(manager, &source).await?; + let reservation = manager + .runtime + .try_reserve_node_bytes(3 * MAX_RECORD_BYTES as usize)?; + let record = Arc::new(StdMutex::new(Responsibility { + spec, + source: source.clone(), + original: None, + event: None, + published: false, + opening_started: false, + opening_joined: false, + execution_error: None, + journal_error: None, + _retained: reservation, + })); + // All protocol mutations already serialize on the manager's activation + // lane. The index/progress locks serve short reads and writes only, so + // inventory can see the original request during lost/paused RPC replies. + self.records()?.insert(cell, record.clone()); + self.accept(&record).await?; + data(&record)?.opening_started = true; + let result = manager + .open_source_locked(source, path) + .await + .map_err(Arc::new); + { + let mut progress = data(&record)?; + progress.opening_joined = true; + if let Err(error) = &result { + progress + .execution_error + .get_or_insert_with(|| error.clone()); + } + progress.event = Some(match result.as_ref().ok().copied() { + Some(receipt) => { + EnrollmentEvent::Established(evidence(&progress, b"opened", Some(receipt))?) + } + None => { + EnrollmentEvent::Retired(evidence(&progress, b"joined-opening-refusal", None)?) + } + }); + progress.published = false; + } + let publication = self.publish(&record).await; + if result.is_err() && publication.is_ok() { + self.records()?.remove(&cell); + } + match result { + Err(error) => Err(retained(error)), + Ok(receipt) => { + publication?; + Ok(receipt) + } + } + } + + async fn accept(&self, record: &Record) -> Result<()> { + let spec = { + let progress = data(record)?; + if progress.original.is_some() { + return Ok(()); + } + progress.spec.clone() + }; + let acceptance = self.journal.accept_enrollment(&spec, now_ms()?).await; + let acceptance = match acceptance { + Ok(acceptance) => acceptance, + Err(source) => return Err(retained(data(record)?.remember_journal(source))), + }; + match acceptance { + FleetEnrollmentAcceptance::New(original) => { + if original.spec() != &spec || original.status() != EnrollmentStatus::Pending { + return Err(Error::Fenced); + } + original.to_bytes().map_err(operation)?; + data(record)?.original = Some(original); + Ok(()) + } + FleetEnrollmentAcceptance::Existing(original) => { + if original.spec() != &spec { + return Err(Error::Fenced); + } + data(record)?.original = Some(original); + Err(Error::Control( + "reader enrollment acceptance reply is unresolved", + )) + } + } + } + + async fn publish(&self, record: &Record) -> Result<()> { + let (original, event) = { + let progress = data(record)?; + if progress.published { + return Ok(()); + } + let Some(event) = progress.event else { + return Ok(()); + }; + let original = progress + .original + .clone() + .ok_or(Error::Control("reader enrollment acceptance is unknown"))?; + (original, event) + }; + let result = self + .journal + .publish_enrollment_result(&original, event, now_ms()?) + .await; + let result = match result { + Ok(result) => result, + Err(source) => return Err(retained(data(record)?.remember_journal(source))), + }; + result.to_bytes().map_err(operation)?; + let agrees = match event { + EnrollmentEvent::Established(evidence) => { + result.status() == EnrollmentStatus::Established + && result.established_evidence() == Some(evidence) + } + EnrollmentEvent::Retired(evidence) => { + result.status() == EnrollmentStatus::Retired + && result.settlement_evidence() == Some(evidence) + } + EnrollmentEvent::Refused(evidence) => { + result.status() == EnrollmentStatus::Refused + && result.settlement_evidence() == Some(evidence) + } + }; + let mut progress = data(record)?; + if !agrees + || result.spec() != &progress.spec + || result.accepted_at_ms() != original.accepted_at_ms() + || progress.event != Some(event) + { + return Err(Error::Fenced); + } + progress.published = true; + Ok(()) + } + + pub(in crate::read_replicas) async fn established(&self, cell: CellId) -> Result<()> { + let record = self + .records()? + .get(&cell) + .cloned() + .ok_or(Error::Control("reader enrollment owner is absent"))?; + if !matches!(data(&record)?.event, Some(EnrollmentEvent::Established(_))) { + return Err(Error::Control("reader enrollment remains unresolved")); + } + self.publish(&record).await + } + + pub(in crate::read_replicas) async fn retire( + &self, + cell: CellId, + receipt: Option, + ) -> Result<()> { + let record = self.records()?.get(&cell).cloned(); + let Some(record) = record else { + return Ok(()); + }; + let never_started = { + let progress = data(&record)?; + if receipt.is_none() && progress.opening_started && !progress.opening_joined { + return Err(Error::Control( + "reader native opening remains unproven after task failure", + )); + } + !progress.opening_started + }; + if never_started { + if receipt.is_some() { + return Err(Error::Fenced); + } + let (spec, accepted_at, digest) = { + let mut progress = data(&record)?; + let digest = match progress.event { + Some(EnrollmentEvent::Refused(digest)) => digest, + None => { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-reader-unexecuted.v1\0"); + hash.update(&progress.spec.to_bytes().map_err(operation)?); + Digest::from_bytes(*hash.finalize().as_bytes()) + } + _ => return Err(Error::Fenced), + }; + progress.event = Some(EnrollmentEvent::Refused(digest)); + ( + progress.spec.clone(), + progress + .original + .as_ref() + .map(EnrollmentRecord::accepted_at_ms), + digest, + ) + }; + let original = match self + .journal + .refuse_unexecuted_enrollment(&spec, digest, now_ms()?) + .await + { + Ok(original) => original, + Err(source) => return Err(retained(data(&record)?.remember_journal(source))), + }; + original.validate_replay(&spec).map_err(operation)?; + original.to_bytes().map_err(operation)?; + if original.status() != EnrollmentStatus::Refused + || original.settlement_evidence() != Some(digest) + || accepted_at.is_some_and(|time| time != original.accepted_at_ms()) + { + return Err(Error::Fenced); + } + self.records()?.remove(&cell); + return Ok(()); + } + { + let mut progress = data(&record)?; + if !matches!(progress.event, Some(EnrollmentEvent::Retired(_))) { + progress.event = Some(EnrollmentEvent::Retired(evidence( + &progress, + b"joined-closure", + receipt, + )?)); + progress.published = false; + } + } + self.publish(&record).await?; + self.records()?.remove(&cell); + Ok(()) + } +} + +fn evidence(record: &Responsibility, phase: &[u8], receipt: Option) -> Result { + let original = record + .original + .as_ref() + .ok_or(Error::Control("reader acceptance is unknown"))?; + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-reader-evidence.v1\0"); + hash.update(&original.to_bytes().map_err(operation)?); + hash.update(phase); + let description = record.source.description(); + hash.update(description.code.as_bytes()); + hash.update(&description.schema.to_be_bytes()); + hash.update(&(record.source.owner().endpoint.len() as u64).to_be_bytes()); + hash.update(record.source.owner().endpoint.as_bytes()); + if let Some(receipt) = receipt { + let EnrollmentRole::Reader { target, position } = &record.spec.role else { + return Err(Error::Fenced); + }; + if receipt.cell != target.cell_id() + || receipt.incarnation != position.incarnation + || receipt.commit_sequence < position.root.commit_sequence + { + return Err(Error::Fenced); + } + if phase == b"opened" && receipt.commit_sequence != position.root.commit_sequence { + return Err(Error::Fenced); + } + hash.update(receipt.cell.as_bytes()); + hash.update(receipt.incarnation.as_bytes()); + hash.update(&receipt.commit_sequence.to_be_bytes()); + } + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) +} diff --git a/crates/cellule-host/src/read_replicas/inventory/mod.rs b/crates/cellule-host/src/read_replicas/inventory/mod.rs new file mode 100644 index 00000000..03e6bfe5 --- /dev/null +++ b/crates/cellule-host/src/read_replicas/inventory/mod.rs @@ -0,0 +1,204 @@ +//! Node-owned reader diagnostics through the existing activation barrier. + +use super::*; +use cellule_runtime::cell::actor::NodeByteReservation; +use cellule_runtime::client::ReadReplicaLifecycleObservation; +use cellule_runtime::identity::Digest; +use cellule_runtime::node::NodeMode; + +const PAGE_BYTES: usize = 1 << 20; +const MAX_PAGE_ENTRIES: usize = 128; + +/// Opaque continuation for one manager session and captured native reader state. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ReaderInventoryCursor { + topology: Digest, + after: CellId, +} + +impl ReaderInventoryCursor { + /// Encodes a fixed-width application continuation. + #[must_use] + pub fn to_bytes(self) -> [u8; 64] { + let mut bytes = [0; 64]; + bytes[..32].copy_from_slice(self.topology.as_bytes()); + bytes[32..].copy_from_slice(self.after.as_bytes()); + bytes + } + /// Decodes the fixed width; topology is rechecked when the manager scans. + pub fn from_bytes(bytes: &[u8]) -> Result { + let bytes: &[u8; 64] = bytes + .try_into() + .map_err(|_| Error::Node("invalid reader inventory cursor width"))?; + let mut topology = [0; 32]; + topology.copy_from_slice(&bytes[..32]); + let mut after = [0; 32]; + after.copy_from_slice(&bytes[32..]); + Ok(Self { + topology: Digest::from_bytes(topology), + after: CellId::from_bytes(after), + }) + } +} + +/// Bounded managed views, retaining their observation's native-byte admission. +pub struct ReaderInventoryPage { + session: SessionId, + topology: Digest, + mode: NodeMode, + observed_at_ms: i64, + closed: bool, + total_views: usize, + entries: Vec, + next: Option, + _memory: NodeByteReservation, +} + +impl ReaderInventoryPage { + /// Returns the manager's exact node boot session. + #[must_use] + pub const fn session(&self) -> SessionId { + self.session + } + /// Returns the fingerprint of all captured native reader states and admission. + #[must_use] + pub const fn topology(&self) -> Digest { + self.topology + } + /// Returns current shared admission mode, independent of reader count. + #[must_use] + pub const fn mode(&self) -> NodeMode { + self.mode + } + /// Returns the supplied capture time; receipts retain their own positions. + #[must_use] + pub const fn observed_at_ms(&self) -> i64 { + self.observed_at_ms + } + /// Reports terminal manager activation closure. + #[must_use] + pub const fn closed(&self) -> bool { + self.closed + } + /// Counts every managed view, including views outside this page. + #[must_use] + pub const fn total_views(&self) -> usize { + self.total_views + } + /// Returns sorted positions and original native lifetimes, without readiness proof. + #[must_use] + pub fn entries(&self) -> &[ReadReplicaLifecycleObservation] { + &self.entries + } + /// Continues only while the captured manager, admission and native states match. + #[must_use] + pub const fn next(&self) -> Option { + self.next + } +} + +impl ReadReplicaManager { + /// Observes managed reader obligations after the current activation completes. + /// + /// Each page retains one MiB from the existing runtime byte ledger. No remote + /// I/O is performed and no new scheduling task is created. Every bounded manager + /// entry is hashed in place, including rows outside the returned page. A + /// changed position, lifetime, closure or admission requires restarting. + /// Matching fingerprints are interval evidence, not an atomic full scan; + /// open lifetime counts can change and return between captures. Receipts + /// remain advisory local positions: + /// canonical lifetime guards retain accepted query/refresh work, including + /// cancelled native jobs. Local joining fences every retained clone; remote + /// authority, producer retirement, replacement policy and host facilities + /// still require independent settlement before taking a node offline. + pub async fn fleet_readers_page( + &self, + cursor: Option, + limit: usize, + now_ms: i64, + ) -> Result { + if !(1..=MAX_PAGE_ENTRIES).contains(&limit) || now_ms < 0 { + return Err(Error::Node("invalid reader inventory bounds")); + } + let memory = self.runtime.try_reserve_node_bytes(PAGE_BYTES)?; + // Activation and removal already use this lane. Observation cannot miss + // an accepted open that is about to enter the managed collection. + let _activation = self.activation.lock().await; + let active = self.active.read().await; + if active.views.len() > MAX_READ_VIEWS { + return Err(Error::Capacity("node read-view inventory bound")); + } + let mode = self.runtime.node_admission().mode()?; + let closed = self.closed.is_cancelled(); + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.reader-native-inventory.v1\0"); + hash.update(self.session.as_bytes()); + hash.update(active.topology.as_bytes()); + hash.update(&[mode as u8, u8::from(closed)]); + hash.update(&(active.views.len() as u64).to_be_bytes()); + let mut cells = active.views.keys().copied().collect::>(); + cells.sort_unstable_by_key(|cell| *cell.as_bytes()); + let start = match cursor { + None => 0, + Some(cursor) => { + cells + .binary_search_by_key(cursor.after.as_bytes(), |cell| *cell.as_bytes()) + .map_err(|_| Error::Node("reader inventory cursor key is absent"))? + + 1 + } + }; + let end = start.saturating_add(limit).min(cells.len()); + let mut entries = Vec::with_capacity(end - start); + // Peer clones can close or refresh outside the manager activation lane. + // Hash their canonical state once per row, copying that same observation + // into the page. Never certify unreturned rows from manager UUID alone. + for (index, cell) in cells.iter().enumerate() { + let reader = active.views.get(cell).ok_or(Error::Control( + "reader inventory view disappeared under activation lane", + ))?; + let observation = reader.lifecycle_observation().await; + let receipt = observation.receipt(); + hash.update(receipt.cell.as_bytes()); + hash.update(receipt.incarnation.as_bytes()); + hash.update(&receipt.commit_sequence.to_be_bytes()); + hash.update(&[ + u8::from(observation.admission_closed()), + u8::from(observation.snapshot_attached()), + ]); + hash.update(&(observation.retained_lifetimes() as u64).to_be_bytes()); + if (start..end).contains(&index) { + entries.push(observation); + } + } + let topology = Digest::from_bytes(*hash.finalize().as_bytes()); + if self.runtime.node_admission().mode()? != mode || self.closed.is_cancelled() != closed { + return Err(Error::Node( + "reader inventory admission changed; restart scan", + )); + } + if cursor.is_some_and(|cursor| cursor.topology != topology) { + return Err(Error::Node( + "reader inventory topology changed; restart scan", + )); + } + let next = if end < cells.len() { + entries.last().map(|last| ReaderInventoryCursor { + topology, + after: last.receipt().cell, + }) + } else { + None + }; + Ok(ReaderInventoryPage { + session: self.session, + topology, + mode, + observed_at_ms: now_ms, + closed, + total_views: cells.len(), + entries, + next, + _memory: memory, + }) + } +} diff --git a/crates/cellule-host/src/read_replicas/maintenance.rs b/crates/cellule-host/src/read_replicas/maintenance.rs new file mode 100644 index 00000000..4e432c20 --- /dev/null +++ b/crates/cellule-host/src/read_replicas/maintenance.rs @@ -0,0 +1,546 @@ +//! Replacement-policy checks followed by the existing native reader closure. +use super::*; +use crate::fleet::{FleetJournalSnapshot, FleetRoster}; +use cellule_runtime::{ + cell::actor::NodeByteReservation, + client::CellDescription, + control::Control, + fleet::operations::{ + EnrollmentRecord, EnrollmentRole, EnrollmentStatus, MAX_RECORD_BYTES, MaintenanceOperation, + }, + identity::NodeId, + node::{MAX_NODE_BYTES, NodeAdvertisement, NodeMode}, + peer::ReplicaPeerClient, +}; + +/// One fresh, authenticated ready reader outside the maintenance node. +/// Its receipt is an observed prefix, not authority to serve or acquire a Cell. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct ReaderReplacement { + /// Exact physical replacement selected by current canonical policy. + pub node: NodeId, + /// Original probed boot, never replaced by another session in this result. + pub session: SessionId, + /// Hash of the probed signed boot's immutable identity. + pub boot_identity: Digest, + /// Ready native position returned through the existing peer protocol. + pub receipt: Receipt, + /// Exact Established replacement request key. + pub enrollment_key: Digest, + /// Complete original Established row digest, including its first timestamps. + pub enrollment_digest: Digest, +} + +/// Checked per-reader evacuation interval, retaining its native metadata charge. +/// This is not complete fleet role settlement. The controller must persist and +/// revalidate replacements, current authority, all remaining roles and inventory +/// barriers before terminal node shutdown. +pub struct ReaderEvacuation { + snapshot: FleetJournalSnapshot, + operation: MaintenanceOperation, + original: EnrollmentRecord, + retired: EnrollmentRecord, + authority: Control, + policy: Option, + minimum: Receipt, + replacements: Vec, + started_at_ms: i64, + finished_at_ms: i64, + _memory: [NodeByteReservation; 2], +} +impl ReaderEvacuation { + /// Complete final native capture barrier, before evidence publication. + #[must_use] + pub fn snapshot(&self) -> &FleetJournalSnapshot { + &self.snapshot + } + /// Exact operation that authorized this original closure. + #[must_use] + pub fn operation(&self) -> &MaintenanceOperation { + &self.operation + } + /// Builds bounded durable history and every canonical replacement page. + /// Applications account these copied metadata buffers. This conversion + /// supplies no journal commit or fresh post-reconstruction confirmation. + pub fn durable_record( + &self, + ) -> Result<( + cellule_runtime::fleet::operations::ReaderEvacuationRecord, + Vec, + )> { + use cellule_runtime::fleet::operations::{ + ReaderEvacuationRecord, ReaderReplacementWitness, + }; + ReaderEvacuationRecord::new( + self.operation.clone(), + ( + Digest::from_bytes( + *blake3::hash( + &self + .snapshot + .head() + .to_bytes() + .map_err(crate::fleet::operation)?, + ) + .as_bytes(), + ), + self.snapshot.registry(), + ), + self.retired.clone(), + Digest::from_bytes( + *blake3::hash(&self.original.to_bytes().map_err(crate::fleet::operation)?) + .as_bytes(), + ), + self.authority.clone(), + self.policy.map(|policy| policy.revision()), + self.policy.map_or(0, |policy| policy.desired_readers()), + self.minimum.commit_sequence, + self.interval(), + self.replacements + .iter() + .map(|entry| ReaderReplacementWitness { + node: entry.node, + session: entry.session, + boot_identity: entry.boot_identity, + enrollment_key: entry.enrollment_key, + enrollment_digest: entry.enrollment_digest, + commit_sequence: entry.receipt.commit_sequence, + }) + .collect(), + ) + .map_err(crate::fleet::operation) + } + + /// Original immutable Established request supplied to this attempt. + #[must_use] + pub fn original(&self) -> &EnrollmentRecord { + &self.original + } + /// Confirmed retirement of that exact request after canonical local joining. + #[must_use] + pub fn retired(&self) -> &EnrollmentRecord { + &self.retired + } + /// Current Cell authority rechecked after closing, without ownership rights. + #[must_use] + pub fn authority(&self) -> &Control { + &self.authority + } + /// Exact desired-count policy, rechecked after local retirement. + #[must_use] + pub const fn policy(&self) -> Option { + self.policy + } + /// Adequate prefix required from every replacement by this original attempt. + #[must_use] + pub const fn minimum(&self) -> Receipt { + self.minimum + } + /// Fresh ready replacements on distinct physical nodes excluding the donor. + #[must_use] + pub fn replacements(&self) -> &[ReaderReplacement] { + &self.replacements + } + /// Capture interval of this attempt. Replayed retirement retains its own + /// original journal timestamps; fresh replacement probes have this interval. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + (self.started_at_ms, self.finished_at_ms) + } +} + +impl ReadReplicaManager { + /// Evacuates one exact Established managed reader under retained maintenance. + /// + /// Authenticate the caller and use the current Evacuating operation. The + /// normal live-owner recruiter creates replacements. This method probes them + /// through the existing authenticated status protocol; no ready spare means + /// no new local close. Both local and signed boot modes must be Draining. + /// Canonical view closure and producer retirement use the same activation + /// lane as ordinary removal. A deadline/cancelled waiter can leave a joined + /// or fenced original view; retry the same request to resume its ownership. + /// A changed policy, owner, boot or unresolved journal result yields no proof. + pub async fn evacuate( + &self, + original: &EnrollmentRecord, + operation: &MaintenanceOperation, + peer: &ReplicaPeerClient, + deadline: tokio::time::Instant, + ) -> Result { + let deadline = evacuation_deadline(operation.deadline_ms(), deadline)?; + tokio::time::timeout_at( + deadline, + self.evacuate_open(original, operation, peer, deadline), + ) + .await + .map_err(|source| Error::Facility { + name: "reader-evacuation-deadline", + source: Box::new(source), + })? + } + + async fn evacuate_open( + &self, + original: &EnrollmentRecord, + operation: &MaintenanceOperation, + peer: &ReplicaPeerClient, + deadline: tokio::time::Instant, + ) -> Result { + // Reserve before encoding or collecting authority and original-record + // copies. These charges stay with the returned evidence after closure. + let records = self + .runtime + .try_reserve_node_metadata_bytes(4 * MAX_RECORD_BYTES as usize + 4096)?; + original.to_bytes().map_err(crate::fleet::operation)?; + let EnrollmentRole::Reader { target, position } = &original.spec().role else { + return Err(Error::Fenced); + }; + if original.status() != EnrollmentStatus::Established + || original.established_evidence().is_none() + || original.spec().target.session != self.session + || original.spec().target.node != operation.node() + || self.session != operation.session() + || target.application().as_bytes() != self.layout.application_id() + || self.runtime.node_admission().mode()? != NodeMode::Draining + { + return Err(Error::Fenced); + } + let enrollment = self.bound_enrollment()?.ok_or(Error::Control( + "reader evacuation requires managed enrollment", + ))?; + let _activation = self.activation.lock().await; + self.ensure_open()?; + let started_at_ms = now_ms()?; + let roster = enrollment + .maintenance_roster(original, operation, deadline, &self.runtime) + .await?; + let current = self + .authority + .load(target.cell_id()) + .await? + .ok_or(Error::CellNotActive)?; + let authority = current.value().clone(); + let root = authority.root.as_ref().ok_or(Error::CellNotActive)?; + let owner = authority.owner.as_ref().ok_or(Error::CellNotActive)?; + if authority.state != ControlState::Serving + || authority.recovery.is_some() + || authority.incarnation != position.incarnation + { + return Err(Error::Fenced); + } + let local = self + .active + .read() + .await + .views + .get(&target.cell_id()) + .cloned(); + let mut minimum = Receipt { + cell: target.cell_id(), + incarnation: authority.incarnation, + commit_sequence: position.root.commit_sequence.max(root.commit_sequence), + }; + if let Some(reader) = &local { + let observed = reader.lifecycle_observation().await; + let receipt = observed.receipt(); + if receipt.cell != minimum.cell || receipt.incarnation != minimum.incarnation { + return Err(Error::Fenced); + } + minimum.commit_sequence = minimum.commit_sequence.max(receipt.commit_sequence); + } else if roster + .enrollments() + .iter() + .find(|row| row.spec() == original.spec()) + .is_none_or(|row| row.status() != EnrollmentStatus::Retired) + { + return Err(Error::Control( + "reader evacuation lacks original local closure", + )); + } + let policy = self.policy.load(minimum.cell).await?.map(|row| row.value()); + if policy.is_some_and(|policy| policy.incarnation() != minimum.incarnation) { + return Err(Error::Fenced); + } + let desired = usize::from(policy.map_or(0, |policy| policy.desired_readers())); + let memory = self.runtime.try_reserve_node_metadata_bytes( + desired * std::mem::size_of::() + 4096, + )?; + // The canonical node codec bounds retained candidate bodies. Account + // those temporary observations separately from the final fixed rows. + let _candidates = self.runtime.try_reserve_node_metadata_bytes( + (desired + 1) + * (2 * MAX_NODE_BYTES as usize + std::mem::size_of::()) + + 4096, + )?; + let own = self + .directory + .load(self.session, now_ms()?) + .await? + .ok_or(Error::Fenced)?; + let own = own.advertisement(); + if !enrollment.matches_boot(own) + || own + .operational_sample() + .is_none_or(|s| s.mode != NodeMode::Draining) + || roster.boot(own.node(), own.session())?.intent().mode() != NodeMode::Draining + { + return Err(Error::Fenced); + } + let selected = self + .directory + .select_readers( + minimum.cell, + owner.session, + authority.code, + desired, + now_ms()?, + MAX_LIVE_NODES, + ) + .await?; + if selected.len() != desired || selected.iter().any(|ad| ad.node() == operation.node()) { + return Err(Error::Capacity( + "reader maintenance replacements unavailable", + )); + } + let expected = CellDescription { + cell: minimum.cell, + incarnation: minimum.incarnation, + code: authority.code, + schema: authority.schema, + }; + let mut replacements = Vec::with_capacity(desired); + for node in selected { + let (enrollment_key, enrollment_digest) = + validate_replacement(&roster, &node, target, minimum)?; + let (receipt, ready) = peer.status(target, node.clone(), expected).await?; + if !ready + || receipt.cell != minimum.cell + || receipt.incarnation != minimum.incarnation + || receipt.commit_sequence < minimum.commit_sequence + { + return Err(Error::ReplicaUnavailable); + } + replacements.push(ReaderReplacement { + node: node.node(), + session: node.session(), + boot_identity: boot_identity(&node)?, + enrollment_key, + enrollment_digest, + receipt, + }); + } + enrollment + .confirm_maintenance_roster(&roster, deadline) + .await?; + self.recheck_replacements(target, &authority, policy, &replacements) + .await?; + drop(roster); + self.remove_locked(minimum.cell).await?; + if let Some(reader) = &local { + let closed = reader.lifecycle_observation().await; + let receipt = closed.receipt(); + if !closed.locally_joined() + || receipt.cell != minimum.cell + || receipt.incarnation != minimum.incarnation + { + return Err(Error::Fenced); + } + // A retained peer clone may finish a refresh after the first probe. + // Every final replacement must also cover this real closed prefix. + minimum.commit_sequence = minimum.commit_sequence.max(receipt.commit_sequence); + } + let after = enrollment + .maintenance_roster(original, operation, deadline, &self.runtime) + .await?; + let retired = after + .enrollments() + .iter() + .find(|row| row.spec() == original.spec()) + .filter(|row| row.status() == EnrollmentStatus::Retired) + .cloned() + .ok_or(Error::Control( + "reader evacuation retirement is unconfirmed", + ))?; + for replacement in &mut replacements { + let node = self + .directory + .load(replacement.session, now_ms()?) + .await? + .ok_or(Error::Fenced)?; + let node = node.advertisement(); + if node.node() != replacement.node || boot_identity(node)? != replacement.boot_identity + { + return Err(Error::Fenced); + } + if validate_replacement(&after, node, target, minimum)? + != (replacement.enrollment_key, replacement.enrollment_digest) + { + return Err(Error::Fenced); + } + let (receipt, ready) = peer.status(target, node.clone(), expected).await?; + if !ready + || receipt.cell != minimum.cell + || receipt.incarnation != minimum.incarnation + || receipt.commit_sequence < minimum.commit_sequence + { + return Err(Error::ReplicaUnavailable); + } + replacement.receipt = receipt; + } + let current = self + .recheck_replacements(target, &authority, policy, &replacements) + .await?; + enrollment + .confirm_maintenance_roster(&after, deadline) + .await?; + Ok(ReaderEvacuation { + snapshot: after.snapshot().clone(), + operation: operation.clone(), + original: original.clone(), + retired, + authority: current, + policy, + minimum, + replacements, + started_at_ms, + finished_at_ms: now_ms()?, + _memory: [records, memory], + }) + } + + async fn recheck_replacements( + &self, + target: &CellTarget, + before: &Control, + policy: Option, + replacements: &[ReaderReplacement], + ) -> Result { + let current = self + .authority + .load(target.cell_id()) + .await? + .ok_or(Error::Fenced)?; + let current = current.value(); + // Publication may advance while a foreign writer serves traffic. Its + // exact lifetime and nonregressing published prefix must remain bound. + if !same_authority(before, current) + || self.policy.load(current.cell).await?.map(|row| row.value()) != policy + { + return Err(Error::Fenced); + } + let owner = current.owner.as_ref().ok_or(Error::Fenced)?; + let selected = self + .directory + .select_readers( + current.cell, + owner.session, + current.code, + replacements.len(), + now_ms()?, + MAX_LIVE_NODES, + ) + .await?; + if selected.len() != replacements.len() { + return Err(Error::Fenced); + } + for (node, replacement) in selected.iter().zip(replacements) { + let fresh = self + .directory + .load(node.session(), now_ms()?) + .await? + .ok_or(Error::Fenced)?; + let node = fresh.advertisement(); + if node.node() != replacement.node + || node.session() != replacement.session + || boot_identity(node)? != replacement.boot_identity + || !node.accepts_new_roles(now_ms()?) + { + return Err(Error::Fenced); + } + } + Ok(current.clone()) + } +} + +fn evacuation_deadline( + operation_deadline_ms: i64, + requested: tokio::time::Instant, +) -> Result { + let captured = tokio::time::Instant::now(); + let remaining = operation_deadline_ms + .checked_sub(now_ms()?) + .filter(|remaining| *remaining > 0) + .ok_or(Error::Node("reader maintenance operation deadline elapsed"))?; + if captured >= requested { + return Err(Error::Node("reader evacuation deadline elapsed")); + } + let limit = captured + .checked_add(Duration::from_millis(remaining.min(30_000) as u64)) + .ok_or(Error::Node("reader maintenance deadline overflow"))?; + Ok(requested.min(limit)) +} + +pub(crate) fn validate_replacement( + roster: &FleetRoster, + node: &NodeAdvertisement, + target: &CellTarget, + minimum: Receipt, +) -> Result<(Digest, Digest)> { + if node.fleet() != roster.snapshot().head().scope().fleet + || !node.accepts_new_roles(now_ms()?) + || roster.boot(node.node(), node.session())?.intent().mode() != NodeMode::Active + { + return Err(Error::Fenced); + } + let mut matching = roster.enrollments().iter().filter(|row| { + row.status() == EnrollmentStatus::Established + && row.spec().target.node == node.node() + && row.spec().target.session == node.session() + && matches!(&row.spec().role,EnrollmentRole::Reader {target:enrolled,position} + if enrolled==target && position.incarnation==minimum.incarnation) + }); + let original = matching.next().ok_or(Error::Control( + "reader replacement lacks Established enrollment", + ))?; + if matching.next().is_some() { + return Err(Error::Control("reader replacement enrollment is ambiguous")); + } + Ok(( + original.spec().key().map_err(crate::fleet::operation)?, + Digest::from_bytes( + *blake3::hash(&original.to_bytes().map_err(crate::fleet::operation)?).as_bytes(), + ), + )) +} + +pub(crate) fn boot_identity(node: &NodeAdvertisement) -> Result { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.reader-maintenance-boot.v1\0"); + hash.update(node.node().as_bytes()); + hash.update(node.session().as_bytes()); + hash.update(node.fleet().as_bytes()); + hash.update(node.certificate().as_bytes()); + hash.update(node.image().as_bytes()); + hash.update(node.release().as_bytes()); + hash.update(&node.verifying_key()?.to_bytes()); + hash.update(node.endpoint().as_bytes()); + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) +} + +/// Shared reader authority lifetime/prefix check for native closure and durable revalidation. +pub(crate) fn same_authority(before: &Control, current: &Control) -> bool { + current.state == ControlState::Serving + && current.recovery.is_none() + && current.cell == before.cell + && current.incarnation == before.incarnation + && current.epoch == before.epoch + && current.owner == before.owner + && current.code == before.code + && current.schema == before.schema + && current.root.as_ref().is_some_and(|root| { + before + .root + .as_ref() + .is_some_and(|old| root.commit_sequence >= old.commit_sequence) + }) +} diff --git a/crates/cellule-host/src/read_replicas/mod.rs b/crates/cellule-host/src/read_replicas/mod.rs index 6bce2c45..93703617 100644 --- a/crates/cellule-host/src/read_replicas/mod.rs +++ b/crates/cellule-host/src/read_replicas/mod.rs @@ -12,9 +12,9 @@ use std::{ use cellule_runtime::{ Error, Result, cell::actor::CellRuntime, - client::{CellReadReplica, Receipt}, - control::{Control, ControlState, authority::CellAuthority}, - identity::{CellId, CellTarget, IncarnationId, SessionId}, + client::{CellReadReplica, ReadReplicaSource, Receipt}, + control::{ControlState, authority::CellAuthority}, + identity::{CellId, CellTarget, Digest, IncarnationId, SessionId}, ltx::{CellReplica, CellStorageLayout, Limits}, node::NodeDirectory, peer::{PeerReplicaControl, PeerReplicaResolver}, @@ -22,17 +22,45 @@ use cellule_runtime::{ registry::Registry, }; use futures_util::future::BoxFuture; +use futures_util::{StreamExt, stream}; use tokio::sync::{Mutex, RwLock}; use tokio_util::sync::CancellationToken; use uuid::Uuid; +pub(crate) mod enrollment; +mod inventory; +pub(crate) mod maintenance; +use enrollment::{ActivationRequest, ReaderEnrollment}; +pub use enrollment::{ + ReaderEnrollmentCompletion, ReaderEnrollmentInventoryCursor, ReaderEnrollmentInventoryPage, + ReaderEnrollmentJobs, +}; +pub use maintenance::{ReaderEvacuation, ReaderReplacement}; +mod reconciliation; mod recruitment; +pub use inventory::{ReaderInventoryCursor, ReaderInventoryPage}; pub use recruitment::ReadReplicaRecruiter; const RECONCILE_INTERVAL: Duration = Duration::from_secs(5); const RECONCILE_BATCH: usize = 64; const RECONCILE_DEADLINE: Duration = Duration::from_secs(30); const MAX_LIVE_NODES: usize = 10_000; +const MAX_READ_VIEWS: usize = 10_000; +const CLOSE_CONCURRENCY: usize = 16; + +struct ActiveReaders { + views: HashMap, + topology: Uuid, +} + +impl Default for ActiveReaders { + fn default() -> Self { + Self { + views: HashMap::new(), + topology: Uuid::now_v7(), + } + } +} /// Admitted immutable readers sharing one node runtime and current placement policy. /// @@ -51,7 +79,10 @@ pub struct ReadReplicaManager { limits: Limits, closed: CancellationToken, activation: Arc>, - active: Arc>>, + active: Arc>, + enrollment: Arc>>>, + enrollment_required: Arc, + activation_started: Arc, } impl ReadReplicaManager { @@ -82,7 +113,86 @@ impl ReadReplicaManager { limits, closed: CancellationToken::new(), activation: Arc::new(Mutex::new(())), - active: Arc::new(RwLock::new(HashMap::new())), + active: Arc::new(RwLock::new(ActiveReaders::default())), + enrollment: Arc::new(std::sync::Mutex::new(None)), + enrollment_required: Arc::new(std::sync::atomic::AtomicBool::new(false)), + activation_started: Arc::new(std::sync::atomic::AtomicBool::new(false)), + } + } + + pub(crate) fn require_enrollment(&self) { + self.enrollment_required + .store(true, std::sync::atomic::Ordering::Release); + } + + pub(crate) fn bind_enrollment(&self, binding: Arc) -> Result<()> { + let mut current = self + .enrollment + .lock() + .map_err(|_| Error::Control("reader enrollment binding poisoned"))?; + if self + .activation_started + .load(std::sync::atomic::Ordering::Acquire) + { + return Err(Error::Control( + "reader enrollment must precede the first activation", + )); + } + if current.is_some() { + return Err(Error::Control("reader enrollment already installed")); + } + *current = Some(binding); + Ok(()) + } + + fn bound_enrollment(&self) -> Result>> { + Ok(self + .enrollment + .lock() + .map_err(|_| Error::Control("reader enrollment binding poisoned"))? + .clone()) + } + + fn activation_enrollment(&self) -> Result>> { + let binding = self + .enrollment + .lock() + .map_err(|_| Error::Control("reader enrollment binding poisoned"))?; + // Serialize the first activation with binding. A pre-binding caller must + // never enter a cancellable path after durable enrollment is installed. + if binding.is_none() + && self + .enrollment_required + .load(std::sync::atomic::Ordering::Acquire) + { + return Err(Error::Control("fleet reader enrollment is not installed")); + } + self.activation_started + .store(true, std::sync::atomic::Ordering::Release); + Ok(binding.clone()) + } + + fn enrollment(&self) -> Result>> { + let binding = self.bound_enrollment()?; + if binding.is_none() + && self + .enrollment_required + .load(std::sync::atomic::Ordering::Acquire) + { + return Err(Error::Control("fleet reader enrollment is not installed")); + } + Ok(binding) + } + + /// Inspects retained original acceptance, evidence and errors for a local + /// reader obligation. This is diagnostic state, not current serving proof. + pub async fn enrollment_completion( + &self, + cell: CellId, + ) -> Result> { + match self.bound_enrollment()? { + Some(enrollment) => enrollment.completion(cell), + None => Ok(None), } } @@ -157,6 +267,11 @@ impl ReadReplicaManager { /// Admits or refreshes a selected snapshot after an authenticated owner hint. pub async fn activate(&self, target: CellTarget, origin: SessionId) -> Result { + if let Some(enrollment) = self.activation_enrollment()? { + return enrollment + .activate(self.clone(), ActivationRequest::Hint(target, origin)) + .await; + } tokio::select! { () = self.closed.cancelled() => Err(Error::RuntimeClosed), result = self.activate_open(target, origin) => result, @@ -166,59 +281,150 @@ impl ReadReplicaManager { async fn activate_open(&self, target: CellTarget, origin: SessionId) -> Result { let _activation = self.activation.lock().await; self.ensure_open()?; - let cell = target.cell_id(); - let control = self - .authority - .load(cell) - .await? - .ok_or(Error::CellNotActive)?; - let control = control.value(); - let owner = control.owner.as_ref().ok_or(Error::Fenced)?; - if control.state != ControlState::Serving - || control.recovery.is_some() - || owner.session != origin - { + let source = CellReadReplica::prepare_source( + &self.registry, + &self.authority, + &self.directory, + target, + ) + .await?; + if source.owner().session != origin { + return Err(Error::Fenced); + } + self.activate_source_locked(source).await + } + + /// Observes a selected exact source without opening a local reader. + /// + /// A fleet adapter can journal Pending using this value's root, epoch and + /// owner before `activate_source`. Selection and admission are rechecked + /// at activation; preparation itself grants no enrollment permit. + pub async fn prepare_source( + &self, + target: CellTarget, + origin: SessionId, + ) -> Result { + self.ensure_open()?; + self.runtime.node_admission().check_new_role()?; + let source = CellReadReplica::prepare_source( + &self.registry, + &self.authority, + &self.directory, + target, + ) + .await?; + if source.owner().session != origin || !self.selected_source(&source).await? { return Err(Error::Fenced); } - if !self.selected(control, origin).await? { + Ok(source) + } + + /// Initially opens an adapter's journaled exact source through canonical activation. + /// Newer publication cannot replace the supplied root. An installed fleet + /// enrollment binding owns Pending, checked publication and joined retirement + /// across waiter cancellation. Without a binding, the adapter must retain + /// this future after acceptance. Once it enters the lane, manager closure + /// joins opening rather than cancelling it. + pub async fn activate_source(&self, source: ReadReplicaSource) -> Result { + if let Some(enrollment) = self.activation_enrollment()? { + return enrollment + .activate(self.clone(), ActivationRequest::Source(Box::new(source))) + .await; + } + self.activate_initial(source).await + } + + async fn activate_initial(&self, source: ReadReplicaSource) -> Result { + let _activation = tokio::select! { + () = self.closed.cancelled() => return Err(Error::RuntimeClosed), + activation = self.activation.lock() => activation, + }; + self.ensure_open()?; + if self + .active + .read() + .await + .views + .contains_key(&source.description().cell) + { + return Err(Error::Control("read view is already installed")); + } + self.activate_source_locked(source).await + } + + async fn activate_source_locked(&self, source: ReadReplicaSource) -> Result { + self.ensure_open()?; + if !self.selected_source(&source).await? { return Err(Error::Fenced); } + let expected = source.description(); + let cell = expected.cell; let path = self.destination(cell).await?; - let existing = { self.active.read().await.get(&cell).cloned() }; + let existing = { self.active.read().await.views.get(&cell).cloned() }; if let Some(existing) = existing { + if let Some(enrollment) = self.enrollment()? { + enrollment.established(cell).await?; + } match existing.refresh(&path).await { Ok(receipt) if self.still_selected(cell).await? => return Ok(receipt), Ok(_) => { - self.remove_locked(cell).await; + self.remove_locked(cell).await?; return Err(Error::Fenced); } Err(Error::Fenced) => { - self.remove_locked(cell).await; + self.remove_locked(cell).await?; } Err(error) => return Err(error), } } + self.runtime.node_admission().check_new_role()?; + if let Some(enrollment) = self.enrollment()? { + return enrollment.open(self, source, path).await; + } + self.open_source_locked(source, path).await + } + + async fn open_source_locked( + &self, + source: ReadReplicaSource, + path: PathBuf, + ) -> Result { + let expected = source.description(); + let cell = expected.cell; + if self.active.read().await.views.len() >= MAX_READ_VIEWS { + return Err(Error::Capacity("node read-view inventory bound")); + } let replica = CellReplica::new( self.layout.clone(), *cell.as_bytes(), - *control.incarnation.as_bytes(), + *expected.incarnation.as_bytes(), self.limits, )?; - let reader = CellReadReplica::open( + let reader = CellReadReplica::open_source( self.runtime.clone(), Arc::clone(&self.registry), self.authority.clone(), self.directory.clone(), replica, - target, + source, &path, ) .await?; let receipt = reader.receipt().await; - if !self.still_selected(cell).await? { - return Err(Error::Fenced); + let selection = self.still_selected(cell).await; + if self.closed.is_cancelled() || !matches!(selection, Ok(true)) { + // Once native opening succeeded, even a provider failure must join + // that view before publishing closure or returning its error. + reader.close_and_join().await; + return Err(if self.closed.is_cancelled() { + Error::RuntimeClosed + } else { + selection.err().unwrap_or(Error::Fenced) + }); } - self.active.write().await.insert(cell, reader); + let mut active = self.active.write().await; + active.topology = Uuid::now_v7(); + active.views.insert(cell, reader); Ok(receipt) } @@ -241,28 +447,49 @@ impl ReadReplicaManager { Ok(directory.join(format!("{}.sqlite", Uuid::now_v7()))) } - async fn selected(&self, control: &Control, origin: SessionId) -> Result { - let Some(policy) = self.policy.load(control.cell).await? else { + async fn selected_source(&self, source: &ReadReplicaSource) -> Result { + let expected = source.description(); + self.selected( + expected.cell, + expected.incarnation, + expected.code, + source.owner().session, + ) + .await + } + + async fn selected( + &self, + cell: CellId, + incarnation: IncarnationId, + code: Digest, + origin: SessionId, + ) -> Result { + let Some(policy) = self.policy.load(cell).await? else { return Ok(false); }; let policy = policy.value(); - if policy.incarnation() != control.incarnation || policy.desired_readers() == 0 { + if policy.incarnation() != incarnation || policy.desired_readers() == 0 { return Ok(false); } let candidates = self .directory .select_readers( - control.cell, + cell, origin, - control.code, + code, usize::from(policy.desired_readers()), now_ms()?, MAX_LIVE_NODES, ) .await?; - Ok(candidates - .iter() - .any(|candidate| candidate.session() == self.session)) + let enrollment = self.bound_enrollment()?; + Ok(candidates.iter().any(|candidate| { + candidate.session() == self.session + && enrollment + .as_ref() + .is_none_or(|binding| binding.matches_boot(candidate)) + })) } async fn still_selected(&self, cell: CellId) -> Result { @@ -276,76 +503,13 @@ impl ReadReplicaManager { let Some(owner) = control.owner.as_ref() else { return Ok(false); }; - self.selected(control, owner.session).await - } - - /// Refreshes admitted snapshots and evicts readers removed from placement. - /// - /// This loop does not discover new Cells; authenticated owner hints call - /// `activate`. Cancellation interrupts provider waits and bounded batches. - pub async fn run(&self, cancellation: CancellationToken) -> Result<()> { - let mut tick = tokio::time::interval(RECONCILE_INTERVAL); - tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); - let mut cursor = 0_usize; - loop { - tokio::select! { - () = cancellation.cancelled() => return Ok(()), - () = self.closed.cancelled() => return Ok(()), - _ = tick.tick() => {} - } - if self.closed.is_cancelled() { - return Ok(()); - } - let mut readers = self.active.read().await.keys().copied().collect::>(); - readers.sort_by_key(|cell| cell.as_bytes().to_owned()); - let count = readers.len().min(RECONCILE_BATCH); - for _ in 0..count { - let index = cursor % readers.len(); - let cell = readers[index]; - // Advance before I/O so an unavailable Cell cannot starve the - // rest of the bounded batch after cancellation or timeout. - cursor = (index + 1) % readers.len(); - tokio::select! { - () = cancellation.cancelled() => return Ok(()), - () = self.closed.cancelled() => return Ok(()), - result = tokio::time::timeout(RECONCILE_DEADLINE, self.refresh_selected(cell)) => { - match result { - Ok(Ok(())) => {}, - Ok(Err(error)) => tracing::warn!(?cell, error = %error, "read replica refresh failed"), - Err(_) => tracing::warn!(?cell, "read replica refresh deadline exceeded"), - } - } - } - } - } - } - - async fn refresh_selected(&self, cell: CellId) -> Result<()> { - // Share activation's lane so an old view cannot evict a replacement - // installed concurrently for the same Cell after an epoch change. - let _activation = self.activation.lock().await; - self.ensure_open()?; - let current = self.active.read().await.get(&cell).cloned(); - let Some(reader) = current else { - return Ok(()); - }; - if !self.still_selected(cell).await? { - self.remove_locked(cell).await; - return Ok(()); - } - let path = self.destination(cell).await?; - match reader.refresh(&path).await { - Ok(_) => Ok(()), - Err(Error::Fenced) => { - // Keep verified warm bytes after owner death. Queries still - // require a live owner; changed authority evicts the view. - if reader.readiness().await.is_err() { - self.remove_locked(cell).await; - } - Ok(()) - } - Err(error) => Err(error), - } + self.selected( + control.cell, + control.incarnation, + control.code, + owner.session, + ) + .await } fn ensure_open(&self) -> Result<()> { @@ -356,25 +520,93 @@ impl ReadReplicaManager { } /// Closes and removes one read view before eviction or writable activation. - pub async fn remove(&self, cell: CellId) { + pub async fn remove(&self, cell: CellId) -> Result<()> { let _activation = self.activation.lock().await; - self.remove_locked(cell).await; + self.remove_locked(cell).await } - async fn remove_locked(&self, cell: CellId) { - if let Some(reader) = self.active.write().await.remove(&cell) { - reader.close(); + async fn remove_locked(&self, cell: CellId) -> Result<()> { + let reader = self.active.read().await.views.get(&cell).cloned(); + let receipt = match &reader { + Some(reader) => Some(reader.close_and_join().await), + None => None, + }; + if let Some(enrollment) = self.bound_enrollment()? { + enrollment.retire(cell, receipt).await?; + } + if reader.is_some() { + // Retain a fenced view and its enrollment until durable retirement. + // Cancellation or publication failure cannot erase this obligation. + let mut active = self.active.write().await; + active.views.remove(&cell); + active.topology = Uuid::now_v7(); } + Ok(()) } - /// Permanently closes activation and every retained read view. - pub async fn shutdown(&self) { - // Cancel provider waits before joining activation's lane; retained peer - // adapters must neither strand drain nor reopen snapshots afterward. + /// Permanently closes activation, joins owned enrollment jobs and retires views. + /// A journal failure retains fenced inventory for a later shutdown attempt. + pub async fn shutdown(&self) -> Result<()> { self.closed.cancel(); + let enrollment = self.bound_enrollment()?; + let mut failure = None; + if let Some(enrollment) = &enrollment { + enrollment.close_admission()?; + if let Err(error) = enrollment.join().await { + failure = Some(error); + } + } let _activation = self.activation.lock().await; - for (_, reader) in self.active.write().await.drain() { - reader.close(); + let views = { + let mut active = self.active.write().await; + active.topology = Uuid::now_v7(); + for reader in active.views.values() { + reader.close(); + } + active + .views + .iter() + .map(|(cell, reader)| (*cell, reader.clone())) + .collect::>() + }; + let closing = stream::iter(views) + .map(|(cell, reader)| async move { (cell, reader.close_and_join().await) }) + .buffer_unordered(CLOSE_CONCURRENCY); + tokio::pin!(closing); + let mut joined = Vec::new(); + while let Some(item) = closing.next().await { + joined.push(item); + } + // Preserve the complete ownership collection across cancellation while + // any native sibling is still joining, as the ordinary manager does. + for (cell, receipt) in joined { + let retirement = match &enrollment { + Some(enrollment) => enrollment.retire(cell, Some(receipt)).await, + None => Ok(()), + }; + match retirement { + Ok(()) => { + self.active.write().await.views.remove(&cell); + } + Err(error) => { + if failure.is_none() { + failure = Some(error); + } + } + } + } + if let Some(enrollment) = &enrollment { + for cell in enrollment.unresolved_cells()? { + if let Err(error) = enrollment.retire(cell, None).await + && failure.is_none() + { + failure = Some(error); + } + } + } + match failure { + Some(error) => Err(error), + None => Ok(()), } } } @@ -398,6 +630,7 @@ impl PeerReplicaResolver for ReadReplicaManager { .active .read() .await + .views .get(&target.cell_id()) .cloned() .ok_or(Error::ReplicaUnavailable) diff --git a/crates/cellule-host/src/read_replicas/reconciliation.rs b/crates/cellule-host/src/read_replicas/reconciliation.rs new file mode 100644 index 00000000..cc31fa06 --- /dev/null +++ b/crates/cellule-host/src/read_replicas/reconciliation.rs @@ -0,0 +1,147 @@ +//! Periodic repair through the original activation and enrollment owners. +use super::*; +use cellule_runtime::cell::actor::NodeByteReservation; + +struct ReconciliationCells { + cells: Vec, + _memory: NodeByteReservation, +} + +impl ReadReplicaManager { + /// Refreshes admitted snapshots and reconciles retained enrollment results. + /// + /// This loop discovers no new Cells. It scans installed views and original + /// producer requests, including joined attempts that never opened a view. + /// Cancellation interrupts waits; accepted opening and cleanup keep their + /// canonical owners. Application replacement policy remains independent. + pub async fn run(&self, cancellation: CancellationToken) -> Result<()> { + let mut tick = tokio::time::interval(RECONCILE_INTERVAL); + tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); + let mut cursor = 0_usize; + loop { + tokio::select! { + () = cancellation.cancelled() => return Ok(()), + () = self.closed.cancelled() => return Ok(()), + _ = tick.tick() => {} + } + if self.closed.is_cancelled() { + return Ok(()); + } + if let Some(enrollment) = self.bound_enrollment()? + && let Err(error) = enrollment.reap().await + { + // Preserve failed joins in their original bank, while allowing + // independent responsibilities to publish or close normally. + tracing::warn!(error = %error, "read enrollment join remains failed"); + } + let readers = match self.reconciliation_cells().await { + Ok(readers) => readers, + Err(error @ Error::Capacity(_)) => { + tracing::warn!(error = %error, "read reconciliation inventory refused"); + continue; + } + Err(error) => return Err(error), + }; + let count = readers.cells.len().min(RECONCILE_BATCH); + for _ in 0..count { + let index = cursor % readers.cells.len(); + let cell = readers.cells[index]; + // Advance before I/O: a failed journal or unavailable Cell must + // not repeatedly consume the first position of every batch. + cursor = (index + 1) % readers.cells.len(); + tokio::select! { + () = cancellation.cancelled() => return Ok(()), + () = self.closed.cancelled() => return Ok(()), + result = tokio::time::timeout(RECONCILE_DEADLINE, self.refresh_selected(cell)) => { + match result { + Ok(Ok(())) => {}, + Ok(Err(error)) => tracing::warn!(?cell, error = %error, "read replica reconciliation failed"), + Err(_) => tracing::warn!(?cell, "read replica reconciliation deadline exceeded"), + } + } + } + } + } + } + + async fn reconciliation_cells(&self) -> Result { + let active = self.active.read().await; + if active.views.len() > MAX_READ_VIEWS { + return Err(Error::Capacity("read reconciliation view bound")); + } + let (mut cells, memory) = match self.bound_enrollment()? { + Some(enrollment) => { + enrollment.reconciliation_cells(&self.runtime, active.views.keys().copied())? + } + None => { + // This index is metadata, including startup and fenced cleanup. + // It grants no native admission or authority to renew a lease. + let memory = self.runtime.try_reserve_node_metadata_bytes( + active.views.len() * std::mem::size_of::() + 4096, + )?; + (active.views.keys().copied().collect::>(), memory) + } + }; + drop(active); + cells.sort_unstable_by_key(|cell| *cell.as_bytes()); + cells.dedup(); + Ok(ReconciliationCells { + cells, + _memory: memory, + }) + } + + async fn refresh_selected(&self, cell: CellId) -> Result<()> { + // A local record can be reconciled only after its original activation + // releases this lane. No absence scan may race an accepted native open. + let _activation = self.activation.lock().await; + self.ensure_open()?; + let current = self.active.read().await.views.get(&cell).cloned(); + let Some(reader) = current else { + if let Some(enrollment) = self.bound_enrollment()? { + // Canonical retirement distinguishes never-started exclusion + // from joined native refusal. Unjoined task failures stay blocked. + enrollment.retire(cell, None).await?; + } + return Ok(()); + }; + if reader.lifecycle_observation().await.admission_closed() { + // A failed/cancelled removal retained the fenced view. Resume its + // same join and journal event before attempting any remote refresh. + return self.remove_locked(cell).await; + } + if self.bound_enrollment()?.is_some() + && self.runtime.node_admission().mode()? == cellule_runtime::node::NodeMode::Draining + { + // Removing a managed view merely because cordon changed selection + // would bypass maintenance replacement policy. Preserve the original + // owner until explicit evacuation or terminal native shutdown. Repair + // its existing establishment reply without refreshing or opening it. + if let Some(enrollment) = self.bound_enrollment()? { + enrollment.established(cell).await?; + } + return Ok(()); + } + if !self.still_selected(cell).await? { + return self.remove_locked(cell).await; + } + if let Some(enrollment) = self.bound_enrollment()? { + // Republish the original opening proof; never reopen or replace its + // pinned request just because the first result reply was lost. + enrollment.established(cell).await?; + } + let path = self.destination(cell).await?; + match reader.refresh(&path).await { + Ok(_) => Ok(()), + Err(Error::Fenced) => { + // Keep verified warm bytes after owner death. Queries still + // require a live owner; changed authority evicts the view. + if reader.readiness().await.is_err() { + self.remove_locked(cell).await?; + } + Ok(()) + } + Err(error) => Err(error), + } + } +} diff --git a/crates/cellule-host/src/status.rs b/crates/cellule-host/src/status.rs index 229f1858..a0fe9e14 100644 --- a/crates/cellule-host/src/status.rs +++ b/crates/cellule-host/src/status.rs @@ -2,6 +2,33 @@ use super::*; +/// Local lifecycle of the original retained host drain task. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum NodeDrainPhase { + /// The canonical facility/runtime/withdrawal sequence is still running. + Running, + /// The sequence returned; its task epilogue has not been joined yet. + Returned, + /// The original task was joined, successfully or with its original failure. + Joined, +} + +/// Bounded local closing diagnostics, independent of fleet authorization. +/// Neither a joined task nor an empty error establishes role or relocation proof. +#[derive(Clone, Debug)] +pub struct NodeDrainObservation { + /// Monotonic identity of this node's retained closing attempt. + pub serial: u64, + /// Actual return/join progress of that original task. + pub phase: NodeDrainPhase, + /// Original resource-sequence or task result, absent while still running. + pub result: Option>>, + /// Earliest original failure across retained closing attempts. + pub first_failure: Option>, + /// Most recent original failure; successful retries do not erase history. + pub latest_failure: Option>, +} + /// Node lifecycle state visible to readiness and shutdown adapters. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum NodeState { @@ -9,6 +36,9 @@ pub enum NodeState { Starting, /// The node serves Cells and may take new ownership. Ready, + /// A confirmed fleet boot exposes management/recovery while retained + /// cordon/drain intent keeps serving and new role admission closed. + Maintenance, /// Scale-down started: the node still serves but takes no new ownership. ScalingDown, /// The node is releasing its Cells and refuses new work. diff --git a/crates/cellule-host/src/tasks.rs b/crates/cellule-host/src/tasks.rs index 3f899b08..030abe7f 100644 --- a/crates/cellule-host/src/tasks.rs +++ b/crates/cellule-host/src/tasks.rs @@ -26,7 +26,7 @@ impl Drop for AbortOnDrop { pub struct CellNodeTaskGroup { pub(super) cancellation: CancellationToken, pub(super) node_shutdown: CancellationToken, - tasks: Mutex>, + tasks: Mutex>>, pub(super) failed: Arc, pub(super) draining: AtomicBool, } @@ -34,12 +34,41 @@ pub struct CellNodeTaskGroup { #[derive(Clone, Copy, PartialEq, Eq)] enum TaskPhase { Work, + RetainedWork, Lease, } struct NodeTask { phase: TaskPhase, - handle: JoinHandle, + abort: tokio::task::AbortHandle, + supervisor: tokio::task::AbortHandle, + join: tokio::sync::Mutex, +} + +type TaskFailure = Arc; + +enum TaskJoin { + Running(JoinHandle), + Finished(std::result::Result<(), TaskFailure>), +} + +impl NodeTask { + async fn join(&self) -> std::result::Result<(), TaskFailure> { + let mut joining = self.join.lock().await; + if let TaskJoin::Running(handle) = &mut *joining { + let result = match handle.await { + Ok(result) => result.map_err(TaskFailure::from), + Err(source) => Err(Arc::new(source) as TaskFailure), + }; + // Commit the consumed handle without another await. Cancellation + // before this point drops only the waiter and keeps the same join. + *joining = TaskJoin::Finished(result); + } + match &*joining { + TaskJoin::Finished(result) => result.clone(), + TaskJoin::Running(_) => Err(Arc::new(Error::Control("node task join incomplete"))), + } + } } impl Drop for CellNodeTaskGroup { @@ -49,7 +78,8 @@ impl Drop for CellNodeTaskGroup { Err(poisoned) => poisoned.into_inner(), }; for task in tasks.iter() { - task.handle.abort(); + task.abort.abort(); + task.supervisor.abort(); } } } @@ -84,7 +114,7 @@ impl CellNodeTaskGroup { } self.tasks .lock() - .map(|tasks| tasks.iter().all(|task| !task.handle.is_finished())) + .map(|tasks| tasks.iter().all(|task| !task.supervisor.is_finished())) .unwrap_or(false) } @@ -136,6 +166,15 @@ impl CellNodeTaskGroup { self.spawn_task(task, TaskPhase::Work) } + /// Watch an independently retained facility join without aborting its + /// waiter on a deadline. The facility still owns and joins accepted work. + pub(super) fn spawn_retained(&self, task: F) -> cellule_runtime::Result<()> + where + F: Future + Send + 'static, + { + self.spawn_task(task, TaskPhase::RetainedWork) + } + fn spawn_task(&self, task: F, phase: TaskPhase) -> cellule_runtime::Result<()> where F: Future + Send + 'static, @@ -150,8 +189,12 @@ impl CellNodeTaskGroup { return Err(Error::Capacity("CellNode task limit reached")); } let failed = Arc::clone(&self.failed); + let task = tokio::spawn(task); + // Deadline abortion targets the work, not its supervisor. The latter + // stays owned until it joins the work's cancellation and destructor. + let abort = task.abort_handle(); let handle = tokio::spawn(async move { - let mut task = AbortOnDrop::new(tokio::spawn(task)); + let mut task = AbortOnDrop::new(task); match task.join().await { Ok(result) => { if result.is_err() { @@ -165,7 +208,12 @@ impl CellNodeTaskGroup { } } }); - tasks.push(NodeTask { phase, handle }); + tasks.push(Arc::new(NodeTask { + phase, + abort, + supervisor: handle.abort_handle(), + join: tokio::sync::Mutex::new(TaskJoin::Running(handle)), + })); Ok(()) } @@ -175,11 +223,18 @@ impl CellNodeTaskGroup { } /// Cancels admission and joins tasks in reverse registration order. + /// + /// Cancelling the caller leaves the original joins owned by this group. + /// Every subsequent drain preserves original task failures as error sources. pub async fn drain(&self) -> FacilityResult { self.drain_until(None).await } /// Cancels admission and joins tasks until an optional absolute deadline. + /// + /// A deadline aborts unfinished ordinary tasks but retains their handles. + /// A later drain joins those tasks and reports their original cancellation + /// errors. Watchers of retained facility work continue until that join. pub async fn drain_until(&self, deadline: Option) -> FacilityResult { self.cancel_work(); self.node_shutdown.cancel(); @@ -188,35 +243,40 @@ impl CellNodeTaskGroup { async fn join_until(&self, deadline: Option, include_lease: bool) -> FacilityResult { let tasks = match self.tasks.lock() { - Ok(mut tasks) => { - let (joining, retained): (Vec<_>, Vec<_>) = std::mem::take(&mut *tasks) - .into_iter() - .partition(|task| include_lease || task.phase == TaskPhase::Work); - *tasks = retained; - joining.into_iter().map(|task| task.handle).collect() - } + // Keep the handles and settled results in the same bounded bank. + // Multiple drain callers share each join rather than taking it away. + Ok(tasks) => tasks + .iter() + .filter(|task| include_lease || task.phase != TaskPhase::Lease) + .cloned() + .collect::>(), Err(poisoned) => { - for task in poisoned.into_inner().drain(..) { - task.handle.abort(); + for task in poisoned.into_inner().iter() { + if task.phase != TaskPhase::RetainedWork + && (include_lease || task.phase != TaskPhase::Lease) + { + task.abort.abort(); + } } return Err(Box::new(std::io::Error::other( "CellNode task group lock poisoned", ))); } }; - let mut tasks = TaskBatch { - tasks, - abort_on_drop: true, - }; let mut first_error = None; - let mut timed_out = false; - while let Some(index) = tasks.tasks.len().checked_sub(1) { + for task in tasks.iter().rev() { let result = match deadline { Some(deadline) => { - match tokio::time::timeout_at(deadline.into(), &mut tasks.tasks[index]).await { + match tokio::time::timeout_at(deadline.into(), task.join()).await { Ok(result) => result, Err(_) => { - timed_out = true; + // Preserve the existing deadline abort policy, but + // retain ownership until a later caller joins it. + for task in &tasks { + if task.phase != TaskPhase::RetainedWork { + task.abort.abort(); + } + } first_error.get_or_insert_with(|| { Box::new(std::io::Error::new( std::io::ErrorKind::TimedOut, @@ -228,37 +288,30 @@ impl CellNodeTaskGroup { } } } - None => (&mut tasks.tasks[index]).await, + None => task.join().await, }; - tasks.tasks.pop(); - match result { - Ok(Ok(())) => {} - Ok(Err(error)) if first_error.is_none() => first_error = Some(error), - Ok(Err(_)) => {} - Err(error) if first_error.is_none() => { - first_error = Some(Box::new(error) as Box) - } - Err(_) => {} + if let Err(source) = result + && first_error.is_none() + { + first_error = Some(Box::new(RetainedTaskFailure(source)) + as Box); } } - if !timed_out { - tasks.abort_on_drop = false; - } first_error.map_or(Ok(()), Err) } } -struct TaskBatch { - pub(super) tasks: Vec>, - pub(super) abort_on_drop: bool, +#[derive(Debug)] +struct RetainedTaskFailure(TaskFailure); + +impl std::fmt::Display for RetainedTaskFailure { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + self.0.fmt(formatter) + } } -impl Drop for TaskBatch { - fn drop(&mut self) { - if self.abort_on_drop { - for task in &self.tasks { - task.abort(); - } - } +impl std::error::Error for RetainedTaskFailure { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + Some(self.0.as_ref()) } } diff --git a/crates/cellule-host/tests/node.rs b/crates/cellule-host/tests/node.rs index a9a2a568..28f81398 100644 --- a/crates/cellule-host/tests/node.rs +++ b/crates/cellule-host/tests/node.rs @@ -103,7 +103,15 @@ mod node { pub mod builder; pub mod components; + pub mod durability; + pub mod fleet_actions; + pub mod fleet_maintenance; + pub mod fleet_receivers; + pub mod fleet_snapshots; + pub mod inventory; pub mod lifecycle; + pub mod movement; pub mod qualification; + pub mod reader_closure; pub mod tasks; } diff --git a/crates/cellule-host/tests/node/components.rs b/crates/cellule-host/tests/node/components.rs index 242ee3c6..fd2ad0cc 100644 --- a/crates/cellule-host/tests/node/components.rs +++ b/crates/cellule-host/tests/node/components.rs @@ -2,6 +2,139 @@ use super::*; +struct UnmanagedDurability; +impl NodeDurabilityProvider for UnmanagedDurability { + fn rotation_required( + self: Arc, + _live_node_limit: usize, + ) -> Pin> + Send>> { + Box::pin(async { Ok(false) }) + } + + fn recruit( + self: Arc, + _limits: ReplicaLimits, + _bytes: u64, + _live: usize, + ) -> Pin>> + Send>> { + Box::pin(async { panic!("rejected fleet provider started recruitment") }) + } +} + +#[tokio::test] +async fn configured_fleet_rejects_an_unmanaged_follower_provider_before_starting_it() { + use cellule_runtime::fleet::operations::{FleetScope, NodeIntent}; + use cellule_runtime::identity::NodeId; + let session = SessionId::from_bytes([248; 16]); + let scope = FleetScope { + fleet: Digest::from_bytes([249; 32]), + application: ApplicationId::from_bytes([3; 16]), + }; + let node = CellNodeBuilder::new(application()) + .with_session(session) + .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 << 20) + .with_replica_host(ReplicaHost::default()) + .with_fleet_startup_intent( + NodeIntent::initial(scope, NodeId::from_bytes([248; 16]), session).unwrap(), + ) + .build() + .unwrap(); + node.install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + node.install_node_lease_for_startup(NodeLeaseGuard::new(0, 60_000).unwrap()) + .unwrap(); + let error = node + .install_node_durability_provider( + Arc::new(UnmanagedDurability), + NodeDurabilitySupervisorConfig::new( + scope.application, + ReplicaLimits::default(), + 1, + 3, + Duration::from_millis(10), + Duration::from_secs(60), + 100, + ) + .unwrap(), + ) + .unwrap_err(); + assert!(matches!( + error, + Error::Control("configured fleet durability requires managed follower enrollment") + )); + assert!( + node.owned_component::(NODE_DURABILITY_PROVIDER_COMPONENT) + .is_none() + ); + assert!(node.runtime().node_durability().is_none()); + assert_eq!(node.state(), NodeState::Starting); + assert!(node.runtime().node_admission().startup_held().unwrap()); + node.shutdown().await.unwrap(); + assert_eq!(node.stats().retained_bytes(), 0); +} + +#[tokio::test] +async fn reader_reconciliation_before_lease_installation_keeps_its_owner_healthy() { + use cellule_runtime::ltx::CellStorageLayout; + use cellule_runtime::node::NodeDirectory; + use cellule_store::Store; + use object_store::{memory::InMemory, path::Path}; + + let node = CellNodeBuilder::new(application()) + .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 << 20) + .with_replica_host(ReplicaHost::default()) + .with_session(SessionId::from_bytes([246; 16])) + .build() + .unwrap(); + node.install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + let root = tempfile::tempdir().unwrap(); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("reader-prelease"), + [3; 16], + ); + let directory = NodeDirectory::new( + layout.clone(), + Digest::from_bytes([2; 32]), + Digest::from_bytes([4; 32]), + Digest::from_bytes([5; 32]), + ); + let manager = node + .install_read_replicas( + layout, + directory, + root.path().into(), + ReplicaLimits::default(), + ) + .unwrap(); + // Drive the same public loop through its first immediate tick before the + // lease exists. This fixes the ordering that raced host startup in CI. + let cancellation = CancellationToken::new(); + let mut running = Box::pin(manager.run(cancellation.clone())); + let early = tokio::select! { + result = &mut running => Some(result), + () = tokio::time::sleep(Duration::from_millis(20)) => None, + }; + cancellation.cancel(); + let outcome = match early { + Some(result) => result, + None => running.await, + }; + let start = node.install_node_lease(NodeLeaseGuard::new(0, 60_000).unwrap()); + let shutdown = node.shutdown().await; + assert!( + outcome.is_ok(), + "pre-lease reconciliation returned {outcome:?}" + ); + start.unwrap(); + shutdown.unwrap(); + assert_eq!(node.state(), NodeState::Stopped); + assert_eq!(node.stats().retained_bytes(), 0); + assert_eq!(node.stats().worker_jobs(), 0); + assert_eq!(node.stats().file_descriptors(), 0); +} + #[tokio::test] async fn node_retains_typed_components_without_duplicate_names() { let node = CellNodeBuilder::new(application()) @@ -27,6 +160,22 @@ async fn node_retains_typed_components_without_duplicate_names() { node.owned_component::("fixture-component").as_deref(), Some(&7) ); + assert_eq!( + node.try_owned_component::("fixture-component") + .unwrap() + .as_deref(), + Some(&7) + ); + assert!( + node.try_owned_component::("fixture-component") + .is_err() + ); + assert!(node.try_owned_component::("unowned").is_err()); + assert!( + node.try_owned_component::("missing") + .unwrap() + .is_none() + ); assert!( node.install_owned_component("fixture-component", Arc::new(8_u64)) .is_err() diff --git a/crates/cellule-host/tests/node/durability/mod.rs b/crates/cellule-host/tests/node/durability/mod.rs new file mode 100644 index 00000000..83daa4be --- /dev/null +++ b/crates/cellule-host/tests/node/durability/mod.rs @@ -0,0 +1,954 @@ +//! Requested rotation, native follower fences and retained supervisor joins. +mod observation; +use super::fleet_actions::clock; +use super::*; +use bytes::Bytes; +use cellule_runtime::cell::catalog::{CatalogEntry, CellCatalog}; +use cellule_runtime::cell::executor::{HandlerOutcome, MutationIdentity, StoredOutcome}; +use cellule_runtime::control::{Owner, authority::CellAuthority}; +use cellule_runtime::follower::FollowerReceipt; +use cellule_runtime::identity::{ + CellTarget, IncarnationId, NamespaceId, NodeId, RequestId, TenantId, +}; +use cellule_runtime::ltx::{CellReplica, CellStorageLayout}; +use cellule_runtime::node::durability::NodeLogAuthority; +use cellule_runtime::node::log_transport::{ + AppendRequest, LocalFollowerTransport, NodeLogTransport, RetireRequest, SealRequest, + TailRequest, +}; +use cellule_store::Store; +use futures_util::future::BoxFuture; +use object_store::{memory::InMemory, path::Path}; +use std::sync::atomic::AtomicU64; + +#[derive(Default)] +struct Authority { + closes: Mutex>, + coverage: Mutex>, + lose_cleanup_reply: AtomicBool, + close_attempts: Mutex>, +} +impl NodeLogAuthority for Authority { + fn activate<'a>(&'a self, _epoch: u64) -> BoxFuture<'a, cellule_runtime::Result<()>> { + Box::pin(async { Ok(()) }) + } + fn advance_coverage<'a>( + &'a self, + epoch: u64, + through: u64, + ) -> BoxFuture<'a, cellule_runtime::Result<()>> { + Box::pin(async move { + self.coverage.lock().unwrap().push((epoch, through)); + Ok(()) + }) + } + fn close<'a>( + &'a self, + retirement: &'a cellule_runtime::node::log::NodeLogRetirementObservation, + ) -> BoxFuture<'a, cellule_runtime::Result<()>> { + let barrier = retirement.barrier(); + Box::pin(async move { + let epoch = barrier.log_epoch(); + self.close_attempts.lock().unwrap().push(epoch); + { + let mut closes = self.closes.lock().unwrap(); + if !closes.contains(&epoch) { + closes.push(epoch); + } + } + if epoch == 2 && self.lose_cleanup_reply.load(Ordering::Acquire) { + return Err(Error::Facility { + name: "test-authority-close", + source: Box::new(std::io::Error::new( + std::io::ErrorKind::ConnectionReset, + "accepted cleanup close reply lost", + )), + }); + } + Ok(()) + }) + } +} + +struct Transport { + locals: Vec<(NodeId, LocalFollowerTransport)>, + requests: Mutex>, + frames: Mutex>, + lose_retire: AtomicBool, + pause_retire: AtomicBool, + entered: tokio::sync::Semaphore, + resume: tokio::sync::Semaphore, +} +impl Transport { + fn local(&self, member: NodeId) -> &LocalFollowerTransport { + &self.locals.iter().find(|(id, _)| *id == member).unwrap().1 + } +} +impl NodeLogTransport for Transport { + fn append<'a>( + &'a self, + member: NodeId, + request: AppendRequest, + ) -> BoxFuture<'a, cellule_runtime::Result> { + self.frames.lock().unwrap().extend(request.frames.clone()); + self.local(member).append(member, request) + } + fn seal<'a>( + &'a self, + member: NodeId, + request: SealRequest, + ) -> BoxFuture<'a, cellule_runtime::Result> { + self.local(member).seal(member, request) + } + fn tail<'a>( + &'a self, + member: NodeId, + request: TailRequest, + ) -> BoxFuture<'a, cellule_runtime::Result>> { + self.local(member).tail(member, request) + } + fn retire<'a>( + &'a self, + member: NodeId, + request: RetireRequest, + ) -> BoxFuture<'a, cellule_runtime::Result> { + Box::pin(async move { + self.requests.lock().unwrap().push((member, request)); + let receipt = self.local(member).retire(member, request).await?; + if member == self.locals[0].0 { + if self.pause_retire.swap(false, Ordering::AcqRel) { + self.entered.add_permits(1); + self.resume.acquire().await.unwrap().forget(); + } + if self.lose_retire.load(Ordering::Acquire) { + return Err(Error::Facility { + name: "test-retirement", + source: Box::new(std::io::Error::new( + std::io::ErrorKind::ConnectionReset, + "native retire response lost", + )), + }); + } + } + Ok(receipt) + }) + } +} + +struct Provider { + session: SessionId, + lease: NodeLeaseGuard, + transport: Arc, + authority: Arc, + next_epoch: AtomicU64, + recruited: Mutex)>>, + events: Mutex>, + fail_recruit: AtomicBool, + pause_recruit: AtomicBool, + entered: tokio::sync::Semaphore, + resume: tokio::sync::Semaphore, + abandoned: Arc, + in_flight: AtomicBool, + wrong_boot: AtomicBool, + wrong_node: AtomicBool, + non_advancing: AtomicBool, +} +impl NodeDurabilityProvider for Provider { + fn rotation_required( + self: Arc, + _live_node_limit: usize, + ) -> Pin> + Send>> { + Box::pin(async { Ok(false) }) + } + + fn recruit( + self: Arc, + limits: ReplicaLimits, + _bytes: u64, + _live: usize, + ) -> Pin>> + Send>> { + Box::pin(async move { + if self.pause_recruit.swap(false, Ordering::AcqRel) { + self.in_flight.store(true, Ordering::Release); + struct AcceptedRecruit { + abandoned: Arc, + finished: bool, + } + impl Drop for AcceptedRecruit { + fn drop(&mut self) { + if !self.finished { + self.abandoned.store(true, Ordering::Release); + } + } + } + let mut accepted = AcceptedRecruit { + abandoned: self.abandoned.clone(), + finished: false, + }; + self.entered.add_permits(1); + self.resume.acquire().await.unwrap().forget(); + accepted.finished = true; + self.in_flight.store(false, Ordering::Release); + } + if self.fail_recruit.swap(false, Ordering::AcqRel) { + return Err(Box::new(std::io::Error::new( + std::io::ErrorKind::ConnectionReset, + "replacement recruitment unavailable", + )) + as Box); + } + let epoch = self.next_epoch.fetch_add(1, Ordering::AcqRel); + let session = if self.wrong_boot.swap(false, Ordering::AcqRel) { + SessionId::from_bytes([238; 16]) + } else { + self.session + }; + let node = if self.wrong_node.swap(false, Ordering::AcqRel) { + NodeId::from_bytes([237; 16]) + } else { + NodeId::from_bytes([240; 16]) + }; + let configured_epoch = if self.non_advancing.swap(false, Ordering::AcqRel) { + 1 + } else { + epoch + }; + let members = if epoch == 1 { + vec![self.transport.locals[0].0, self.transport.locals[1].0] + } else { + vec![self.transport.locals[1].0, self.transport.locals[2].0] + }; + self.recruited + .lock() + .unwrap() + .push((epoch, members.clone())); + Ok(Some(NodeDurabilityConfig::new( + session, + node, + configured_epoch, + members, + self.transport.clone(), + self.authority.clone(), + self.lease.clone(), + limits, + Default::default(), + )?)) + }) + } + fn rotation_event(&self, event: NodeDurabilityRotation) { + self.events.lock().unwrap().push(event); + } +} + +struct Fixture { + node: Arc, + provider: Arc, + directories: Vec, + source: tempfile::TempDir, + target: CellTarget, + replica: CellReplica, + authority: CellAuthority, + handle: cellule_runtime::cell::actor::CellHandle, + withdrawn: Arc, +} +struct DrainBarrier { + entered: tokio::sync::Semaphore, + resume: tokio::sync::Semaphore, +} +impl Fixture { + async fn new() -> Self { + Self::with_threshold(u64::MAX).await + } + async fn with_threshold(max_frames: u64) -> Self { + Self::with_drain_barrier(max_frames, None).await + } + async fn with_drain_barrier(max_frames: u64, barrier: Option>) -> Self { + let session = SessionId::from_bytes([239; 16]); + let node = Arc::new( + CellNodeBuilder::new(application()) + .with_runtime( + SqlWorkerPool::new(1, 8) + .unwrap() + .with_native_memory_limit(128 << 20) + .unwrap(), + 64 << 20, + ) + .with_replica_host( + ReplicaHost::default().with_local_disk_budget(DiskBudget::new(8 << 30)), + ) + .with_session(session) + .build() + .unwrap(), + ); + let shutdown = CancellationToken::new(); + let tasks = node + .install_task_group(CancellationToken::new(), shutdown.clone()) + .unwrap(); + let now = clock(); + let lease = NodeLeaseGuard::new(now, now + 60_000).unwrap(); + node.install_node_lease_for_startup(lease.clone()).unwrap(); + let members = [ + NodeId::from_bytes([241; 16]), + NodeId::from_bytes([242; 16]), + NodeId::from_bytes([243; 16]), + ]; + let directories: Vec<_> = members + .iter() + .map(|_| tempfile::tempdir().unwrap()) + .collect(); + let locals = members + .iter() + .zip(&directories) + .map(|(member, directory)| { + let store = FollowerStore::open( + directory.path().to_owned(), + ReplicaLimits::default(), + DiskBudget::new(1 << 30), + ) + .unwrap(); + (*member, LocalFollowerTransport::new(*member, store)) + }) + .collect(); + let transport = Arc::new(Transport { + locals, + requests: Mutex::new(Vec::new()), + frames: Mutex::new(Vec::new()), + lose_retire: AtomicBool::new(false), + pause_retire: AtomicBool::new(false), + entered: tokio::sync::Semaphore::new(0), + resume: tokio::sync::Semaphore::new(0), + }); + let provider = Arc::new(Provider { + session, + lease, + transport, + authority: Arc::new(Authority::default()), + next_epoch: AtomicU64::new(1), + recruited: Mutex::new(Vec::new()), + events: Mutex::new(Vec::new()), + fail_recruit: AtomicBool::new(false), + pause_recruit: AtomicBool::new(false), + entered: tokio::sync::Semaphore::new(0), + resume: tokio::sync::Semaphore::new(0), + abandoned: Arc::new(AtomicBool::new(false)), + in_flight: AtomicBool::new(false), + wrong_boot: AtomicBool::new(false), + wrong_node: AtomicBool::new(false), + non_advancing: AtomicBool::new(false), + }); + let withdrawn = Arc::new(AtomicBool::new(false)); + let observed_withdrawal = withdrawn.clone(); + let observed_provider = provider.clone(); + tasks + .spawn_lease_maintenance(async move { + shutdown.cancelled().await; + observed_withdrawal.store(true, Ordering::Release); + if observed_provider.in_flight.load(Ordering::Acquire) { + return Err(Error::Control( + "session withdrew during accepted recruitment", + )); + } + Ok(()) + }) + .unwrap(); + let limits = ReplicaLimits { + max_database_bytes: 64 << 20, + max_capture_bytes: 16 << 20, + ..ReplicaLimits::default() + }; + if let Some(barrier) = barrier { + // Reverse facility order puts this gate after the original native + // supervisor join, before runtime closure clears weak receipts. + node.install_facility( + CellNodeFacility::new("test-native-join-barrier", move || { + let barrier = Arc::clone(&barrier); + async move { + barrier.entered.add_permits(1); + barrier.resume.acquire().await.unwrap().forget(); + Ok(()) + } + }) + .unwrap(), + ) + .unwrap(); + } + node.install_node_durability_provider( + provider.clone(), + NodeDurabilitySupervisorConfig::new( + ApplicationId::from_bytes([3; 16]), + limits, + 1, + 3, + Duration::from_millis(10), + Duration::from_millis(5), + max_frames, + ) + .unwrap(), + ) + .unwrap(); + until(|| node.runtime().node_durability().is_some()).await; + node.start().unwrap(); + let source = tempfile::tempdir().unwrap(); + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + ApplicationId::from_bytes([3; 16]), + NamespaceId::from_bytes([2; 16]), + b"requested-rotation", + ) + .unwrap(); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("requested-rotation"), + [3; 16], + ); + let incarnation = IncarnationId::from_bytes([244; 16]); + let replica = CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + limits, + ) + .unwrap(); + let catalog = CellCatalog::new(layout.clone(), target.tenant()); + let code = node.application().registry().module_digests()[0]; + let proof = catalog + .provision(CatalogEntry::new(&target, CatalogRole::Sql, code, 1).unwrap()) + .await + .unwrap(); + let authority = CellAuthority::new(layout); + let initial = authority + .create_initial( + &proof, + incarnation, + Owner { + session, + endpoint: "https://rotation.internal:8789".into(), + }, + ) + .await + .unwrap(); + let handle = node + .runtime() + .bootstrap( + proof, + replica.clone(), + authority.clone(), + initial, + source.path().join("cell.sqlite"), + |transaction| { + transaction.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES (16)", + )?; + Ok(()) + }, + ) + .await + .unwrap(); + Self { + node, + provider, + directories, + source, + target, + replica, + authority, + handle, + withdrawn, + } + } + async fn command(&self, id: u8, expected: u8) -> StoredOutcome { + let now = clock(); + let result = self + .handle + .execute( + MutationIdentity { + request_id: RequestId::from_bytes([id; 16]), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }, + Digest::from_bytes([id; 32]), + now, + 64, + 64, + move |transaction| { + transaction.execute("UPDATE counter SET value = value + 1", [])?; + Ok(HandlerOutcome::Success(vec![expected])) + }, + ) + .await + .unwrap(); + assert!(matches!(&result, StoredOutcome::Success { result, .. } if result == &[expected])); + result + } + async fn readback(&self, expected: u8) { + let control = self + .authority + .load(self.target.cell_id()) + .await + .unwrap() + .unwrap(); + let root = control.value().ltx_root().unwrap(); + let destination = self.source.path().join("restored.sqlite"); + let verified = self.replica.open_root(&root).await.unwrap(); + assert_eq!(verified.restore(&destination).await.unwrap(), root.position); + let database = rusqlite::Connection::open(destination).unwrap(); + assert_eq!( + database + .query_row("SELECT value FROM counter", [], |row| row.get::<_, u8>(0)) + .unwrap(), + expected + ); + assert_eq!( + database + .query_row( + "SELECT result FROM sys_requests WHERE request_id = ?1", + [RequestId::from_bytes([245; 16]).as_bytes().as_slice()], + |row| row.get::<_, Vec>(0) + ) + .unwrap(), + vec![17] + ); + } +} + +async fn until(mut ready: impl FnMut() -> bool) { + tokio::time::timeout(Duration::from_secs(5), async { + while !ready() { + tokio::time::sleep(Duration::from_millis(1)).await; + } + }) + .await + .unwrap(); +} +async fn entered(signal: &tokio::sync::Semaphore) { + tokio::time::timeout(Duration::from_secs(5), signal.acquire()) + .await + .unwrap() + .unwrap() + .forget(); +} +fn error_contains(mut error: &(dyn std::error::Error + 'static), text: &str) -> bool { + loop { + if error.to_string().contains(text) { + return true; + } + match error.source() { + Some(source) => error = source, + None => return false, + } + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn requested_rotation_keeps_strict_scope_through_retries_and_continued_writes() { + let test = Fixture::new().await; + let first = test.command(245, 17).await; + test.provider + .transport + .lose_retire + .store(true, Ordering::Release); + assert!(matches!( + test.node.request_node_log_rotation(999), + Err(Error::Control("node-log rotation request has stale scope")) + )); + let request = test.node.request_node_log_rotation(1).unwrap(); + until(|| request.observe().unwrap().first_failure().is_some()).await; + until(|| test.provider.transport.requests.lock().unwrap().len() >= 4).await; + assert!(test.provider.authority.closes.lock().unwrap().is_empty()); + assert_eq!( + request.observe().unwrap().phase(), + NodeLogRotationPhase::Retiring + ); + assert!(request.observe().unwrap().completion().is_none()); + assert_eq!( + test.node.request_node_log_rotation(1).unwrap().log_epoch(), + 1 + ); + let second = test.command(246, 18).await; + assert_eq!(second.commit_sequence(), first.commit_sequence() + 1); + test.provider.fail_recruit.store(true, Ordering::Release); + test.provider + .transport + .lose_retire + .store(false, Ordering::Release); + until(|| request.observe().unwrap().completion().is_some()).await; + let observation = request.observe().unwrap(); + assert!(error_contains( + observation.first_failure().unwrap().as_ref(), + "native retire response lost" + )); + assert!(error_contains( + observation.latest_failure().unwrap().as_ref(), + "replacement recruitment unavailable" + )); + let completion = observation.completion().unwrap(); + assert_eq!(completion.retirement().barrier().log_epoch(), 1); + assert_eq!(completion.retirement().barrier().covered_through(), 1); + assert_eq!(completion.replacement_epoch(), 2); + assert!(Arc::ptr_eq( + completion, + test.node + .request_node_log_rotation(1) + .unwrap() + .observe() + .unwrap() + .completion() + .unwrap() + )); + assert_eq!( + test.provider.recruited.lock().unwrap()[1].1, + vec![NodeId::from_bytes([242; 16]), NodeId::from_bytes([243; 16])] + ); + test.command(247, 19).await; + test.handle.drain().await.unwrap(); + for directory in &test.directories[..2] { + let store = FollowerStore::open( + directory.path().to_owned(), + ReplicaLimits::default(), + DiskBudget::new(1 << 30), + ) + .unwrap(); + let page = store.fleet_lanes_page(None, 128, clock()).await.unwrap(); + let old = page + .entries() + .iter() + .find(|entry| entry.epoch == 1) + .unwrap(); + assert_eq!( + old.state, + cellule_runtime::follower::FollowerLaneState::Retired + ); + assert_eq!(old.retired_through, Some(1)); + let frame = test.provider.transport.frames.lock().unwrap()[0].clone(); + assert!( + store + .append(test.provider.session, 1, vec![frame], 0) + .await + .is_err() + ); + } + test.readback(19).await; + test.node.shutdown().await.unwrap(); + assert_eq!(test.node.stats().retained_bytes(), 0); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn dropped_request_handle_does_not_cancel_rotation() { + let test = Fixture::new().await; + test.command(245, 17).await; + test.handle.drain().await.unwrap(); + test.provider + .transport + .pause_retire + .store(true, Ordering::Release); + let request = test.node.request_node_log_rotation(1).unwrap(); + entered(&test.provider.transport.entered).await; + drop(request); + let recovered = test.node.node_log_rotation_request(1).unwrap().unwrap(); + assert!(recovered.observe().unwrap().completion().is_none()); + test.provider.transport.resume.add_permits(1); + until(|| recovered.observe().unwrap().completion().is_some()).await; + assert_eq!(test.provider.recruited.lock().unwrap().len(), 2); + assert_eq!(*test.provider.authority.closes.lock().unwrap(), vec![1]); + test.readback(17).await; + test.node.shutdown().await.unwrap(); + assert_eq!(test.node.stats().retained_bytes(), 0); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn host_deadline_retains_accepted_recruitment_until_native_cleanup_joins() { + let test = Fixture::new().await; + test.command(245, 17).await; + test.handle.drain().await.unwrap(); + test.provider.pause_recruit.store(true, Ordering::Release); + let request = test.node.request_node_log_rotation(1).unwrap(); + entered(&test.provider.entered).await; + assert_eq!( + request.observe().unwrap().phase(), + NodeLogRotationPhase::Recruiting + ); + let result = test + .node + .shutdown_until(Instant::now() + Duration::from_millis(20)) + .await; + assert!(result.is_err()); + assert_eq!(test.node.state(), NodeState::Draining); + assert!(!test.provider.abandoned.load(Ordering::Acquire)); + assert!(!test.withdrawn.load(Ordering::Acquire)); + assert!(request.observe().unwrap().completion().is_none()); + assert_eq!(test.provider.recruited.lock().unwrap().len(), 1); + until(|| test.node.runtime().is_shutting_down()).await; + assert!(matches!( + test.node.runtime().try_reserve_node_bytes(1), + Err(Error::RuntimeClosed) + )); + let retained = test.node.stats().retained_bytes(); + let supervisor = test + .node + .fleet_durability_supervisor(clock()) + .unwrap() + .unwrap(); + assert_eq!(supervisor.state, NodeDurabilitySupervisorState::Running); + assert!(supervisor.cancellation_requested); + let inventory = supervisor.rotations.unwrap(); + assert!(!inventory.stopped); + assert_eq!(inventory.running_epoch, Some(1)); + let pending = inventory.pending.unwrap(); + assert_eq!(pending.epoch, 1); + assert_eq!(pending.progress.phase(), NodeLogRotationPhase::Recruiting); + assert!(Arc::ptr_eq( + pending.progress.retirement().unwrap(), + request.observe().unwrap().retirement().unwrap() + )); + assert_eq!(test.node.stats().retained_bytes(), retained); + test.provider.resume.add_permits(1); + test.node.shutdown().await.unwrap(); + assert!(!test.provider.abandoned.load(Ordering::Acquire)); + assert_eq!(test.node.state(), NodeState::Stopped); + assert!(test.withdrawn.load(Ordering::Acquire)); + assert_eq!(*test.provider.authority.closes.lock().unwrap(), vec![1, 2]); + assert_eq!(test.node.stats().retained_bytes(), 0); + assert!(request.observe().is_err()); + assert!(test.node.node_log_rotation_request(1).unwrap().is_none()); + test.readback(17).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn cancelled_host_drain_waiter_retains_native_retirement_and_original_request() { + let barrier = Arc::new(DrainBarrier { + entered: tokio::sync::Semaphore::new(0), + resume: tokio::sync::Semaphore::new(0), + }); + let test = Fixture::with_drain_barrier(u64::MAX, Some(Arc::clone(&barrier))).await; + test.command(245, 17).await; + test.handle.drain().await.unwrap(); + test.provider + .transport + .pause_retire + .store(true, Ordering::Release); + let request = test.node.request_node_log_rotation(1).unwrap(); + entered(&test.provider.transport.entered).await; + let node = test.node.clone(); + let waiter = tokio::spawn(async move { node.shutdown().await }); + until(|| test.node.state() == NodeState::Draining).await; + waiter.abort(); + assert!(waiter.await.unwrap_err().is_cancelled()); + assert_eq!( + request.observe().unwrap().phase(), + NodeLogRotationPhase::Retiring + ); + assert!(!test.withdrawn.load(Ordering::Acquire)); + assert!(request.observe().unwrap().completion().is_none()); + test.provider.transport.resume.add_permits(1); + entered(&barrier.entered).await; + assert_eq!( + request.observe().unwrap().phase(), + NodeLogRotationPhase::Interrupted + ); + let interrupted = test + .node + .node_log_rotation_request(1) + .unwrap() + .unwrap() + .observe() + .unwrap(); + assert!(interrupted.retirement().is_some()); + assert!(interrupted.completion().is_none()); + assert_eq!(test.provider.recruited.lock().unwrap().len(), 1); + assert!(!test.withdrawn.load(Ordering::Acquire)); + let closing = test.node.drain_observation().unwrap().unwrap(); + assert_eq!(closing.serial, 1); + assert_eq!(closing.phase, NodeDrainPhase::Running); + assert!(closing.result.is_none()); + barrier.resume.add_permits(1); + test.node.shutdown().await.unwrap(); + let joined = test.node.drain_observation().unwrap().unwrap(); + assert_eq!(joined.serial, 1); + assert_eq!(joined.phase, NodeDrainPhase::Joined); + assert!(joined.result.unwrap().is_ok()); + assert!(request.observe().is_err()); + assert_eq!(test.node.state(), NodeState::Stopped); + assert_eq!(*test.provider.authority.closes.lock().unwrap(), vec![1]); + assert!(test.withdrawn.load(Ordering::Acquire)); + assert_eq!(test.node.stats().retained_bytes(), 0); + test.readback(17).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn automatic_rotation_in_flight_refuses_retroactive_strict_request() { + let test = Fixture::with_threshold(1).await; + test.provider + .transport + .pause_retire + .store(true, Ordering::Release); + test.command(245, 17).await; + entered(&test.provider.transport.entered).await; + let result = test.node.request_node_log_rotation(1); + let retained = test.node.node_log_rotation_request(1).unwrap(); + test.provider.transport.resume.add_permits(1); + until(|| { + test.node + .runtime() + .node_durability() + .unwrap() + .1 + .log_epoch() + .unwrap() + == 2 + }) + .await; + assert!(matches!( + result, + Err(Error::Control( + "node-log automatic rotation already running" + )) + )); + assert!(retained.is_none()); + test.handle.drain().await.unwrap(); + test.readback(17).await; + test.node.shutdown().await.unwrap(); + assert_eq!(test.node.stats().retained_bytes(), 0); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn rotation_history_is_bounded_and_weak_handles_do_not_hold_runtime_resources() { + let test = Fixture::new().await; + test.command(245, 17).await; + test.handle.drain().await.unwrap(); + let first = test.node.request_node_log_rotation(1).unwrap(); + until(|| first.observe().unwrap().completion().is_some()).await; + let completed_bytes = test.node.stats().retained_bytes(); + test.provider.pause_recruit.store(true, Ordering::Release); + let second = test.node.request_node_log_rotation(2).unwrap(); + entered(&test.provider.entered).await; + assert_eq!( + test.node.stats().retained_bytes(), + completed_bytes + 4 * 1024 + ); + assert!(matches!( + test.node.request_node_log_rotation(3), + Err(Error::Capacity("node-log rotation request already pending")) + )); + assert!(first.observe().unwrap().completion().is_some()); + test.provider.resume.add_permits(1); + until(|| second.observe().unwrap().completion().is_some()).await; + assert!(first.observe().is_err()); + assert!(test.node.node_log_rotation_request(1).unwrap().is_none()); + assert_eq!(test.node.stats().retained_bytes(), completed_bytes); + test.node.shutdown().await.unwrap(); + assert!(second.observe().is_err()); + assert_eq!(test.node.stats().retained_bytes(), 0); + test.readback(17).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn invalid_replacement_boot_node_or_epoch_never_builds_or_closes_foreign_scope() { + for bad in 0..3 { + let test = Fixture::new().await; + test.command(245, 17).await; + test.handle.drain().await.unwrap(); + match bad { + 0 => test.provider.wrong_boot.store(true, Ordering::Release), + 1 => test.provider.wrong_node.store(true, Ordering::Release), + _ => test.provider.non_advancing.store(true, Ordering::Release), + } + let request = test.node.request_node_log_rotation(1).unwrap(); + until(|| request.observe().unwrap().completion().is_some()).await; + let observed = request.observe().unwrap(); + assert!(error_contains( + observed.first_failure().unwrap().as_ref(), + "node-log replacement identity or epoch differs" + )); + assert_eq!(observed.completion().unwrap().replacement_epoch(), 3); + assert_eq!( + test.node + .runtime() + .node_durability() + .unwrap() + .1 + .identity() + .unwrap(), + (test.provider.session, NodeId::from_bytes([240; 16]), 3) + ); + assert_eq!(*test.provider.authority.closes.lock().unwrap(), vec![1]); + assert!( + test.provider + .transport + .requests + .lock() + .unwrap() + .iter() + .all( + |(_, request)| request.leader_session == test.provider.session + && request.log_epoch == 1 + ) + ); + test.readback(17).await; + test.node.shutdown().await.unwrap(); + assert_eq!(*test.provider.authority.closes.lock().unwrap(), vec![1, 3]); + assert_eq!(test.node.stats().retained_bytes(), 0); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn lost_cleanup_close_replies_keep_the_same_generation_owned_across_deadlines() { + let test = Fixture::new().await; + test.command(245, 17).await; + test.handle.drain().await.unwrap(); + test.provider.pause_recruit.store(true, Ordering::Release); + test.provider + .authority + .lose_cleanup_reply + .store(true, Ordering::Release); + let request = test.node.request_node_log_rotation(1).unwrap(); + entered(&test.provider.entered).await; + assert!( + test.node + .shutdown_until(Instant::now() + Duration::from_millis(20)) + .await + .is_err() + ); + test.provider.resume.add_permits(1); + until(|| request.observe().unwrap().first_failure().is_some()).await; + until(|| { + test.provider + .authority + .close_attempts + .lock() + .unwrap() + .iter() + .filter(|epoch| **epoch == 2) + .count() + >= 2 + }) + .await; + let observed = request.observe().unwrap(); + assert_eq!(observed.phase(), NodeLogRotationPhase::Recruiting); + assert!(observed.completion().is_none()); + assert!(error_contains( + observed.first_failure().unwrap().as_ref(), + "accepted cleanup close reply lost" + )); + assert!( + test.node + .shutdown_until(Instant::now() + Duration::from_millis(20)) + .await + .is_err() + ); + assert_eq!(test.node.state(), NodeState::Draining); + assert!(!test.withdrawn.load(Ordering::Acquire)); + assert_eq!(test.provider.recruited.lock().unwrap().len(), 2); + test.provider + .authority + .lose_cleanup_reply + .store(false, Ordering::Release); + test.node.shutdown().await.unwrap(); + assert_eq!(*test.provider.authority.closes.lock().unwrap(), vec![1, 2]); + assert!(test.withdrawn.load(Ordering::Acquire)); + assert_eq!(test.node.stats().retained_bytes(), 0); + assert!(error_contains( + observed.first_failure().unwrap().as_ref(), + "accepted cleanup close reply lost" + )); + test.readback(17).await; +} diff --git a/crates/cellule-host/tests/node/durability/observation.rs b/crates/cellule-host/tests/node/durability/observation.rs new file mode 100644 index 00000000..757f4cc5 --- /dev/null +++ b/crates/cellule-host/tests/node/durability/observation.rs @@ -0,0 +1,187 @@ +//! Public supervision captures use the original native owner and request bank. +use super::*; +use std::sync::atomic::AtomicUsize; + +#[derive(Default)] +struct IdleProvider { + calls: AtomicUsize, +} +impl NodeDurabilityProvider for IdleProvider { + fn rotation_required( + self: Arc, + _live_node_limit: usize, + ) -> Pin> + Send>> { + Box::pin(async { Ok(false) }) + } + + fn recruit( + self: Arc, + _limits: ReplicaLimits, + _bytes: u64, + _live: usize, + ) -> Pin>> + Send>> { + self.calls.fetch_add(1, Ordering::Relaxed); + Box::pin(async { Ok(None) }) + } +} + +fn configuration() -> NodeDurabilitySupervisorConfig { + NodeDurabilitySupervisorConfig::new( + ApplicationId::from_bytes([3; 16]), + ReplicaLimits::default(), + 1, + 2, + Duration::from_millis(10), + Duration::from_secs(3600), + u64::MAX, + ) + .unwrap() +} + +#[tokio::test] +async fn metadata_admission_precedes_provider_start_and_releases_after_join() { + let node = CellNodeBuilder::new(application()) + .with_runtime(SqlWorkerPool::new(1, 4).unwrap(), 16 << 10) + .with_replica_host(ReplicaHost::default()) + .with_session(SessionId::from_bytes([91; 16])) + .build() + .unwrap(); + node.install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + let provider = Arc::new(IdleProvider::default()); + let budget = node + .runtime() + .try_reserve_node_metadata_bytes(node.stats().retained_capacity_bytes()) + .unwrap(); + assert!(matches!( + node.install_node_durability_provider(provider.clone(), configuration()), + Err(Error::Capacity(_)) + )); + assert_eq!(provider.calls.load(Ordering::Relaxed), 0); + assert!(node.fleet_durability_supervisor(clock()).unwrap().is_none()); + assert!( + node.owned_component::(NODE_DURABILITY_PROVIDER_COMPONENT) + .is_none() + ); + drop(budget); + assert!(matches!( + node.runtime().try_reserve_node_bytes(1), + Err(Error::Fenced) + )); + node.install_node_durability_provider(provider.clone(), configuration()) + .unwrap(); + until(|| provider.calls.load(Ordering::Relaxed) > 0).await; + assert_eq!(node.stats().retained_bytes(), 4 * 1024); + let before = node.stats().retained_bytes(); + for _ in 0..20 { + let observed = node.fleet_durability_supervisor(clock()).unwrap().unwrap(); + assert_eq!(observed.application, ApplicationId::from_bytes([3; 16])); + assert_eq!(observed.session, SessionId::from_bytes([91; 16])); + assert_eq!(observed.state, NodeDurabilitySupervisorState::Running); + let inventory = observed.rotations.unwrap(); + assert!(!inventory.stopped); + assert!(inventory.pending.is_none() && inventory.completed.is_none()); + } + assert_eq!(node.stats().retained_bytes(), before); + assert!(matches!( + node.runtime().try_reserve_node_bytes(1), + Err(Error::Fenced) + )); + node.shutdown().await.unwrap(); + assert_eq!(node.stats().retained_bytes(), 0); + assert!(node.fleet_durability_supervisor(clock()).unwrap().is_none()); +} + +#[tokio::test] +async fn unbound_invalid_and_wrong_typed_supervisor_capture_are_distinct() { + let node = CellNodeBuilder::new(application()) + .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 << 20) + .with_replica_host(ReplicaHost::default()) + .with_session(SessionId::from_bytes([92; 16])) + .build() + .unwrap(); + assert!(node.fleet_durability_supervisor(clock()).unwrap().is_none()); + assert!(matches!( + node.fleet_durability_supervisor(-1), + Err(Error::Node(_)) + )); + node.install_owned_component_with_drain( + "node-durability-supervisor", + Arc::new(7_u8), + || async { Ok(()) }, + ) + .unwrap(); + assert!(matches!( + node.fleet_durability_supervisor(clock()), + Err(Error::Control(_)) + )); + node.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn failed_member_and_original_completed_rotation_remain_visible_without_new_admission() { + let test = Fixture::new().await; + test.command(245, 17).await; + test.provider + .transport + .lose_retire + .store(true, Ordering::Release); + let request = test.node.request_node_log_rotation(1).unwrap(); + until(|| request.observe().unwrap().first_failure().is_some()).await; + let original = request.observe().unwrap().first_failure().unwrap().clone(); + let stats = test.node.stats(); + let budget = test + .node + .runtime() + .try_reserve_node_bytes(stats.retained_capacity_bytes() - stats.retained_bytes()) + .unwrap(); + let retained = test.node.stats().retained_bytes(); + for _ in 0..20 { + let observed = test + .node + .fleet_durability_supervisor(clock()) + .unwrap() + .unwrap(); + assert_eq!(observed.state, NodeDurabilitySupervisorState::Running); + let inventory = observed.rotations.unwrap(); + assert_eq!(inventory.running_epoch, Some(1)); + let pending = inventory.pending.unwrap(); + assert_eq!(pending.epoch, 1); + assert_eq!(pending.progress.phase(), NodeLogRotationPhase::Retiring); + assert!(Arc::ptr_eq( + pending.progress.first_failure().unwrap(), + &original + )); + assert!(pending.progress.completion().is_none()); + } + assert_eq!(test.node.stats().retained_bytes(), retained); + drop(budget); + test.provider + .transport + .lose_retire + .store(false, Ordering::Release); + until(|| request.observe().unwrap().completion().is_some()).await; + let completed = request.observe().unwrap().completion().unwrap().clone(); + let observed = test + .node + .fleet_durability_supervisor(clock()) + .unwrap() + .unwrap(); + assert_eq!(observed.state, NodeDurabilitySupervisorState::Running); + let inventory = observed.rotations.unwrap(); + assert!(inventory.pending.is_none()); + let latest = inventory.completed.unwrap(); + assert_eq!(latest.epoch, 1); + assert!(Arc::ptr_eq( + latest.progress.completion().unwrap(), + &completed + )); + assert!(Arc::ptr_eq( + latest.progress.first_failure().unwrap(), + &original + )); + assert_eq!(completed.replacement_epoch(), 2); + test.node.shutdown().await.unwrap(); + assert_eq!(test.node.stats().retained_bytes(), 0); + test.readback(17).await; +} diff --git a/crates/cellule-host/tests/node/fleet_actions.rs b/crates/cellule-host/tests/node/fleet_actions.rs new file mode 100644 index 00000000..3a2f3839 --- /dev/null +++ b/crates/cellule-host/tests/node/fleet_actions.rs @@ -0,0 +1,1198 @@ +//! Atomic acceptance, lost replies, retained source proof and shutdown joins. + +use super::*; +use cellule_host::fleet::{ + FleetActionAcceptance, FleetActionJournal, FleetAdapterFuture, FleetCellInputs, + FleetCellProvider, FleetJournalSnapshot, FleetRecoveryInputs, FleetSnapshotRequest, +}; +use cellule_runtime::cell::actor::{CellHandle, CellInventoryEntry}; +use cellule_runtime::cell::catalog::{CatalogEntry, CellCatalog}; +use cellule_runtime::control::{Owner, authority::CellAuthority}; +use cellule_runtime::fleet::operations::*; +use cellule_runtime::identity::{CellTarget, IncarnationId, NamespaceId, NodeId, TenantId}; +use cellule_runtime::ltx::{CellReplica, CellStorageLayout}; +use cellule_store::Store; +use object_store::{memory::InMemory, path::Path}; +use std::collections::HashMap; +use std::sync::atomic::AtomicUsize; + +pub(super) fn clock() -> i64 { + i64::try_from( + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_millis(), + ) + .unwrap() +} + +pub(super) fn scope() -> FleetScope { + FleetScope { + fleet: Digest::from_bytes([200; 32]), + application: ApplicationId::from_bytes([3; 16]), + } +} + +struct JournalState { + head: FleetHead, + registry: RegistryVersion, + records: HashMap)>, + bases: HashMap, + recovery_bases: HashMap, + recovery_evidence: HashMap, + intent: NodeIntent, +} + +pub(super) struct Journal { + state: Mutex, + accepts: AtomicUsize, + publications: AtomicUsize, + lose_publication_reply: AtomicBool, + block_publication: AtomicBool, + entered: tokio::sync::Notify, + resume: tokio::sync::Semaphore, + pub(super) lose_basis_reply: AtomicBool, + pub(super) block_basis: AtomicBool, + pub(super) basis_entered: tokio::sync::Notify, + pub(super) basis_resume: tokio::sync::Semaphore, + pub(super) basis_writes: AtomicUsize, + pub(super) panic_basis: AtomicBool, + pub(super) lose_recovery_evidence_reply: AtomicBool, + pub(super) block_inspections: AtomicBool, + pub(super) inspection_entered: tokio::sync::Notify, + pub(super) inspection_resume: tokio::sync::Semaphore, + pub(super) snapshot_calls: AtomicUsize, + pub(super) pause_snapshot_post: AtomicBool, + pub(super) snapshot_post_entered: tokio::sync::Notify, + pub(super) snapshot_post_resume: tokio::sync::Semaphore, +} + +impl Journal { + pub(super) fn snapshot_request( + &self, + subject: cellule_host::fleet::FleetSnapshotSubject, + nonce: u8, + ) -> FleetSnapshotRequest { + let state = self.state.lock().unwrap(); + let expected = FleetJournalSnapshot::new(state.head.clone(), state.registry).unwrap(); + let now = clock(); + let deadline = (now + 3_000).min(state.head.controller().unwrap().expires_at_ms); + FleetSnapshotRequest::new( + expected, + Digest::from_bytes([nonce; 32]), + state.intent.node(), + state.intent.session(), + subject, + 128, + now, + deadline, + ) + .unwrap() + } + pub(super) fn advance_snapshot_registry(&self) { + let mut state = self.state.lock().unwrap(); + state.registry = state.registry.advance(state.registry.revision()).unwrap(); + } + + fn new() -> Self { + Self { + state: Mutex::new(JournalState { + head: FleetHead::new(scope(), clock()).unwrap(), + registry: RegistryVersion::new(scope()).unwrap(), + records: HashMap::new(), + bases: HashMap::new(), + recovery_bases: HashMap::new(), + recovery_evidence: HashMap::new(), + intent: NodeIntent::initial( + scope(), + NodeId::from_bytes([201; 16]), + SessionId::from_bytes([201; 16]), + ) + .unwrap(), + }), + accepts: AtomicUsize::new(0), + publications: AtomicUsize::new(0), + lose_publication_reply: AtomicBool::new(false), + block_publication: AtomicBool::new(false), + entered: tokio::sync::Notify::new(), + resume: tokio::sync::Semaphore::new(0), + lose_basis_reply: AtomicBool::new(false), + block_basis: AtomicBool::new(false), + basis_entered: tokio::sync::Notify::new(), + basis_resume: tokio::sync::Semaphore::new(0), + basis_writes: AtomicUsize::new(0), + panic_basis: AtomicBool::new(false), + lose_recovery_evidence_reply: AtomicBool::new(false), + block_inspections: AtomicBool::new(false), + inspection_entered: tokio::sync::Notify::new(), + inspection_resume: tokio::sync::Semaphore::new(0), + snapshot_calls: AtomicUsize::new(0), + pause_snapshot_post: AtomicBool::new(false), + snapshot_post_entered: tokio::sync::Notify::new(), + snapshot_post_resume: tokio::sync::Semaphore::new(0), + } + } + + pub(super) fn reset_preparing(&self, spec: MoveAttemptSpec) { + let mut state = self.state.lock().unwrap(); + assert!(state.records.is_empty()); + let now = clock(); + state.head = FleetHead::new(scope(), now) + .unwrap() + .claim( + FleetProfile::default(), + 0, + SessionId::from_bytes([206; 16]), + now, + ) + .unwrap(); + drop(state); + self.transition(JournalTransition::Allocate(spec.clone())); + self.transition(JournalTransition::Attempt { + id: spec.id, + event: AttemptEvent::BeginPrepare, + }); + } + + pub(super) fn reset_maintenance(&self, node: NodeId, session: SessionId) -> FleetAction { + let now = clock(); + let mut state = self.state.lock().unwrap(); + assert!(state.records.is_empty()); + state.head = FleetHead::new(scope(), now) + .unwrap() + .claim( + FleetProfile::default(), + 0, + SessionId::from_bytes([206; 16]), + now, + ) + .unwrap(); + drop(state); + self.transition(JournalTransition::BeginMaintenance( + MaintenanceOperation::new( + OperationId::from_bytes([207; 16]).unwrap(), + Digest::from_bytes([208; 32]), + node, + session, + 2, + now, + now + 60_000, + ) + .unwrap(), + )); + self.maintenance_action(MaintenanceAction::Cordon) + } + + pub(super) fn maintenance_action(&self, effect: MaintenanceAction) -> FleetAction { + self.state + .lock() + .unwrap() + .head + .maintenance_action(effect, clock()) + .unwrap() + } + + pub(super) fn hold_next_result(&self) { + self.block_publication.store(true, Ordering::SeqCst); + } + + pub(super) async fn wait_for_result_publication(&self) { + tokio::time::timeout(Duration::from_secs(5), self.entered.notified()) + .await + .unwrap(); + } + + pub(super) fn resume_result_publication(&self) { + self.resume.add_permits(1); + } + + pub(super) fn transition(&self, event: JournalTransition) { + let mut state = self.state.lock().unwrap(); + state.head = state + .head + .transition( + FleetProfile::default(), + state.head.revision(), + state.head.controller().unwrap().epoch, + clock(), + event, + ) + .unwrap(); + } + + pub(super) fn action(&self, id: AttemptId, effect: MovementAction) -> FleetAction { + self.state + .lock() + .unwrap() + .head + .movement_action(id, effect, clock()) + .unwrap() + } + + pub(super) fn basis(&self, key: Digest) -> Option { + self.state.lock().unwrap().bases.get(&key).cloned() + } + + pub(super) fn recovery_basis(&self, key: Digest) -> Option { + self.state.lock().unwrap().recovery_bases.get(&key).cloned() + } + pub(super) fn recovery_evidence(&self, key: Digest) -> Option { + self.state + .lock() + .unwrap() + .recovery_evidence + .get(&key) + .cloned() + } + pub(super) fn current_attempt(&self) -> MoveAttempt { + self.state.lock().unwrap().head.attempts()[0].clone() + } + pub(super) fn accepted_count(&self) -> usize { + self.accepts.load(Ordering::SeqCst) + } + pub(super) fn registry(&self) -> RegistryVersion { + self.state.lock().unwrap().registry + } + + pub(super) fn lose_next_result_reply(&self) { + self.lose_publication_reply.store(true, Ordering::SeqCst); + } +} + +impl FleetActionJournal for Journal { + fn authorize_snapshot<'a>( + &'a self, + request: &'a FleetSnapshotRequest, + now_ms: i64, + ) -> FleetAdapterFuture<'a, ()> { + Box::pin(async move { + let call = self.snapshot_calls.fetch_add(1, Ordering::SeqCst); + if self.block_inspections.load(Ordering::SeqCst) { + self.inspection_entered.notify_one(); + self.inspection_resume.acquire().await.unwrap().forget(); + } + if call == 1 && self.pause_snapshot_post.load(Ordering::SeqCst) { + self.snapshot_post_entered.notify_one(); + self.snapshot_post_resume.acquire().await.unwrap().forget(); + } + let state = self.state.lock().unwrap(); + let snapshot = FleetJournalSnapshot::new(state.head.clone(), state.registry)?; + request.authorize_against(&snapshot, &state.intent, now_ms)?; + Ok(()) + }) + } + + fn authorize_inspection<'a>( + &'a self, + request: &'a FleetInspectionRequest, + now_ms: i64, + ) -> FleetAdapterFuture<'a, ()> { + Box::pin(async move { + if self.block_inspections.load(Ordering::SeqCst) { + self.inspection_entered.notify_one(); + self.inspection_resume.acquire().await.unwrap().forget(); + } + let state = self.state.lock().unwrap(); + request.authorize_against(&state.head, state.registry, now_ms)?; + Ok(()) + }) + } + + fn accept_action<'a>( + &'a self, + action: &'a FleetAction, + node: NodeId, + session: SessionId, + now_ms: i64, + ) -> FleetAdapterFuture<'a, FleetActionAcceptance> { + Box::pin(async move { + let mut state = self.state.lock().unwrap(); + let key = action.key()?; + if let Some((accepted, result)) = state.records.get(&key) { + accepted.validate_replay(action, node, session)?; + return Ok(FleetActionAcceptance::Existing { + accepted: accepted.clone(), + result: result.clone().map(Box::new), + }); + } + // One critical section linearizes head validation and acceptance. + let accepted = + AcceptedFleetAction::new(action.clone(), &state.head, node, session, now_ms)?; + state.records.insert(key, (accepted.clone(), None)); + self.accepts.fetch_add(1, Ordering::SeqCst); + Ok(FleetActionAcceptance::New(accepted)) + }) + } + + fn publish_action_result<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + result: &'a FleetActionOutcome, + ) -> FleetAdapterFuture<'a, ()> { + Box::pin(async move { + self.publications.fetch_add(1, Ordering::SeqCst); + if self.block_publication.swap(false, Ordering::SeqCst) { + self.entered.notify_one(); + self.resume.acquire().await.unwrap().forget(); + } + accepted.validate_result(result)?; + { + let mut state = self.state.lock().unwrap(); + let (original, previous) = state.records.get_mut(&result.action_key).unwrap(); + if original != accepted { + return Err(Box::new(OperationError::Conflict) + as Box); + } + if let Some(previous) = previous + && previous != result + && !matches!(previous.outcome, FleetOutcome::Unknown) + { + return Err(Box::new(OperationError::Conflict) + as Box); + } + *previous = Some(result.clone()); + } + if self.lose_publication_reply.swap(false, Ordering::SeqCst) { + return Err(Box::new(std::io::Error::other( + "injected lost result publication reply", + )) + as Box); + } + Ok(()) + }) + } + + fn load_movement_action<'a>( + &'a self, + scope: FleetScope, + attempt: AttemptId, + effect: MovementAction, + node: NodeId, + session: SessionId, + ) -> FleetAdapterFuture<'a, Option> { + Box::pin(async move { + let state = self.state.lock().unwrap(); + Ok(state.records.values().find_map(|(accepted, result)| { + let FleetActionKind::Movement { + action, + attempt: original, + } = accepted.action().kind() + else { + return None; + }; + (accepted.action().scope() == scope + && original.spec().id == attempt + && *action == effect + && accepted.node() == node + && accepted.session() == session) + .then(|| FleetActionAcceptance::Existing { + accepted: accepted.clone(), + result: result.clone().map(Box::new), + }) + })) + }) + } + + fn record_acquisition_basis<'a>( + &'a self, + basis: &'a AcquisitionBasis, + ) -> FleetAdapterFuture<'a, AcquisitionBasis> { + Box::pin(async move { + if self.block_basis.swap(false, Ordering::SeqCst) { + self.basis_entered.notify_one(); + self.basis_resume.acquire().await.unwrap().forget(); + } + assert!( + !self.panic_basis.swap(false, Ordering::SeqCst), + "injected acquisition-basis panic" + ); + let retained = { + let mut state = self.state.lock().unwrap(); + let key = basis.accepted().action().key()?; + if state + .records + .get(&key) + .is_none_or(|(original, _)| original != basis.accepted()) + { + return Err(Box::new(OperationError::Conflict) + as Box); + } + if let Some(original) = state.bases.get(&key) { + if original.accepted() != basis.accepted() + || original.control() != basis.control() + { + return Err(Box::new(OperationError::Conflict) + as Box); + } + original.clone() + } else { + state.bases.insert(key, basis.clone()); + self.basis_writes.fetch_add(1, Ordering::SeqCst); + basis.clone() + } + }; + if self.lose_basis_reply.swap(false, Ordering::SeqCst) { + return Err(Box::new(std::io::Error::other( + "injected lost acquisition-basis reply", + )) + as Box); + } + Ok(retained) + }) + } + + fn load_acquisition_basis<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + ) -> FleetAdapterFuture<'a, Option> { + Box::pin(async move { + let basis = self + .state + .lock() + .unwrap() + .bases + .get(&accepted.action().key()?) + .cloned(); + if basis + .as_ref() + .is_some_and(|basis| basis.accepted() != accepted) + { + return Err( + Box::new(OperationError::Conflict) as Box + ); + } + Ok(basis) + }) + } + fn record_recovery_basis<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + basis: &'a RecoveryBasis, + ) -> FleetAdapterFuture<'a, RecoveryBasis> { + Box::pin(async move { + if self.block_basis.swap(false, Ordering::SeqCst) { + self.basis_entered.notify_one(); + self.basis_resume.acquire().await.unwrap().forget(); + } + basis.validate_acceptance(accepted)?; + let retained = { + let mut state = self.state.lock().unwrap(); + let key = accepted.action().key()?; + if state + .records + .get(&key) + .is_none_or(|(original, _)| original != accepted) + { + return Err(Box::new(OperationError::Conflict) + as Box); + } + if let Some(original) = state.recovery_bases.get(&key) { + original.validate_acceptance(accepted)?; + if original.control() != basis.control() { + return Err(Box::new(OperationError::Conflict) + as Box); + } + original.clone() + } else { + state.recovery_bases.insert(key, basis.clone()); + self.basis_writes.fetch_add(1, Ordering::SeqCst); + basis.clone() + } + }; + if self.lose_basis_reply.swap(false, Ordering::SeqCst) { + return Err( + Box::new(std::io::Error::other("injected lost recovery-basis reply")) + as Box, + ); + } + Ok(retained) + }) + } + fn load_recovery_basis<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + ) -> FleetAdapterFuture<'a, Option> { + Box::pin(async move { + let basis = self + .state + .lock() + .unwrap() + .recovery_bases + .get(&accepted.action().key()?) + .cloned(); + if let Some(basis) = &basis { + basis.validate_acceptance(accepted)?; + } + Ok(basis) + }) + } + fn record_recovery_evidence<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + evidence: &'a RecoveryEvidence, + ) -> FleetAdapterFuture<'a, RecoveryEvidence> { + Box::pin(async move { + evidence.basis().validate_acceptance(accepted)?; + let retained = { + let mut state = self.state.lock().unwrap(); + let key = accepted.action().key()?; + if state.recovery_bases.get(&key) != Some(evidence.basis()) { + return Err(Box::new(OperationError::Conflict) + as Box); + } + if let Some(original) = state.recovery_evidence.get(&key) { + if original.basis() != evidence.basis() + || original.restored() != evidence.restored() + { + return Err(Box::new(OperationError::Conflict) + as Box); + } + original.clone() + } else { + state.recovery_evidence.insert(key, evidence.clone()); + evidence.clone() + } + }; + if self + .lose_recovery_evidence_reply + .swap(false, Ordering::SeqCst) + { + return Err(Box::new(std::io::Error::other( + "injected lost recovery evidence reply", + )) + as Box); + } + Ok(retained) + }) + } + fn load_recovery_evidence<'a>( + &'a self, + accepted: &'a AcceptedFleetAction, + ) -> FleetAdapterFuture<'a, Option> { + Box::pin(async move { + let evidence = self + .state + .lock() + .unwrap() + .recovery_evidence + .get(&accepted.action().key()?) + .cloned(); + if let Some(evidence) = &evidence { + evidence.basis().validate_acceptance(accepted)?; + } + Ok(evidence) + }) + } +} + +struct NoCells; + +impl FleetCellProvider for NoCells { + fn cell_inputs<'a>( + &'a self, + _spec: &'a MoveAttemptSpec, + ) -> FleetAdapterFuture<'a, FleetCellInputs> { + Box::pin(async { + Err(Box::new(std::io::Error::other( + "source release must not resolve receiver inputs", + )) as Box) + }) + } + fn recovery_inputs<'a>( + &'a self, + _spec: &'a MoveAttemptSpec, + ) -> FleetAdapterFuture<'a, FleetRecoveryInputs> { + Box::pin(async { + Err(Box::new(std::io::Error::other( + "source must not resolve receiver recovery", + )) as Box) + }) + } +} + +pub(super) struct Fixture { + pub(super) node: Arc, + pub(super) journal: Arc, + pub(super) action: FleetAction, + pub(super) authority: CellAuthority, + pub(super) handle: CellHandle, + pub(super) lease: NodeLeaseGuard, + lease_shutdown: CancellationToken, + tasks: Arc, + pub(super) _root: tempfile::TempDir, +} + +pub(super) async fn fixture() -> Fixture { + let session = SessionId::from_bytes([201; 16]); + let physical = NodeId::from_bytes([201; 16]); + let journal = Arc::new(Journal::new()); + let node = Arc::new( + CellNodeBuilder::new(application()) + .with_runtime( + SqlWorkerPool::new(1, 8) + .unwrap() + .with_native_memory_limit(128 << 20) + .unwrap(), + 64 << 20, + ) + .with_replica_host( + ReplicaHost::default().with_local_disk_budget(DiskBudget::new(8 << 30)), + ) + .with_session(session) + .build() + .unwrap(), + ); + let lease_shutdown = CancellationToken::new(); + let tasks = node + .install_task_group(CancellationToken::new(), lease_shutdown.clone()) + .unwrap(); + node.install_fleet_actions(scope(), physical, journal.clone(), Arc::new(NoCells)) + .unwrap(); + let lease = NodeLeaseGuard::new(clock(), clock() + 60_000).unwrap(); + node.install_node_lease(lease.clone()).unwrap(); + let root = tempfile::tempdir().unwrap(); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("host-fleet-actions"), + [3; 16], + ); + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + scope().application, + NamespaceId::from_bytes([2; 16]), + b"host-fleet-actions", + ) + .unwrap(); + let incarnation = IncarnationId::from_bytes([202; 16]); + let replica = CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + ReplicaLimits { + max_database_bytes: 64 << 20, + max_capture_bytes: 16 << 20, + ..ReplicaLimits::default() + }, + ) + .unwrap(); + let code = node.application().registry().module_digests()[0]; + let proof = CellCatalog::new(layout.clone(), target.tenant()) + .provision(CatalogEntry::new(&target, CatalogRole::Sql, code, 1).unwrap()) + .await + .unwrap(); + let authority = CellAuthority::new(layout); + let initial = authority + .create_initial( + &proof, + incarnation, + Owner { + session, + endpoint: "https://fleet-source.internal:8789".into(), + }, + ) + .await + .unwrap(); + let handle = node + .runtime() + .bootstrap( + proof, + replica, + authority.clone(), + initial, + root.path().join("source.sqlite"), + |transaction| { + transaction.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES (42)", + )?; + Ok(()) + }, + ) + .await + .unwrap(); + let owner = tokio::time::timeout(Duration::from_secs(5), async { + loop { + let page = node.runtime().fleet_cells_page(None, 128).await.unwrap(); + if let Some(CellInventoryEntry::Owned(owner)) = page.entries().first() + && owner.stable_observations == 2 + && owner.cost.is_some() + { + return (**owner).clone(); + } + drop(page); + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + let now = clock(); + let spec = MoveAttemptSpec { + id: AttemptId { + operation: OperationId::from_bytes([203; 16]).unwrap(), + sequence: 1, + }, + target, + incarnation, + source_node: physical, + source: session, + generation: owner.generation, + source_epoch: owner.position.unwrap().epoch, + destination_node: NodeId::from_bytes([204; 16]), + destination: SessionId::from_bytes([204; 16]), + cost: owner.cost.unwrap(), + snapshot_digest: Digest::from_bytes([205; 32]), + deadline_ms: now + 60_000, + }; + let mut head = journal.state.lock().unwrap().head.clone(); + head = head + .claim( + FleetProfile::default(), + head.revision(), + SessionId::from_bytes([206; 16]), + now, + ) + .unwrap(); + for event in [ + JournalTransition::Allocate(spec.clone()), + JournalTransition::Attempt { + id: spec.id, + event: AttemptEvent::BeginPrepare, + }, + JournalTransition::Attempt { + id: spec.id, + event: AttemptEvent::Reserved(ReceiverReservation { + session: spec.destination, + expires_at_ms: spec.deadline_ms, + }), + }, + JournalTransition::Attempt { + id: spec.id, + event: AttemptEvent::BeginRelease, + }, + ] { + head = head + .transition( + FleetProfile::default(), + head.revision(), + head.controller().unwrap().epoch, + now, + event, + ) + .unwrap(); + } + let action = head + .movement_action(spec.id, MovementAction::Release, now) + .unwrap(); + journal.state.lock().unwrap().head = head; + Fixture { + node, + journal, + action, + authority, + handle, + lease, + lease_shutdown, + tasks, + _root: root, + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn concurrent_duplicate_release_has_one_acceptance_and_exact_durable_result() { + let fixture = fixture().await; + let before = fixture + .authority + .load(fixture.handle.cell_id()) + .await + .unwrap() + .unwrap(); + let (a, b) = tokio::join!( + fixture + .node + .apply_fleet_action(fixture.action.clone(), clock()), + fixture + .node + .apply_fleet_action(fixture.action.clone(), clock()) + ); + let a = a.unwrap(); + let b = b.unwrap(); + assert!(a.committed && b.committed); + assert_eq!(a.outcome, b.outcome); + let FleetOutcome::Released(position) = &a.outcome.outcome else { + panic!("not released") + }; + assert_eq!(Some(&position.root), before.value().root.as_ref()); + assert_eq!(fixture.journal.accepts.load(Ordering::SeqCst), 1); + assert_eq!(fixture.node.stats().active_cells(), 0); + assert_eq!( + fixture + .journal + .state + .lock() + .unwrap() + .records + .get(&fixture.action.key().unwrap()) + .unwrap() + .1, + Some(a.outcome.clone()) + ); + fixture.node.shutdown().await.unwrap(); + assert_eq!(fixture.node.stats().retained_bytes(), 0); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn dropped_waiter_keeps_release_and_shutdown_joins_result_publication() { + let fixture = fixture().await; + fixture + .journal + .block_publication + .store(true, Ordering::SeqCst); + let node = Arc::clone(&fixture.node); + let action = fixture.action.clone(); + let waiter = tokio::spawn(async move { node.apply_fleet_action(action, clock()).await }); + tokio::time::timeout(Duration::from_secs(5), fixture.journal.entered.notified()) + .await + .unwrap(); + assert_eq!(fixture.node.stats().active_cells(), 0); + waiter.abort(); + assert!(waiter.await.unwrap_err().is_cancelled()); + let node = Arc::clone(&fixture.node); + let shutdown = tokio::spawn(async move { node.shutdown().await }); + tokio::task::yield_now().await; + assert!(!shutdown.is_finished()); + fixture.journal.resume.add_permits(1); + tokio::time::timeout(Duration::from_secs(5), shutdown) + .await + .unwrap() + .unwrap() + .unwrap(); + assert_eq!(fixture.node.state(), NodeState::Stopped); + assert!(matches!( + fixture + .journal + .state + .lock() + .unwrap() + .records + .get(&fixture.action.key().unwrap()) + .unwrap() + .1 + .as_ref() + .unwrap() + .outcome, + FleetOutcome::Released(_) + )); + assert_eq!(fixture.node.stats().retained_bytes(), 0); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn lost_publication_reply_retries_original_proof_without_releasing_again() { + let fixture = fixture().await; + fixture + .journal + .lose_publication_reply + .store(true, Ordering::SeqCst); + let first = fixture + .node + .apply_fleet_action(fixture.action.clone(), clock()) + .await + .unwrap(); + assert!(!first.committed && first.journal_error.is_some()); + assert!(matches!(first.outcome.outcome, FleetOutcome::Released(_))); + // Advance authority before retrying the lost source result. Its original + // root must survive; a fresh authority read would now name another owner. + let receiver_session = SessionId::from_bytes([204; 16]); + let receiver = CellNodeBuilder::new(application()) + .with_runtime( + SqlWorkerPool::new(1, 8) + .unwrap() + .with_native_memory_limit(128 << 20) + .unwrap(), + 64 << 20, + ) + .with_replica_host(ReplicaHost::default().with_local_disk_budget(DiskBudget::new(8 << 30))) + .with_session(receiver_session) + .build() + .unwrap(); + receiver + .install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + receiver + .install_node_lease(NodeLeaseGuard::new(clock(), clock() + 60_000).unwrap()) + .unwrap(); + let replica = CellReplica::new( + fixture.authority.layout().clone(), + *fixture.handle.cell_id().as_bytes(), + *fixture.handle.incarnation().as_bytes(), + ReplicaLimits { + max_database_bytes: 64 << 20, + max_capture_bytes: 16 << 20, + ..ReplicaLimits::default() + }, + ) + .unwrap(); + let idle = fixture + .authority + .load(fixture.handle.cell_id()) + .await + .unwrap() + .unwrap(); + let activated = receiver + .runtime() + .acquire_idle_restored( + fixture.handle.catalog().clone(), + replica, + fixture.authority.clone(), + idle, + fixture._root.path().join("receiver.sqlite"), + Owner { + session: receiver_session, + endpoint: "https://fleet-receiver.internal:8789".into(), + }, + ) + .await + .unwrap(); + let now = clock(); + activated + .execute( + cellule_runtime::cell::executor::MutationIdentity { + request_id: cellule_runtime::identity::RequestId::from_bytes([207; 16]), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }, + Digest::from_bytes([207; 32]), + now, + 64, + 64, + |transaction| { + transaction.execute("UPDATE counter SET value = 99", [])?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + vec![99], + )) + }, + ) + .await + .unwrap(); + let newer = fixture + .authority + .load(fixture.handle.cell_id()) + .await + .unwrap() + .unwrap(); + let FleetOutcome::Released(original) = &first.outcome.outcome else { + panic!() + }; + assert!(newer.value().epoch > original.epoch); + assert!(newer.value().root.as_ref().unwrap().commit_sequence > original.root.commit_sequence); + let final_result = tokio::time::timeout(Duration::from_secs(5), async { + loop { + let result = fixture + .node + .apply_fleet_action(fixture.action.clone(), clock()) + .await + .unwrap(); + if result.committed { + return result; + } + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + assert_eq!(first.outcome, final_result.outcome); + assert_eq!(fixture.journal.accepts.load(Ordering::SeqCst), 1); + assert!(fixture.journal.publications.load(Ordering::SeqCst) >= 2); + assert_eq!(fixture.node.stats().active_cells(), 0); + fixture.node.shutdown().await.unwrap(); + receiver.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn timed_out_drain_retains_publication_and_can_join_the_same_task_later() { + let fixture = fixture().await; + fixture + .journal + .block_publication + .store(true, Ordering::SeqCst); + let node = Arc::clone(&fixture.node); + let action = fixture.action.clone(); + let waiter = tokio::spawn(async move { node.apply_fleet_action(action, clock()).await }); + tokio::time::timeout(Duration::from_secs(5), fixture.journal.entered.notified()) + .await + .unwrap(); + waiter.abort(); + assert!(waiter.await.unwrap_err().is_cancelled()); + assert!( + fixture + .node + .drain_until(Some(Instant::now() + Duration::from_millis(50))) + .await + .is_err() + ); + assert_ne!(fixture.node.state(), NodeState::Stopped); + assert!( + fixture + .journal + .state + .lock() + .unwrap() + .records + .get(&fixture.action.key().unwrap()) + .unwrap() + .1 + .is_none() + ); + assert!(fixture.node.stats().retained_bytes() >= 2 * MAX_RECORD_BYTES as usize); + fixture.journal.resume.add_permits(1); + tokio::time::timeout(Duration::from_secs(5), fixture.node.shutdown()) + .await + .unwrap() + .unwrap(); + assert_eq!(fixture.node.state(), NodeState::Stopped); + assert!(matches!( + fixture + .journal + .state + .lock() + .unwrap() + .records + .get(&fixture.action.key().unwrap()) + .unwrap() + .1 + .as_ref() + .unwrap() + .outcome, + FleetOutcome::Released(_) + )); + assert_eq!(fixture.node.stats().retained_bytes(), 0); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn host_deadline_cannot_cancel_runtime_drain_of_an_accepted_query() { + let fixture = fixture().await; + let tasks = Arc::clone(&fixture.tasks); + let withdrawal = fixture.lease_shutdown.clone(); + let runtime = fixture.node.runtime(); + tasks + .spawn_lease_maintenance(async move { + withdrawal.cancelled().await; + assert_eq!(runtime.stats().active_cells(), 0); + Ok::<(), Error>(()) + }) + .unwrap(); + let (started, observed) = std::sync::mpsc::channel(); + let (release, resumed) = std::sync::mpsc::channel(); + let handle = fixture.handle.clone(); + let query = tokio::spawn(async move { + handle + .query(64, 64, move |connection| { + started.send(()).unwrap(); + resumed.recv().unwrap(); + let value = connection + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; + Ok(value.to_be_bytes().to_vec()) + }) + .await + }); + tokio::task::spawn_blocking(move || observed.recv().unwrap()) + .await + .unwrap(); + assert!( + fixture + .node + .drain_until(Some(Instant::now() + Duration::from_millis(50))) + .await + .is_err() + ); + assert_ne!(fixture.node.state(), NodeState::Stopped); + assert!(!query.is_finished()); + assert!(!fixture.lease_shutdown.is_cancelled()); + release.send(()).unwrap(); + assert_eq!(query.await.unwrap().unwrap(), 42_i64.to_be_bytes()); + tokio::time::timeout(Duration::from_secs(5), fixture.node.shutdown()) + .await + .unwrap() + .unwrap(); + assert_eq!(fixture.node.state(), NodeState::Stopped); + assert_eq!(fixture.node.stats().active_cells(), 0); + assert_eq!(fixture.node.stats().retained_bytes(), 0); + assert!(fixture.lease_shutdown.is_cancelled()); + let final_control = fixture + .authority + .load(fixture.handle.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!( + final_control.value().state, + cellule_runtime::control::ControlState::Idle + ); + assert!(final_control.value().owner.is_none()); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn existing_acceptance_without_result_never_repeats_unobserved_release() { + let fixture = fixture().await; + let _ = fixture + .journal + .accept_action( + &fixture.action, + NodeId::from_bytes([201; 16]), + SessionId::from_bytes([201; 16]), + clock(), + ) + .await + .unwrap(); + let before = fixture + .authority + .load(fixture.handle.cell_id()) + .await + .unwrap() + .unwrap(); + let result = fixture + .node + .apply_fleet_action(fixture.action.clone(), clock()) + .await + .unwrap(); + assert!(result.committed && result.execution_error.is_some()); + assert!(matches!(result.outcome.outcome, FleetOutcome::Unknown)); + assert_eq!(fixture.node.stats().active_cells(), 1); + assert_eq!( + fixture + .authority + .load(fixture.handle.cell_id()) + .await + .unwrap() + .unwrap() + .value(), + before.value() + ); + fixture.node.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn stale_authorization_fails_before_acceptance_or_source_effect() { + let fixture = fixture().await; + { + let mut state = fixture.journal.state.lock().unwrap(); + state.head = state + .head + .claim( + FleetProfile::default(), + state.head.revision(), + state.head.controller().unwrap().claimant, + clock(), + ) + .unwrap(); + } + assert!( + fixture + .node + .apply_fleet_action(fixture.action.clone(), clock()) + .await + .is_err() + ); + assert_eq!(fixture.journal.accepts.load(Ordering::SeqCst), 0); + assert_eq!(fixture.node.stats().active_cells(), 1); + assert!(fixture.journal.state.lock().unwrap().records.is_empty()); + fixture.node.shutdown().await.unwrap(); +} diff --git a/crates/cellule-host/tests/node/fleet_maintenance.rs b/crates/cellule-host/tests/node/fleet_maintenance.rs new file mode 100644 index 00000000..05222003 --- /dev/null +++ b/crates/cellule-host/tests/node/fleet_maintenance.rs @@ -0,0 +1,413 @@ +//! Exact boot-bound cordon through the public node-owned action boundary. + +use super::fleet_actions::{Fixture, clock, fixture}; +use super::*; +use cellule_runtime::fleet::operations::*; +use cellule_runtime::identity::NodeId; +use cellule_runtime::node::NodeMode; + +async fn release_action(fixture: &Fixture) -> FleetAction { + let FleetActionKind::Movement { attempt, .. } = fixture.action.kind() else { + panic!("fixture lacks source identity") + }; + let spec = attempt.spec().clone(); + fixture.journal.reset_preparing(spec.clone()); + fixture.journal.transition(JournalTransition::Attempt { + id: spec.id, + event: AttemptEvent::Reserved(ReceiverReservation { + session: spec.destination, + expires_at_ms: spec.deadline_ms, + }), + }); + fixture + .journal + .transition(JournalTransition::BeginMaintenance( + MaintenanceOperation::new( + spec.id.operation, + Digest::from_bytes([221; 32]), + spec.source_node, + spec.source, + 2, + clock(), + spec.deadline_ms, + ) + .unwrap(), + )); + let cordon = fixture + .journal + .maintenance_action(MaintenanceAction::Cordon); + let result = fixture + .node + .apply_fleet_action(cordon, clock()) + .await + .unwrap(); + assert!(result.committed && result.execution_error.is_none()); + assert_eq!(result.outcome.outcome, FleetOutcome::Cordoned); + fixture + .journal + .transition(JournalTransition::Maintenance(MaintenanceEvent::Cordoned)); + fixture.journal.transition(JournalTransition::Maintenance( + MaintenanceEvent::BeginEvacuation, + )); + fixture.journal.transition(JournalTransition::Attempt { + id: spec.id, + event: AttemptEvent::BeginMaintenanceRelease, + }); + fixture + .journal + .action(spec.id, MovementAction::ReleaseMaintenance) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn busy_maintenance_release_preserves_sql_receipt_after_waiter_or_publication_loss() { + use cellule_runtime::cell::actor::CellInventoryEntry; + use cellule_runtime::cell::executor::{HandlerOutcome, MutationIdentity, Resolution}; + use cellule_runtime::control::Owner; + use cellule_runtime::identity::RequestId; + use cellule_runtime::ltx::CellReplica; + + for (drop_waiter, lose_result) in [(false, false), (true, false), (false, true)] { + let fixture = fixture().await; + let action = release_action(&fixture).await; + if lose_result { + fixture.journal.lose_next_result_reply(); + } + let now = clock(); + let identity = MutationIdentity { + request_id: RequestId::from_bytes([222; 16]), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }; + let digest = Digest::from_bytes([223; 32]); + let entered = Arc::new(tokio::sync::Notify::new()); + let signal = entered.clone(); + let (resume, paused) = std::sync::mpsc::channel(); + let handle = fixture.handle.clone(); + let command = tokio::spawn(async move { + handle + .execute(identity, digest, now, 64, 64, move |transaction| { + signal.notify_one(); + paused.recv().unwrap(); + transaction.execute("UPDATE counter SET value = 77", [])?; + Ok(HandlerOutcome::Success(vec![77])) + }) + .await + }); + tokio::time::timeout(Duration::from_secs(3), entered.notified()) + .await + .unwrap(); + let node = fixture.node.clone(); + let input = action.clone(); + let waiter = tokio::spawn(async move { node.apply_fleet_action(input, clock()).await }); + let quiesced = tokio::time::timeout(Duration::from_secs(3), async { + loop { + let page = fixture + .node + .runtime() + .fleet_cells_page(None, 128) + .await + .unwrap(); + if page.entries().iter().any(|entry| { + matches!(entry, + CellInventoryEntry::Owned(owner) if owner.quiescing) + }) { + break; + } + drop(page); + tokio::task::yield_now().await; + } + }) + .await; + let foreground = tokio::time::timeout( + Duration::from_millis(500), + fixture.handle.query(64, 64, |_| { + panic!("new foreground work crossed maintenance closure") + }), + ) + .await; + if drop_waiter { + waiter.abort(); + } + // Release and join accepted SQL before asserting any preflight evidence. + resume.send(()).unwrap(); + let receipt = command.await.unwrap().unwrap(); + assert!(quiesced.is_ok()); + assert!(matches!(foreground, Ok(Err(Error::CellDraining)))); + if drop_waiter { + assert!(waiter.await.unwrap_err().is_cancelled()); + } else { + let first = waiter.await.unwrap().unwrap(); + assert_eq!(first.committed, !lose_result); + assert_eq!(first.journal_error.is_some(), lose_result); + assert!(first.execution_error.is_none()); + } + let completed = fixture + .node + .apply_fleet_action(action.clone(), clock()) + .await + .unwrap(); + assert!(completed.committed && completed.execution_error.is_none()); + let FleetOutcome::Released(position) = &completed.outcome.outcome else { + panic!("maintenance did not release: {completed:?}") + }; + let idle = fixture + .authority + .load(fixture.handle.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!( + idle.value().state, + cellule_runtime::control::ControlState::Idle + ); + assert_eq!(idle.value().root.as_ref(), Some(&position.root)); + assert_eq!(position.root.commit_sequence, receipt.commit_sequence()); + assert_eq!(fixture.journal.accepted_count(), 2); // Cordon and explicit source release. + + let session = SessionId::from_bytes([204; 16]); + let receiver = CellNodeBuilder::new(application()) + .with_runtime( + SqlWorkerPool::new(1, 8) + .unwrap() + .with_native_memory_limit(128 << 20) + .unwrap(), + 64 << 20, + ) + .with_replica_host( + ReplicaHost::default().with_local_disk_budget(DiskBudget::new(8 << 30)), + ) + .with_session(session) + .build() + .unwrap(); + receiver + .install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + receiver + .install_node_lease(NodeLeaseGuard::new(clock(), clock() + 60_000).unwrap()) + .unwrap(); + let replica = CellReplica::new( + fixture.authority.layout().clone(), + *fixture.handle.cell_id().as_bytes(), + *fixture.handle.incarnation().as_bytes(), + ReplicaLimits { + max_database_bytes: 64 << 20, + max_capture_bytes: 16 << 20, + ..ReplicaLimits::default() + }, + ) + .unwrap(); + let destination = receiver + .runtime() + .acquire_idle_restored( + fixture.handle.catalog().clone(), + replica, + fixture.authority.clone(), + idle, + fixture._root.path().join("receiver.sqlite"), + Owner { + session, + endpoint: "https://fleet-receiver.internal:8789".into(), + }, + ) + .await + .unwrap(); + assert_eq!( + destination + .resolve(identity, digest, clock(), 64) + .await + .unwrap(), + Resolution::Committed(receipt) + ); + assert_eq!( + destination + .query(64, 64, |connection| { + let value = connection + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; + Ok(value.to_be_bytes().to_vec()) + }) + .await + .unwrap(), + 77_i64.to_be_bytes() + ); + // Retained release is historical proof even after authority moves on. + let replay = fixture + .node + .apply_fleet_action(action, clock()) + .await + .unwrap(); + assert_eq!(replay.outcome, completed.outcome); + assert_eq!(fixture.journal.accepted_count(), 2); + receiver.shutdown().await.unwrap(); + assert_eq!(receiver.stats().active_cells(), 0); + assert_eq!(receiver.stats().retained_bytes(), 0); + close(fixture).await; + } +} + +fn cordon(fixture: &Fixture) -> FleetAction { + fixture.journal.reset_maintenance( + NodeId::from_bytes([201; 16]), + SessionId::from_bytes([201; 16]), + ) +} + +async fn check_existing_owner(fixture: &Fixture) { + let value = fixture + .handle + .query(64, 64, |connection| { + let value = connection + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; + Ok(value.to_be_bytes().to_vec()) + }) + .await + .unwrap(); + assert_eq!(value, 42_i64.to_be_bytes()); + assert_eq!(fixture.node.stats().active_cells(), 1); +} + +async fn close(fixture: Fixture) { + fixture.node.shutdown().await.unwrap(); + assert_eq!(fixture.node.stats().active_cells(), 0); + assert_eq!(fixture.node.stats().retained_bytes(), 0); + assert_eq!(fixture.node.state(), NodeState::Stopped); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn committed_cordon_closes_shared_role_gate_keeps_owner_and_is_idempotent() { + let fixture = fixture().await; + let action = cordon(&fixture); + let gate = fixture.node.runtime().node_admission(); + assert_eq!(gate.mode().unwrap(), NodeMode::Active); + let (a, b) = tokio::join!( + fixture.node.apply_fleet_action(action.clone(), clock()), + fixture.node.apply_fleet_action(action.clone(), clock()), + ); + let a = a.unwrap(); + let b = b.unwrap(); + assert!(a.committed && b.committed); + assert_eq!(a.outcome, b.outcome); + assert_eq!(a.outcome.outcome, FleetOutcome::Cordoned); + assert_eq!(fixture.journal.accepted_count(), 1); + assert_eq!(gate.mode().unwrap(), NodeMode::Draining); + assert!(matches!(gate.check_new_role(), Err(Error::CellDraining))); + // Even another lifecycle cordon cannot clear terminal maintenance intent. + gate.cordon().unwrap(); + assert_eq!(gate.mode().unwrap(), NodeMode::Draining); + assert!(fixture.node.is_ready()); + check_existing_owner(&fixture).await; + close(fixture).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn lost_publication_reply_replays_original_cordon_and_keeps_gate_closed() { + let fixture = fixture().await; + let action = cordon(&fixture); + fixture.journal.lose_next_result_reply(); + let original = fixture + .node + .apply_fleet_action(action.clone(), clock()) + .await + .unwrap(); + assert!(!original.committed); + assert!(original.journal_error.is_some()); + assert_eq!( + fixture.node.runtime().node_admission().mode().unwrap(), + NodeMode::Draining + ); + let replay = fixture + .node + .apply_fleet_action(action, clock()) + .await + .unwrap(); + assert!(replay.committed); + assert_eq!(replay.accepted, original.accepted); + assert_eq!(replay.outcome, original.outcome); + assert_eq!(fixture.journal.accepted_count(), 1); + check_existing_owner(&fixture).await; + close(fixture).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn dropped_cordon_waiter_does_not_cancel_node_owned_publication() { + let fixture = fixture().await; + let action = cordon(&fixture); + fixture.journal.hold_next_result(); + let node = fixture.node.clone(); + let waiter = tokio::spawn(async move { node.apply_fleet_action(action, clock()).await }); + fixture.journal.wait_for_result_publication().await; + waiter.abort(); + assert!(waiter.await.unwrap_err().is_cancelled()); + assert_eq!( + fixture.node.runtime().node_admission().mode().unwrap(), + NodeMode::Draining + ); + fixture.journal.resume_result_publication(); + close(fixture).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn wrong_boot_stale_intent_and_unsupported_roles_do_not_accept_effects() { + let fixture = fixture().await; + let wrong = fixture.journal.reset_maintenance( + NodeId::from_bytes([201; 16]), + SessionId::from_bytes([202; 16]), + ); + assert!( + fixture + .node + .apply_fleet_action(wrong, clock()) + .await + .is_err() + ); + assert_eq!(fixture.journal.accepted_count(), 0); + assert_eq!( + fixture.node.runtime().node_admission().mode().unwrap(), + NodeMode::Active + ); + let stale = cordon(&fixture); + fixture.journal.transition(JournalTransition::Maintenance( + MaintenanceEvent::ExtendDeadline(clock() + 120_000), + )); + assert!( + fixture + .node + .apply_fleet_action(stale, clock()) + .await + .is_err() + ); + fixture + .journal + .transition(JournalTransition::Maintenance(MaintenanceEvent::Cordoned)); + fixture.journal.transition(JournalTransition::Maintenance( + MaintenanceEvent::BeginEvacuation, + )); + let roles = fixture + .journal + .maintenance_action(MaintenanceAction::SettleRoles); + assert!( + fixture + .node + .apply_fleet_action(roles, clock()) + .await + .is_err() + ); + let inspection = FleetInspectionRequest::new( + fixture + .journal + .maintenance_action(MaintenanceAction::Inspect), + fixture.journal.registry(), + Digest::from_bytes([211; 32]), + NodeId::from_bytes([201; 16]), + SessionId::from_bytes([201; 16]), + clock() + 30_000, + ) + .unwrap(); + assert!(fixture.node.inspect_fleet_action(inspection).await.is_err()); + assert_eq!(fixture.journal.accepted_count(), 0); + assert_eq!( + fixture.node.runtime().node_admission().mode().unwrap(), + NodeMode::Active + ); + check_existing_owner(&fixture).await; + close(fixture).await; +} diff --git a/crates/cellule-host/tests/node/fleet_receivers/mod.rs b/crates/cellule-host/tests/node/fleet_receivers/mod.rs new file mode 100644 index 00000000..ad092cfe --- /dev/null +++ b/crates/cellule-host/tests/node/fleet_receivers/mod.rs @@ -0,0 +1,1780 @@ +//! Journal-bound receiver admission, takeover, evidence and cancellation. + +use super::fleet_actions::{Fixture, clock, fixture, scope}; +use super::*; +use cellule_host::fleet::{ + FleetActionCompletion, FleetActionJournal, FleetAdapterFuture, FleetCellInputs, + FleetCellProvider, FleetRecoveryInputs, +}; +use cellule_runtime::cell::actor::{CellInventoryEntry, ReceiverState}; +use cellule_runtime::cell::executor::{HandlerOutcome, MutationIdentity, Resolution}; +use cellule_runtime::control::{ControlState, Owner}; +use cellule_runtime::fleet::operations::*; +use cellule_runtime::identity::RequestId; +use cellule_runtime::ltx::CellReplica; + +mod prefix; +mod successor; +mod suffix; + +struct Cells(FleetCellInputs, Arc>>); + +impl FleetCellProvider for Cells { + fn cell_inputs<'a>( + &'a self, + _spec: &'a MoveAttemptSpec, + ) -> FleetAdapterFuture<'a, FleetCellInputs> { + Box::pin(async move { Ok(self.0.clone()) }) + } + fn recovery_inputs<'a>( + &'a self, + _spec: &'a MoveAttemptSpec, + ) -> FleetAdapterFuture<'a, FleetRecoveryInputs> { + Box::pin(async { + self.1.lock().unwrap().clone().ok_or_else(|| { + Box::new(std::io::Error::other( + "canonical recovery proof unavailable", + )) as Box + }) + }) + } +} + +struct Movement { + source: Fixture, + receiver: Arc, + inputs: FleetCellInputs, + spec: MoveAttemptSpec, + recovery_inputs: Arc>>, +} + +impl Movement { + async fn new(memory: usize) -> Self { + Self::with_deadline(memory, 60_000).await + } + + async fn with_deadline(memory: usize, duration_ms: i64) -> Self { + let source = fixture().await; + let FleetActionKind::Movement { attempt, .. } = source.action.kind() else { + panic!() + }; + let mut spec = attempt.spec().clone(); + spec.deadline_ms = clock() + duration_ms; + source.journal.reset_preparing(spec.clone()); + let inputs = FleetCellInputs { + catalog: source.handle.catalog().clone(), + replica: CellReplica::new( + source.authority.layout().clone(), + *spec.target.cell_id().as_bytes(), + *spec.incarnation.as_bytes(), + ReplicaLimits { + max_database_bytes: 64 << 20, + max_capture_bytes: 16 << 20, + ..ReplicaLimits::default() + }, + ) + .unwrap(), + authority: source.authority.clone(), + destination: source._root.path().join("fleet-receiver.sqlite"), + owner: Owner { + session: spec.destination, + endpoint: "https://fleet-receiver.internal:8789".into(), + }, + }; + let receiver = Arc::new( + CellNodeBuilder::new(application()) + .with_runtime( + SqlWorkerPool::new(1, 8) + .unwrap() + .with_native_memory_limit(memory) + .unwrap(), + 64 << 20, + ) + .with_replica_host( + ReplicaHost::default().with_local_disk_budget(DiskBudget::new(8 << 30)), + ) + .with_session(spec.destination) + .build() + .unwrap(), + ); + receiver + .install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + let recovery_inputs = Arc::new(Mutex::new(None)); + receiver + .install_fleet_actions( + scope(), + spec.destination_node, + source.journal.clone(), + Arc::new(Cells(inputs.clone(), recovery_inputs.clone())), + ) + .unwrap(); + receiver + .install_node_lease(NodeLeaseGuard::new(clock(), clock() + 60_000).unwrap()) + .unwrap(); + Self { + source, + receiver, + inputs, + spec, + recovery_inputs, + } + } + + fn event(&self, event: AttemptEvent) { + self.source.journal.transition(JournalTransition::Attempt { + id: self.spec.id, + event, + }); + } + + fn action(&self, effect: MovementAction) -> FleetAction { + self.source.journal.action(self.spec.id, effect) + } + + fn inspection(&self, nonce: u8) -> FleetInspectionRequest { + FleetInspectionRequest::new( + self.action(MovementAction::Inspect), + self.source.journal.registry(), + Digest::from_bytes([nonce; 32]), + self.spec.destination_node, + self.spec.destination, + clock() + 10_000, + ) + .unwrap() + } + + async fn inspect(&self, nonce: u8) -> Arc { + let request = self.inspection(nonce); + let observation = self + .receiver + .inspect_fleet_action(request.clone()) + .await + .unwrap(); + observation.validate_for(&request, clock(), 10_000).unwrap(); + observation + } + + async fn prepare(&self) -> Arc { + apply(&self.receiver, self.action(MovementAction::Prepare)).await + } + + async fn release(&self) -> PublishedPosition { + let reserved = self.prepare().await; + let FleetOutcome::Reserved(reservation) = reserved.outcome.outcome else { + panic!("not reserved: {:?}", reserved) + }; + assert!(reserved.committed && reserved.execution_error.is_none()); + self.event(AttemptEvent::Reserved(reservation)); + self.event(AttemptEvent::BeginRelease); + let released = apply(&self.source.node, self.action(MovementAction::Release)).await; + let FleetOutcome::Released(position) = &released.outcome.outcome else { + panic!("not released: {:?}", released) + }; + assert!(released.committed && released.execution_error.is_none()); + self.event(AttemptEvent::Released(position.clone())); + self.event(AttemptEvent::BeginActivate); + position.clone() + } + + async fn start_recovery(&self, idle: bool) { + let prepared = self.prepare().await; + let FleetOutcome::Reserved(reservation) = prepared.outcome.outcome else { + panic!("not prepared") + }; + self.event(AttemptEvent::Reserved(reservation)); + self.event(AttemptEvent::BeginRelease); + if idle { + // The canonical source release happened, but its action result was + // never observed. Do not invent Released from the later Idle root. + self.source.handle.drain().await.unwrap(); + } + self.source.lease.fence(); + assert!(matches!( + self.source.handle.query(1, 1, |_| Ok(Vec::new())).await, + Err(Error::Fenced) | Err(Error::CellDraining) + )); + let now = clock(); + let image = Digest::from_bytes([221; 32]); + let release = Digest::from_bytes([222; 32]); + let directory = cellule_runtime::node::NodeDirectory::new( + self.inputs.authority.layout().clone(), + scope().fleet, + image, + release, + ); + let key = ed25519_dalek::SigningKey::from_bytes(&[223; 32]); + let signed = |node, session, issued, expires| { + cellule_runtime::node::NodeAdvertisement::sign( + node, + session, + "https://recovery.internal:8789".into(), + scope().fleet, + Digest::from_bytes([224; 32]), + image, + release, + &key, + 1, + issued, + expires, + vec![self.source.node.application().registry().module_digests()[0]], + vec![1], + cellule_runtime::node::NodeFailureDomain::default(), + cellule_runtime::node::NodeCapacity { + free_memory_bytes: 128 << 20, + free_disk_bytes: 8 << 30, + follower_free_bytes: 8 << 30, + follower_retained_bytes: 0, + job_credits: 1, + log_protocol: cellule_runtime::node::NODE_LOG_PROTOCOL_VERSION, + }, + ) + .unwrap() + }; + directory + .create( + signed( + self.spec.source_node, + self.spec.source, + now - 10_000, + now - 1, + ), + now - 10_000, + ) + .await + .unwrap(); + directory + .create( + signed( + self.spec.destination_node, + self.spec.destination, + now, + now + 10_000, + ), + now, + ) + .await + .unwrap(); + let takeover = directory + .claim_expired_for_takeover(self.spec.source, self.spec.destination, now) + .await + .unwrap(); + *self.recovery_inputs.lock().unwrap() = Some(FleetRecoveryInputs { + takeover, + manifests: cellule_runtime::recovery::manifest::RecoveryManifestStore::new( + self.inputs.authority.layout().clone(), + self.inputs.replica.limits(), + ), + }); + self.event(AttemptEvent::OutcomeUnknown); + self.event(AttemptEvent::BeginRecover); + } + + async fn shutdown(&self) { + self.receiver.shutdown().await.unwrap(); + self.source.node.shutdown().await.unwrap(); + for node in [&self.receiver, &self.source.node] { + assert_eq!(node.state(), NodeState::Stopped); + let stats = node.stats(); + assert_eq!(stats.active_cells(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.retained_bytes(), 0); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.file_descriptors(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); + } + } +} + +async fn apply(node: &CellNode, action: FleetAction) -> Arc { + tokio::time::timeout(Duration::from_secs(5), async { + loop { + match node.apply_fleet_action(action.clone(), clock()).await { + Ok(result) => return result, + Err(error) if matches!(error.as_ref(), Error::FleetOperation(error) if matches!(error.as_ref(), OperationError::Busy)) => tokio::task::yield_now().await, + Err(error) => panic!("action failed: {error:?}"), + } + } + }).await.unwrap() +} + +async fn counter(handle: &cellule_runtime::cell::actor::CellHandle) -> i64 { + let bytes = handle + .query(64, 64, |connection| { + Ok(connection + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))? + .to_be_bytes() + .to_vec()) + }) + .await + .unwrap(); + i64::from_be_bytes(bytes.try_into().unwrap()) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn local_eviction_before_fleet_release_proves_refusal_and_joins_receiver_credit() { + let movement = Movement::new(128 << 20).await; + let prepared = movement.prepare().await; + let FleetOutcome::Reserved(reservation) = prepared.outcome.outcome else { + panic!("receiver did not reserve: {prepared:?}") + }; + assert!(prepared.committed && prepared.execution_error.is_none()); + movement.event(AttemptEvent::Reserved(reservation)); + assert_eq!( + movement.receiver.stats().local_disk_reserved_bytes(), + movement.spec.cost.disk_bytes + ); + + // Emergency local shedding uses this same canonical eviction path. The + // exact fleet action has not been accepted while that independent job runs. + assert_eq!( + movement.source.node.runtime().evict_idle(1).await.unwrap(), + 1 + ); + tokio::time::timeout(Duration::from_secs(5), async { + loop { + let current = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + if current.value().state == ControlState::Idle + && movement.source.node.stats().active_cells() == 0 + { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + let idle = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + movement.event(AttemptEvent::BeginRelease); + let action = movement.action(MovementAction::Release); + let refused = apply(&movement.source.node, action.clone()).await; + assert!(refused.committed && refused.journal_error.is_none()); + assert_eq!( + refused.outcome.outcome, + FleetOutcome::Rejected(DrainBlocker::IncompleteObservation) + ); + assert!(matches!( + refused.execution_error.as_deref(), + Some(Error::CellNotActive) + )); + let replay = apply(&movement.source.node, action).await; + assert_eq!(replay.outcome, refused.outcome); + assert_eq!( + movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap() + .value(), + idle.value() + ); + + movement.event(AttemptEvent::ReleaseRefused( + DrainBlocker::IncompleteObservation, + )); + movement.event(AttemptEvent::BeginCancel); + let cancelled = apply(&movement.receiver, movement.action(MovementAction::Cancel)).await; + assert!(cancelled.committed && cancelled.execution_error.is_none()); + assert_eq!(cancelled.outcome.outcome, FleetOutcome::ReceiverCleaned); + assert_eq!(movement.receiver.stats().local_disk_reserved_bytes(), 0); + movement.event(AttemptEvent::Cancelled); + assert!( + movement + .source + .journal + .current_attempt() + .released() + .is_none() + ); + assert!( + movement + .source + .journal + .current_attempt() + .activated() + .is_none() + ); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn accepted_release_without_result_stays_unknown_after_local_eviction() { + let movement = Movement::new(128 << 20).await; + let prepared = movement.prepare().await; + let FleetOutcome::Reserved(reservation) = prepared.outcome.outcome else { + panic!("receiver did not reserve: {prepared:?}") + }; + movement.event(AttemptEvent::Reserved(reservation)); + movement.event(AttemptEvent::BeginRelease); + let action = movement.action(MovementAction::Release); + assert!(matches!( + movement + .source + .journal + .accept_action( + &action, + movement.spec.source_node, + movement.spec.source, + clock() + ) + .await + .unwrap(), + cellule_host::fleet::FleetActionAcceptance::New(_) + )); + + // Journal acceptance with no original result cannot establish whether its + // source effect ran. An independently released Idle root grants no proof. + assert_eq!( + movement.source.node.runtime().evict_idle(1).await.unwrap(), + 1 + ); + tokio::time::timeout(Duration::from_secs(5), async { + loop { + let current = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + if current.value().state == ControlState::Idle + && movement.source.node.stats().active_cells() == 0 + { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + let idle = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + let result = apply(&movement.source.node, action).await; + assert!(result.committed && result.execution_error.is_some()); + assert_eq!(result.outcome.outcome, FleetOutcome::Unknown); + assert_eq!( + movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap() + .value(), + idle.value() + ); + assert_eq!( + movement.receiver.stats().local_disk_reserved_bytes(), + movement.spec.cost.disk_bytes + ); + assert!( + movement + .source + .journal + .current_attempt() + .released() + .is_none() + ); + assert_eq!( + movement.source.journal.current_attempt().phase(), + AttemptPhase::Releasing + ); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn fresh_inspection_bypasses_historical_cache_and_rechecks_stopped_actor() { + let movement = Movement::new(128 << 20).await; + let preparation = movement.prepare().await; + // Simulate a retained older-format Inspect response. It must stay readable + // as history, but the host cannot dispatch it as a fresh observation. + let old_action = movement.action(MovementAction::Inspect); + let accepted = match movement + .source + .journal + .accept_action( + &old_action, + movement.spec.destination_node, + movement.spec.destination, + clock(), + ) + .await + .unwrap() + { + cellule_host::fleet::FleetActionAcceptance::New(accepted) => accepted, + _ => panic!("not original Inspect record"), + }; + let cached = FleetActionOutcome { + scope: scope(), + action_key: old_action.key().unwrap(), + node: movement.spec.destination_node, + session: movement.spec.destination, + observed_at_ms: clock(), + outcome: preparation.outcome.outcome.clone(), + }; + movement + .source + .journal + .publish_action_result(&accepted, &cached) + .await + .unwrap(); + assert!(matches!(cached.outcome, FleetOutcome::Reserved(_))); + movement.release().await; + let activated = apply( + &movement.receiver, + movement.action(MovementAction::Activate), + ) + .await; + let FleetOutcome::Activated(original) = &activated.outcome.outcome else { + panic!("{activated:?}") + }; + movement.event(AttemptEvent::Activated(original.clone())); + let request = movement.inspection(230); + let first = movement + .receiver + .inspect_fleet_action(request.clone()) + .await + .unwrap(); + first.validate_for(&request, clock(), 10_000).unwrap(); + let FleetOutcome::Activated(serving) = &first.outcome().outcome else { + panic!("{first:?}") + }; + assert_eq!(serving.position, original.position); + let current = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + let handle = movement + .receiver + .runtime() + .local_handle(movement.inputs.catalog.clone(), ¤t) + .await + .unwrap() + .unwrap(); + let now = clock(); + handle + .execute( + MutationIdentity { + request_id: RequestId::from_bytes([231; 16]), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }, + Digest::from_bytes([231; 32]), + now, + 64, + 64, + |transaction| { + transaction.execute("UPDATE counter SET value=101", [])?; + Ok(HandlerOutcome::Success(vec![101])) + }, + ) + .await + .unwrap(); + let next = movement.inspection(232); + let second = movement + .receiver + .inspect_fleet_action(next.clone()) + .await + .unwrap(); + second.validate_for(&next, clock(), 10_000).unwrap(); + assert!(first.validate_for(&next, clock(), 10_000).is_err()); + let FleetOutcome::Activated(newer) = &second.outcome().outcome else { + panic!("{second:?}") + }; + assert!(newer.position.root.commit_sequence > serving.position.root.commit_sequence); + assert_eq!(counter(&handle).await, 101); + let retained = movement + .source + .journal + .load_movement_action( + scope(), + movement.spec.id, + MovementAction::Inspect, + movement.spec.destination_node, + movement.spec.destination, + ) + .await + .unwrap() + .unwrap(); + assert!( + matches!(retained, cellule_host::fleet::FleetActionAcceptance::Existing { result: Some(result), .. } if *result == cached) + ); + assert!( + movement + .receiver + .apply_fleet_action(movement.action(MovementAction::Inspect), clock()) + .await + .is_err() + ); + handle.drain().await.unwrap(); + let stopped = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(stopped.value().state, ControlState::Idle); + let error = movement + .receiver + .inspect_fleet_action(movement.inspection(233)) + .await + .unwrap_err(); + assert!(matches!( + error.as_ref(), + Error::Fenced | Error::CellDraining + )); + assert_eq!(movement.receiver.stats().active_cells(), 0); + assert_eq!( + movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap() + .value(), + stopped.value() + ); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn fresh_recovery_inspection_never_retries_acquisition_with_only_retained_input() { + let movement = Movement::new(128 << 20).await; + movement.start_recovery(false).await; + movement + .source + .journal + .lose_basis_reply + .store(true, Ordering::SeqCst); + let recover = movement.action(MovementAction::Recover); + let failed = apply(&movement.receiver, recover.clone()).await; + assert!(matches!(failed.outcome.outcome, FleetOutcome::Unknown)); + assert!(failed.execution_error.is_some()); + let before = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + let writes = movement.source.journal.basis_writes.load(Ordering::SeqCst); + let request = movement.inspection(234); + let observed = movement + .receiver + .inspect_fleet_action(request.clone()) + .await + .unwrap(); + observed.validate_for(&request, clock(), 10_000).unwrap(); + assert!(matches!(observed.outcome().outcome, FleetOutcome::Unknown)); + assert_eq!( + movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap() + .value(), + before.value() + ); + assert_eq!( + movement.source.journal.basis_writes.load(Ordering::SeqCst), + writes + ); + assert_eq!(movement.receiver.stats().active_cells(), 0); + // Explicit replay of the accepted effect remains the only resumption path. + let resumed = apply(&movement.receiver, recover).await; + assert!(matches!( + resumed.outcome.outcome, + FleetOutcome::Recovered(_) + )); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn dropped_inspection_waiters_share_action_bound_and_are_joined_by_shutdown() { + let movement = Movement::new(128 << 20).await; + let journal = &movement.source.journal; + journal.block_inspections.store(true, Ordering::SeqCst); + let node = movement.receiver.clone(); + let a = movement.inspection(235); + let first = tokio::spawn(async move { node.inspect_fleet_action(a).await }); + tokio::time::timeout( + Duration::from_secs(5), + journal.inspection_entered.notified(), + ) + .await + .unwrap(); + let node = movement.receiver.clone(); + let b = movement.inspection(236); + let second = tokio::spawn(async move { node.inspect_fleet_action(b).await }); + tokio::time::timeout( + Duration::from_secs(5), + journal.inspection_entered.notified(), + ) + .await + .unwrap(); + let third = movement + .receiver + .inspect_fleet_action(movement.inspection(237)) + .await + .unwrap_err(); + assert!(matches!( + third.as_ref(), + Error::Capacity("fleet action receipt bound") + )); + first.abort(); + second.abort(); + assert!(movement.receiver.stats().retained_bytes() >= 6 * MAX_RECORD_BYTES as usize); + let node = movement.receiver.clone(); + let shutdown = tokio::spawn(async move { node.shutdown().await }); + tokio::task::yield_now().await; + assert!(!shutdown.is_finished()); + journal.block_inspections.store(false, Ordering::SeqCst); + journal.inspection_resume.add_permits(2); + shutdown.await.unwrap().unwrap(); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn admitted_movement_preserves_receipt_and_inspection_after_local_receipt_retirement() { + let movement = Movement::new(128 << 20).await; + let now = clock(); + let identity = MutationIdentity { + request_id: RequestId::from_bytes([211; 16]), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }; + let request_digest = Digest::from_bytes([211; 32]); + let acknowledged = movement + .source + .handle + .execute(identity, request_digest, now, 64, 64, |transaction| { + transaction.execute("UPDATE counter SET value = 77", [])?; + Ok(HandlerOutcome::Success(vec![77])) + }) + .await + .unwrap(); + // Durable acknowledgement can precede the actor joining publication and + // refreshing demand. Idle movement requires settled native observations; + // otherwise BusyExecution is a valid, permanently retained action refusal. + tokio::time::timeout(Duration::from_secs(5), async { + loop { + let page = movement + .source + .node + .runtime() + .fleet_cells_page(None, 128) + .await + .unwrap(); + if let Some(CellInventoryEntry::Owned(owner)) = page.entries().first() + && owner.target == movement.spec.target + && owner.generation == movement.spec.generation + && owner.incarnation == movement.spec.incarnation + && owner.stable_observations == 2 + && owner.cost.is_some() + && owner.blockers.is_empty() + && owner.position.as_ref().is_some_and(|position| { + position.epoch == movement.spec.source_epoch + && position.root.commit_sequence >= acknowledged.commit_sequence() + }) + { + break; + } + drop(page); + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + let released = movement.release().await; + assert!(released.root.commit_sequence >= acknowledged.commit_sequence()); + assert_eq!(movement.source.node.stats().active_cells(), 0); + let activation = movement.action(MovementAction::Activate); + let (a, b) = tokio::join!( + apply(&movement.receiver, activation.clone()), + apply(&movement.receiver, activation.clone()) + ); + assert!(a.committed && b.committed && a.execution_error.is_none()); + assert_eq!(a.outcome, b.outcome); + let FleetOutcome::Activated(evidence) = &a.outcome.outcome else { + panic!("not activated: {a:?}") + }; + assert_eq!(evidence.session, movement.spec.destination); + assert_eq!(evidence.position.root, released.root); + let basis = movement + .source + .journal + .basis(activation.key().unwrap()) + .unwrap(); + assert_eq!(basis.position().unwrap(), released); + basis.validate_result(&a.outcome).unwrap(); + assert_eq!(movement.source.journal.accepted_count(), 3); + let serving = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + let handle = movement + .receiver + .runtime() + .local_handle(movement.inputs.catalog.clone(), &serving) + .await + .unwrap() + .unwrap(); + assert_eq!(counter(&handle).await, 77); + assert_eq!( + handle + .resolve(identity, request_digest, clock(), 64) + .await + .unwrap(), + Resolution::Committed(acknowledged) + ); + movement.event(AttemptEvent::Activated(evidence.clone())); + let cleanup = apply(&movement.receiver, movement.action(MovementAction::Cancel)).await; + assert!(cleanup.committed && cleanup.execution_error.is_none()); + assert!(matches!( + cleanup.outcome.outcome, + FleetOutcome::ReceiverCleaned + )); + movement.event(AttemptEvent::ReceiverCleaned); + let inspection = movement.inspect(250).await; + assert!(matches!( + inspection.outcome().outcome, + FleetOutcome::Activated(_) + )); + assert!( + movement + .receiver + .runtime() + .prepared_receiver(movement.spec.id) + .unwrap() + .is_none() + ); + assert_eq!(movement.receiver.stats().active_cells(), 1); + assert_eq!(movement.receiver.stats().worker_jobs(), 0); + assert!(movement.receiver.stats().local_disk_reserved_bytes() < movement.spec.cost.disk_bytes); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn prepare_refusal_keeps_source_serving_without_partial_receiver_credit() { + let movement = Movement::new(64 << 10).await; + let before = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + let result = movement.prepare().await; + assert!(result.committed && result.execution_error.is_some()); + assert!(matches!( + result.outcome.outcome, + FleetOutcome::Rejected(DrainBlocker::ReceiverCapacity) + )); + assert!( + movement + .receiver + .runtime() + .prepared_receiver(movement.spec.id) + .unwrap() + .is_none() + ); + let stats = movement.receiver.stats(); + assert_eq!(stats.active_cells(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.file_descriptors(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); + let current = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(current.value(), before.value()); + assert_eq!(counter(&movement.source.handle).await, 42); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn duplicate_prepare_and_cancel_return_original_evidence_without_double_reserving() { + let movement = Movement::new(128 << 20).await; + let action = movement.action(MovementAction::Prepare); + let (a, b) = tokio::join!( + apply(&movement.receiver, action.clone()), + apply(&movement.receiver, action) + ); + assert!(a.committed && b.committed); + assert_eq!(a.outcome, b.outcome); + assert_eq!(movement.source.journal.accepted_count(), 1); + assert_eq!(movement.receiver.stats().active_cells(), 1); + assert_eq!(movement.receiver.stats().worker_jobs(), 1); + assert_eq!( + movement.receiver.stats().local_disk_reserved_bytes(), + movement.spec.cost.disk_bytes + ); + movement.event(AttemptEvent::BeginCancel); + let cancel = movement.action(MovementAction::Cancel); + let first = apply(&movement.receiver, cancel.clone()).await; + assert!(first.committed && matches!(first.outcome.outcome, FleetOutcome::ReceiverCleaned)); + let duplicate = apply(&movement.receiver, cancel).await; + assert_eq!(first.outcome, duplicate.outcome); + assert_eq!(movement.receiver.stats().active_cells(), 0); + assert_eq!(movement.receiver.stats().worker_jobs(), 0); + assert_eq!(movement.receiver.stats().local_disk_reserved_bytes(), 0); + movement.event(AttemptEvent::Cancelled); + let inspected = movement.inspect(250).await; + assert!(matches!( + inspected.outcome().outcome, + FleetOutcome::ReceiverCleaned + )); + assert_eq!(counter(&movement.source.handle).await, 42); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn lost_basis_reply_prevents_takeover_until_the_original_basis_is_confirmed() { + let movement = Movement::new(128 << 20).await; + let released = movement.release().await; + movement + .source + .journal + .lose_basis_reply + .store(true, Ordering::SeqCst); + let action = movement.action(MovementAction::Activate); + let first = apply(&movement.receiver, action.clone()).await; + assert!(first.committed && first.execution_error.is_some()); + assert!(matches!(first.outcome.outcome, FleetOutcome::Unknown)); + let retained = movement + .source + .journal + .basis(action.key().unwrap()) + .unwrap(); + assert_eq!(retained.position().unwrap(), released); + let idle = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(idle.value(), retained.control()); + assert_eq!(idle.value().state, ControlState::Idle); + assert_eq!( + movement + .receiver + .runtime() + .prepared_receiver(movement.spec.id) + .unwrap() + .unwrap() + .state() + .unwrap(), + ReceiverState::Prepared + ); + let activated = apply(&movement.receiver, action.clone()).await; + assert!(activated.committed && activated.execution_error.is_none()); + assert!(matches!( + activated.outcome.outcome, + FleetOutcome::Activated(_) + )); + assert_eq!( + movement + .source + .journal + .basis(action.key().unwrap()) + .unwrap(), + retained + ); + assert_eq!( + movement.source.journal.basis_writes.load(Ordering::SeqCst), + 1 + ); + assert_eq!(movement.source.journal.accepted_count(), 3); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn dropped_waiter_cannot_cancel_a_basis_write_or_the_owned_activation() { + let movement = Movement::new(128 << 20).await; + movement.release().await; + movement + .source + .journal + .block_basis + .store(true, Ordering::SeqCst); + let node = movement.receiver.clone(); + let action = movement.action(MovementAction::Activate); + let issued = action.clone(); + let waiter = tokio::spawn(async move { node.apply_fleet_action(issued, clock()).await }); + tokio::time::timeout( + Duration::from_secs(5), + movement.source.journal.basis_entered.notified(), + ) + .await + .unwrap(); + assert!( + movement + .source + .journal + .basis(action.key().unwrap()) + .is_none() + ); + let idle = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(idle.value().state, ControlState::Idle); + waiter.abort(); + assert!(waiter.await.unwrap_err().is_cancelled()); + assert_eq!( + movement + .receiver + .runtime() + .prepared_receiver(movement.spec.id) + .unwrap() + .unwrap() + .state() + .unwrap(), + ReceiverState::Prepared + ); + movement.source.journal.basis_resume.add_permits(1); + let result = apply(&movement.receiver, action).await; + assert!(result.committed && result.execution_error.is_none()); + assert!(matches!(result.outcome.outcome, FleetOutcome::Activated(_))); + assert_eq!( + movement.source.journal.basis_writes.load(Ordering::SeqCst), + 1 + ); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn expired_unused_credit_is_cleaned_after_release_then_canonical_cold_activation_resumes() { + let movement = Movement::with_deadline(128 << 20, 1_000).await; + let released = movement.release().await; + while clock() < movement.spec.deadline_ms { + tokio::time::sleep(Duration::from_millis(5)).await; + } + assert_eq!( + movement.receiver.stats().local_disk_reserved_bytes(), + movement.spec.cost.disk_bytes + ); + movement.event(AttemptEvent::BeginCancel); + let cleanup = apply(&movement.receiver, movement.action(MovementAction::Cancel)).await; + assert!(cleanup.committed && cleanup.execution_error.is_none()); + assert!(matches!( + cleanup.outcome.outcome, + FleetOutcome::ReceiverCleaned + )); + assert_eq!(movement.receiver.stats().active_cells(), 0); + assert_eq!(movement.receiver.stats().worker_jobs(), 0); + assert_eq!(movement.receiver.stats().local_disk_reserved_bytes(), 0); + movement.event(AttemptEvent::ReceiverCleaned); + movement.event(AttemptEvent::BeginActivate); + let activated = apply( + &movement.receiver, + movement.action(MovementAction::Activate), + ) + .await; + assert!( + activated.committed && activated.execution_error.is_none(), + "{activated:?}" + ); + let FleetOutcome::Activated(evidence) = &activated.outcome.outcome else { + panic!("not activated: {activated:?}") + }; + assert_eq!(evidence.position.root, released.root); + assert_eq!(movement.receiver.stats().active_cells(), 1); + let serving = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + let handle = movement + .receiver + .runtime() + .local_handle(movement.inputs.catalog.clone(), &serving) + .await + .unwrap() + .unwrap(); + assert_eq!(counter(&handle).await, 42); + movement.event(AttemptEvent::Activated(evidence.clone())); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn lost_cleanup_reply_after_release_retains_the_original_resource_evidence() { + let movement = Movement::new(128 << 20).await; + let released = movement.release().await; + movement.event(AttemptEvent::BeginCancel); + movement.source.journal.lose_next_result_reply(); + let action = movement.action(MovementAction::Cancel); + let first = apply(&movement.receiver, action.clone()).await; + assert!(!first.committed && first.journal_error.is_some()); + assert!(matches!( + first.outcome.outcome, + FleetOutcome::ReceiverCleaned + )); + assert_eq!(movement.receiver.stats().active_cells(), 0); + assert_eq!(movement.receiver.stats().worker_jobs(), 0); + let confirmed = apply(&movement.receiver, action).await; + assert!(confirmed.committed && confirmed.execution_error.is_none()); + assert_eq!(first.outcome, confirmed.outcome); + let idle = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(idle.value().state, ControlState::Idle); + assert_eq!(idle.value().root.as_ref(), Some(&released.root)); + movement.event(AttemptEvent::ReceiverCleaned); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn receiver_result_retry_preserves_basis_and_original_root_after_successor_publication() { + let movement = Movement::new(128 << 20).await; + let released = movement.release().await; + movement.source.journal.lose_next_result_reply(); + let action = movement.action(MovementAction::Activate); + let first = apply(&movement.receiver, action.clone()).await; + assert!(!first.committed && first.journal_error.is_some()); + let basis = movement + .source + .journal + .basis(action.key().unwrap()) + .unwrap(); + let serving = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + let handle = movement + .receiver + .runtime() + .local_handle(movement.inputs.catalog.clone(), &serving) + .await + .unwrap() + .unwrap(); + let now = clock(); + handle + .execute( + MutationIdentity { + request_id: RequestId::from_bytes([212; 16]), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }, + Digest::from_bytes([212; 32]), + now, + 64, + 64, + |transaction| { + transaction.execute("UPDATE counter SET value = 99", [])?; + Ok(HandlerOutcome::Success(vec![99])) + }, + ) + .await + .unwrap(); + let confirmed = apply(&movement.receiver, action.clone()).await; + assert!(confirmed.committed && confirmed.execution_error.is_none()); + assert_eq!(confirmed.outcome, first.outcome); + assert_eq!( + movement + .source + .journal + .basis(action.key().unwrap()) + .unwrap(), + basis + ); + assert_eq!(basis.position().unwrap(), released); + let FleetOutcome::Activated(evidence) = &confirmed.outcome.outcome else { + panic!() + }; + movement.event(AttemptEvent::Activated(evidence.clone())); + let inspected = movement.inspect(250).await; + let FleetOutcome::Activated(current) = &inspected.outcome().outcome else { + panic!("not serving: {inspected:?}") + }; + assert!(current.position.root.commit_sequence > evidence.position.root.commit_sequence); + assert_eq!(counter(&handle).await, 99); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn ordinary_acquisition_on_the_preferred_session_joins_after_unused_credit_cleanup() { + let movement = Movement::new(128 << 20).await; + let released = movement.release().await; + let idle = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + let runtime = movement.receiver.runtime(); + let inputs = movement.inputs.clone(); + let cold = tokio::spawn(async move { + runtime + .acquire_idle_restored( + inputs.catalog, + inputs.replica, + inputs.authority, + idle, + inputs + .destination + .with_file_name("ordinary-receiver.sqlite"), + inputs.owner, + ) + .await + }); + tokio::time::timeout(Duration::from_secs(5), async { + loop { + let current = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + if current + .value() + .owner + .as_ref() + .is_some_and(|owner| owner.session == movement.spec.destination) + { + assert_eq!(current.value().root.as_ref(), Some(&released.root)); + break; + } + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + // Ordinary takeover owns authority while its affine SQL open waits for the + // unused prepared token. Cleanup frees that token, not the new writer. + assert!(!cold.is_finished()); + assert_eq!( + movement + .receiver + .runtime() + .prepared_receiver(movement.spec.id) + .unwrap() + .unwrap() + .state() + .unwrap(), + ReceiverState::Prepared + ); + movement.event(AttemptEvent::BeginCancel); + let cleanup = apply(&movement.receiver, movement.action(MovementAction::Cancel)).await; + assert!(cleanup.committed && cleanup.execution_error.is_none()); + assert!(matches!( + cleanup.outcome.outcome, + FleetOutcome::ReceiverCleaned + )); + movement.event(AttemptEvent::ReceiverCleaned); + let handle = tokio::time::timeout(Duration::from_secs(5), cold) + .await + .unwrap() + .unwrap() + .unwrap(); + assert_eq!(counter(&handle).await, 42); + movement.event(AttemptEvent::BeginActivate); + let action = movement.action(MovementAction::Activate); + let result = apply(&movement.receiver, action.clone()).await; + assert!(result.committed && result.execution_error.is_none()); + let FleetOutcome::Activated(evidence) = &result.outcome.outcome else { + panic!("not serving: {result:?}") + }; + assert_eq!(evidence.position.epoch, released.epoch + 1); + // This action observed the ordinary winner; it performed no acquisition CAS. + assert!( + movement + .source + .journal + .basis(action.key().unwrap()) + .is_none() + ); + movement.event(AttemptEvent::Activated(evidence.clone())); + let inspected = movement.inspect(250).await; + assert!(matches!( + inspected.outcome().outcome, + FleetOutcome::Activated(_) + )); + assert_eq!(movement.receiver.stats().active_cells(), 1); + assert_eq!(movement.receiver.stats().worker_jobs(), 0); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn action_task_failure_retains_its_original_error_across_repeated_shutdown() { + let movement = Movement::new(128 << 20).await; + movement.release().await; + movement + .source + .journal + .panic_basis + .store(true, Ordering::SeqCst); + let failed = movement + .receiver + .apply_fleet_action(movement.action(MovementAction::Activate), clock()) + .await + .unwrap_err(); + let original = original_task_failure(failed.as_ref()); + assert!(original.is_panic()); + assert!( + original + .to_string() + .contains("injected acquisition-basis panic") + ); + let first = movement.receiver.shutdown().await.unwrap_err(); + let second = movement.receiver.shutdown().await.unwrap_err(); + let first_source = original_task_failure(&first); + let second_source = original_task_failure(&second); + assert_eq!(original.id(), first_source.id()); + assert_eq!(first_source.id(), second_source.id()); + assert_eq!(original.to_string(), first_source.to_string()); + assert_eq!(first_source.to_string(), second_source.to_string()); + assert_ne!(movement.receiver.state(), NodeState::Stopped); + let idle = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(idle.value().state, ControlState::Idle); + // The host still joined canonical runtime cleanup after the facility failed. + // Retained fatal evidence prevents a later drain from claiming success. + assert_eq!(movement.receiver.stats().active_cells(), 0); + assert_eq!(movement.receiver.stats().worker_jobs(), 0); + assert_eq!(movement.receiver.stats().retained_bytes(), 0); + assert_eq!(movement.receiver.stats().file_descriptors(), 0); + assert_eq!(movement.receiver.stats().local_disk_reserved_bytes(), 0); + assert!(movement.receiver.shutdown().await.is_err()); + assert_ne!(movement.receiver.state(), NodeState::Stopped); + movement.source.node.shutdown().await.unwrap(); +} + +fn original_task_failure<'a>( + mut error: &'a (dyn std::error::Error + 'static), +) -> &'a tokio::task::JoinError { + loop { + if let Some(error) = error.downcast_ref::() { + return error; + } + error = error + .source() + .expect("original task source must be preserved"); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn fatal_action_drain_joins_a_sibling_inspection_before_returning_original_failure() { + let movement = Movement::new(128 << 20).await; + movement.release().await; + let journal = &movement.source.journal; + journal.block_basis.store(true, Ordering::SeqCst); + let node = movement.receiver.clone(); + let action = movement.action(MovementAction::Activate); + let activation = tokio::spawn(async move { node.apply_fleet_action(action, clock()).await }); + tokio::time::timeout(Duration::from_secs(5), journal.basis_entered.notified()) + .await + .unwrap(); + journal.block_inspections.store(true, Ordering::SeqCst); + let node = movement.receiver.clone(); + let request = movement.inspection(238); + let inspection = tokio::spawn(async move { node.inspect_fleet_action(request).await }); + tokio::time::timeout( + Duration::from_secs(5), + journal.inspection_entered.notified(), + ) + .await + .unwrap(); + // Inject the failure at the same before-claim boundary after both finite + // jobs are accepted. The sibling remains deliberately paused in its adapter. + journal.panic_basis.store(true, Ordering::SeqCst); + journal.basis_resume.add_permits(1); + let failed = activation.await.unwrap().unwrap_err(); + let original = original_task_failure(failed.as_ref()).id(); + inspection.abort(); + let node = movement.receiver.clone(); + let mut shutdown = tokio::spawn(async move { node.shutdown().await }); + assert!( + tokio::time::timeout(Duration::from_secs(1), &mut shutdown) + .await + .is_err() + ); + assert!(!shutdown.is_finished()); + journal.block_inspections.store(false, Ordering::SeqCst); + journal.inspection_resume.add_permits(1); + let failure = shutdown.await.unwrap().unwrap_err(); + assert_eq!(original_task_failure(&failure).id(), original); + let repeated = movement.receiver.shutdown().await.unwrap_err(); + assert_eq!(original_task_failure(&repeated).id(), original); + assert_ne!(movement.receiver.state(), NodeState::Stopped); + let stats = movement.receiver.stats(); + assert_eq!(stats.active_cells(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.retained_bytes(), 0); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.file_descriptors(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); + movement.source.node.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn failed_source_recovery_preserves_receipt_and_immutable_basis_after_root_advances() { + let movement = Movement::new(128 << 20).await; + let now = clock(); + let identity = MutationIdentity { + request_id: RequestId::from_bytes([231; 16]), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }; + let digest = Digest::from_bytes([231; 32]); + let receipt = movement + .source + .handle + .execute(identity, digest, now, 64, 64, |tx| { + tx.execute("UPDATE counter SET value = 77", [])?; + Ok(HandlerOutcome::Success(vec![77])) + }) + .await + .unwrap(); + movement.start_recovery(false).await; + let source_control = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(source_control.value().state, ControlState::Serving); + movement.source.journal.lose_next_result_reply(); + let action = movement.action(MovementAction::Recover); + let recovered = apply(&movement.receiver, action.clone()).await; + assert!(!recovered.committed && recovered.execution_error.is_none()); + let FleetOutcome::Recovered(e) = &recovered.outcome.outcome else { + panic!("not recovered: {recovered:?}") + }; + assert_eq!(e.recovery.basis().control(), source_control.value()); + assert_eq!( + e.recovery.position().unwrap().root, + source_control.value().root.clone().unwrap() + ); + let original = movement + .source + .journal + .recovery_evidence(action.key().unwrap()) + .unwrap(); + let current = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + let handle = movement + .receiver + .runtime() + .local_handle(movement.inputs.catalog.clone(), ¤t) + .await + .unwrap() + .unwrap(); + assert_eq!(counter(&handle).await, 77); + assert_eq!( + handle.resolve(identity, digest, clock(), 64).await.unwrap(), + Resolution::Committed(receipt) + ); + let now = clock(); + handle + .execute( + MutationIdentity { + request_id: RequestId::from_bytes([232; 16]), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }, + Digest::from_bytes([232; 32]), + now, + 64, + 64, + |tx| { + tx.execute("UPDATE counter SET value = 99", [])?; + Ok(HandlerOutcome::Success(vec![99])) + }, + ) + .await + .unwrap(); + let republished = apply(&movement.receiver, action).await; + assert!(republished.committed); + assert_eq!(republished.outcome, recovered.outcome); + assert_eq!( + movement + .source + .journal + .recovery_evidence(republished.outcome.action_key) + .unwrap(), + original + ); + movement.event(AttemptEvent::Recovered(e.clone())); + let history = movement.source.journal.current_attempt(); + assert_eq!(history.phase(), AttemptPhase::Recovered); + assert!(history.released().is_none() && history.activated().is_none()); + let cleaned = apply(&movement.receiver, movement.action(MovementAction::Cancel)).await; + assert!(cleaned.committed && cleaned.execution_error.is_none()); + assert!(matches!( + cleaned.outcome.outcome, + FleetOutcome::ReceiverCleaned + )); + movement.event(AttemptEvent::ReceiverCleaned); + let inspected = movement.inspect(250).await; + let FleetOutcome::Recovered(fresh) = &inspected.outcome().outcome else { + panic!("not recovered: {inspected:?}") + }; + assert_eq!(fresh.recovery, original); + assert!( + fresh.serving.position.root.commit_sequence + > fresh.recovery.position().unwrap().root.commit_sequence + ); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn unobserved_release_cas_recovers_idle_input_without_fabricating_clean_release() { + let movement = Movement::new(128 << 20).await; + movement.start_recovery(true).await; + let result = apply(&movement.receiver, movement.action(MovementAction::Recover)).await; + assert!(result.committed && result.execution_error.is_none()); + let FleetOutcome::Recovered(evidence) = &result.outcome.outcome else { + panic!("not recovered: {result:?}") + }; + assert_eq!( + evidence.recovery.basis().control().state, + ControlState::Idle + ); + assert_eq!( + evidence.recovery.basis().control().epoch, + movement.spec.source_epoch + ); + movement.event(AttemptEvent::Recovered(evidence.clone())); + assert!( + movement + .source + .journal + .current_attempt() + .released() + .is_none() + ); + assert_eq!(movement.receiver.stats().active_cells(), 1); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn lost_recovery_basis_reply_retains_original_input_and_prevents_cas_until_confirmation() { + let movement = Movement::new(128 << 20).await; + movement.start_recovery(false).await; + movement + .source + .journal + .lose_basis_reply + .store(true, Ordering::SeqCst); + let action = movement.action(MovementAction::Recover); + let unknown = apply(&movement.receiver, action.clone()).await; + assert!(unknown.committed && unknown.execution_error.is_some()); + assert!(matches!(unknown.outcome.outcome, FleetOutcome::Unknown)); + let original = movement + .source + .journal + .recovery_basis(action.key().unwrap()) + .unwrap(); + assert_eq!( + movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap() + .value(), + original.control() + ); + assert_eq!(movement.receiver.stats().active_cells(), 0); + assert_eq!(movement.receiver.stats().worker_jobs(), 0); + let recovered = apply(&movement.receiver, action.clone()).await; + assert!(recovered.committed && recovered.execution_error.is_none()); + assert!(matches!( + recovered.outcome.outcome, + FleetOutcome::Recovered(_) + )); + assert_eq!( + movement + .source + .journal + .recovery_basis(action.key().unwrap()) + .unwrap(), + original + ); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn lost_recovery_evidence_reply_cannot_admit_actor_or_claim_completion() { + let movement = Movement::new(128 << 20).await; + movement.start_recovery(false).await; + movement + .source + .journal + .lose_recovery_evidence_reply + .store(true, Ordering::SeqCst); + let action = movement.action(MovementAction::Recover); + let failed = apply(&movement.receiver, action.clone()).await; + assert!(failed.committed && failed.execution_error.is_some()); + assert!(matches!(failed.outcome.outcome, FleetOutcome::Unknown)); + let evidence = movement + .source + .journal + .recovery_evidence(action.key().unwrap()) + .unwrap(); + let current = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(current.value().state, ControlState::Idle); + assert_eq!( + current.value().root.as_ref(), + evidence.restored().root.as_ref() + ); + assert_eq!(current.value().epoch, evidence.restored().epoch); + assert_eq!(movement.receiver.stats().active_cells(), 0); + assert_eq!(movement.receiver.stats().worker_jobs(), 0); + let repeated = apply(&movement.receiver, action.clone()).await; + assert!(matches!(repeated.outcome.outcome, FleetOutcome::Unknown)); + assert!(repeated.execution_error.is_some()); + assert_eq!( + movement.source.journal.current_attempt().phase(), + AttemptPhase::Recovering + ); + assert_eq!( + movement + .source + .journal + .current_attempt() + .spec() + .cost + .disk_bytes, + movement.spec.cost.disk_bytes + ); + // Ordinary canonical recovery can acquire the safe Idle root after rollback. + // Fleet inspection then proves current serving against the retained original + // recovery result, without fabricating a second clean source release. + let ordinary = movement + .receiver + .runtime() + .acquire_idle_restored( + movement.inputs.catalog.clone(), + movement.inputs.replica.clone(), + movement.inputs.authority.clone(), + current, + movement.inputs.destination.clone(), + movement.inputs.owner.clone(), + ) + .await + .unwrap(); + assert_eq!(counter(&ordinary).await, 42); + let finished = apply(&movement.receiver, action).await; + assert!(finished.committed && finished.execution_error.is_none()); + let FleetOutcome::Recovered(finished_evidence) = &finished.outcome.outcome else { + panic!("not recovered: {finished:?}") + }; + assert_eq!(finished_evidence.recovery, evidence); + assert!(finished_evidence.serving.position.epoch > evidence.restored().epoch); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn dropped_recovery_waiter_leaves_owned_basis_recording_and_takeover_running() { + let movement = Movement::new(128 << 20).await; + movement.start_recovery(false).await; + movement + .source + .journal + .block_basis + .store(true, Ordering::SeqCst); + let node = movement.receiver.clone(); + let action = movement.action(MovementAction::Recover); + let dispatched = action.clone(); + let waiter = tokio::spawn(async move { node.apply_fleet_action(dispatched, clock()).await }); + tokio::time::timeout( + Duration::from_secs(5), + movement.source.journal.basis_entered.notified(), + ) + .await + .unwrap(); + assert!( + movement + .source + .journal + .recovery_basis(action.key().unwrap()) + .is_none() + ); + assert_eq!( + movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .epoch, + movement.spec.source_epoch + ); + waiter.abort(); + assert!(waiter.await.unwrap_err().is_cancelled()); + movement.source.journal.basis_resume.add_permits(1); + let result = apply(&movement.receiver, action).await; + assert!(result.committed && result.execution_error.is_none()); + assert!(matches!(result.outcome.outcome, FleetOutcome::Recovered(_))); + movement.shutdown().await; +} diff --git a/crates/cellule-host/tests/node/fleet_receivers/prefix.rs b/crates/cellule-host/tests/node/fleet_receivers/prefix.rs new file mode 100644 index 00000000..7c600625 --- /dev/null +++ b/crates/cellule-host/tests/node/fleet_receivers/prefix.rs @@ -0,0 +1,191 @@ +//! Fresh native inspection still requires preparation lineage and origin bytes. +use super::*; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn fresh_serving_refuses_missing_lineage_after_successor_publication() { + inspect_missing_evidence(false).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn fresh_serving_refuses_missing_origin_root_despite_a_native_actor() { + inspect_missing_evidence(true).await; +} + +async fn inspect_missing_evidence(origin: bool) { + let movement = Movement::new(128 << 20).await; + movement.release().await; + let activated = apply( + &movement.receiver, + movement.action(MovementAction::Activate), + ) + .await; + assert!(activated.committed && activated.execution_error.is_none()); + let FleetOutcome::Activated(evidence) = &activated.outcome.outcome else { + panic!("not activated") + }; + movement.event(AttemptEvent::Activated(evidence.clone())); + let current = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + let handle = movement + .receiver + .runtime() + .local_handle(movement.inputs.catalog.clone(), ¤t) + .await + .unwrap() + .unwrap(); + let now = clock(); + handle + .execute( + MutationIdentity { + request_id: RequestId::from_bytes([213; 16]), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }, + Digest::from_bytes([213; 32]), + now, + 64, + 64, + |transaction| { + transaction.execute("UPDATE counter SET value = 99", [])?; + Ok(HandlerOutcome::Success(vec![99])) + }, + ) + .await + .unwrap(); + let advanced = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .unwrap(); + movement.inspect(247).await; + let layout = movement.inputs.authority.layout(); + let path = if origin { + layout.incarnation_object_path( + &advanced.cell, + &advanced.incarnation, + &advanced.digest, + cellule_runtime::ltx::CellObjectKind::Root, + ) + } else { + layout.root_lineage_path(&advanced.cell, &advanced.incarnation, &advanced.digest) + }; + let (original, _) = layout.store().get_with_etag(&path).await.unwrap(); + layout.store().delete(&path).await.unwrap(); + assert_eq!(counter(&handle).await, 99); + let accepted = movement.source.journal.accepted_count(); + let error = movement + .receiver + .inspect_fleet_action(movement.inspection(248)) + .await + .unwrap_err(); + if origin { + assert!(matches!(error.as_ref(), Error::Ltx(_))); + } else { + assert!( + matches!(error.as_ref(), Error::RootLineageIncomplete { root } if *root == advanced) + ); + } + assert_eq!(movement.source.journal.accepted_count(), accepted); + assert_eq!( + movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .ltx_root(), + Some(advanced) + ); + // Restore the exact original bytes, never fabricate historical inputs or + // rerun acquisition. A new nonce must inspect current evidence again. + layout.store().create_strict(&path, original).await.unwrap(); + let restored = movement.inspect(249).await; + assert!(matches!( + restored.outcome().outcome, + FleetOutcome::Activated(_) + )); + movement.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn fresh_recovery_requires_canonical_acquisition_even_with_a_live_actor() { + inspect_acquisition(false).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn corrupt_acquisition_cannot_certify_an_idle_source_recovery() { + inspect_acquisition(true).await; +} + +async fn inspect_acquisition(corrupt: bool) { + let movement = Movement::new(128 << 20).await; + movement.start_recovery(corrupt).await; + let action = movement.action(MovementAction::Recover); + let result = apply(&movement.receiver, action.clone()).await; + assert!(result.committed && result.execution_error.is_none()); + let FleetOutcome::Recovered(recovered) = &result.outcome.outcome else { + panic!("not recovered") + }; + movement.event(AttemptEvent::Recovered(recovered.clone())); + let cleaned = apply(&movement.receiver, movement.action(MovementAction::Cancel)).await; + assert!(matches!( + cleaned.outcome.outcome, + FleetOutcome::ReceiverCleaned + )); + movement.event(AttemptEvent::ReceiverCleaned); + movement.inspect(241).await; + let restored = recovered.recovery.restored(); + let layout = movement.inputs.authority.layout(); + let path = layout.acquisition_record_path( + restored.cell.as_bytes(), + restored.incarnation.as_bytes(), + restored.epoch, + ); + let (original, _) = layout.store().get_with_etag(&path).await.unwrap(); + layout.store().delete(&path).await.unwrap(); + if corrupt { + layout + .store() + .create_strict(&path, bytes::Bytes::from_static(b"corrupt-acquisition")) + .await + .unwrap(); + } + // Historical replay preserves its original committed outcome; it cannot + // replace the independent fresh native inspection or restore missing proof. + assert_eq!( + apply(&movement.receiver, action).await.outcome, + result.outcome + ); + let error = movement + .receiver + .inspect_fleet_action(movement.inspection(242)) + .await + .unwrap_err(); + if corrupt { + assert!(matches!(error.as_ref(), Error::Control(_))); + layout.store().delete(&path).await.unwrap(); + } else { + assert!( + matches!(error.as_ref(), Error::AcquisitionHistoryIncomplete { epoch, .. } + if *epoch==restored.epoch) + ); + } + layout.store().create_strict(&path, original).await.unwrap(); + assert!(matches!( + movement.inspect(243).await.outcome().outcome, + FleetOutcome::Recovered(_) + )); + movement.shutdown().await; +} diff --git a/crates/cellule-host/tests/node/fleet_receivers/successor.rs b/crates/cellule-host/tests/node/fleet_receivers/successor.rs new file mode 100644 index 00000000..5a6deaee --- /dev/null +++ b/crates/cellule-host/tests/node/fleet_receivers/successor.rs @@ -0,0 +1,179 @@ +use super::*; +use cellule_runtime::identity::NodeId; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn actual_successor_inspection_does_not_acquire_or_settle_preferred_credit() { + inspect_successor(false).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn preferred_node_new_session_requires_its_own_read_only_successor_request() { + inspect_successor(true).await; +} + +async fn inspect_successor(preferred_physical_node: bool) { + let movement = Movement::new(128 << 20).await; + let released = movement.release().await; + let node_id = if preferred_physical_node { + movement.spec.destination_node + } else { + NodeId::from_bytes([203; 16]) + }; + let session = SessionId::from_bytes([205; 16]); + let mut inputs = movement.inputs.clone(); + inputs.destination = inputs.destination.with_file_name("actual-successor.sqlite"); + inputs.owner = Owner { + session, + endpoint: "https://actual-successor.internal:8789".into(), + }; + let node = CellNodeBuilder::new(application()) + .with_runtime( + SqlWorkerPool::new(1, 8) + .unwrap() + .with_native_memory_limit(128 << 20) + .unwrap(), + 64 << 20, + ) + .with_replica_host(ReplicaHost::default().with_local_disk_budget(DiskBudget::new(8 << 30))) + .with_session(session) + .build() + .unwrap(); + node.install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + node.install_fleet_actions( + scope(), + node_id, + movement.source.journal.clone(), + Arc::new(Cells(inputs.clone(), Arc::new(Mutex::new(None)))), + ) + .unwrap(); + node.install_node_lease(NodeLeaseGuard::new(clock(), clock() + 60_000).unwrap()) + .unwrap(); + let make_request = |nonce| { + FleetInspectionRequest::new( + movement.action(MovementAction::Inspect), + movement.source.journal.registry(), + Digest::from_bytes([nonce; 32]), + node_id, + session, + clock() + 10_000, + ) + .unwrap() + }; + // A known release permits a read, but Idle authority and no actor cannot + // manufacture serving. Inspection leaves the pinned authority unchanged. + let idle = inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert!(node.inspect_fleet_action(make_request(231)).await.is_err()); + assert_eq!( + inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap() + .value(), + idle.value() + ); + let handle = node + .runtime() + .acquire_idle_restored( + inputs.catalog.clone(), + inputs.replica.clone(), + inputs.authority.clone(), + idle, + inputs.destination.clone(), + inputs.owner.clone(), + ) + .await + .unwrap(); + assert_eq!(counter(&handle).await, 42); + let current = inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + let accepts = movement.source.journal.accepted_count(); + let charged = movement.receiver.stats().local_disk_reserved_bytes(); + assert!(charged > 0); + let request = make_request(232); + let observed = node.inspect_fleet_action(request.clone()).await.unwrap(); + observed.validate_for(&request, clock(), 10_000).unwrap(); + let FleetOutcome::Activated(evidence) = &observed.outcome().outcome else { + panic!("not actual serving: {observed:?}") + }; + assert_eq!((evidence.node, evidence.session), (node_id, session)); + assert_eq!(evidence.position.root, released.root); + assert_eq!(evidence.position.epoch, released.epoch + 1); + assert_eq!( + inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap() + .value(), + current.value() + ); + assert_eq!(movement.source.journal.accepted_count(), accepts); + assert_eq!( + movement.receiver.stats().local_disk_reserved_bytes(), + charged + ); + assert_eq!( + movement + .receiver + .runtime() + .prepared_receiver(movement.spec.id) + .unwrap() + .unwrap() + .state() + .unwrap(), + ReceiverState::Prepared + ); + assert!( + node.apply_fleet_action(movement.action(MovementAction::Activate), clock()) + .await + .is_err() + ); + assert_eq!(movement.source.journal.accepted_count(), accepts); + movement.event(AttemptEvent::Activated(evidence.clone())); + let attempt = movement.source.journal.current_attempt(); + assert_eq!(attempt.next_action(), MovementAction::Cancel); + assert!(!attempt.receiver_resources_settled()); + let cleanup = apply(&movement.receiver, movement.action(MovementAction::Cancel)).await; + assert!(cleanup.committed && cleanup.execution_error.is_none()); + assert!(matches!( + cleanup.outcome.outcome, + FleetOutcome::ReceiverCleaned + )); + movement.event(AttemptEvent::ReceiverCleaned); + assert_eq!( + movement.source.journal.current_attempt().next_action(), + MovementAction::Retire + ); + let fresh = make_request(233); + assert!(matches!( + node.inspect_fleet_action(fresh) + .await + .unwrap() + .outcome() + .outcome, + FleetOutcome::Activated(_) + )); + handle.drain().await.unwrap(); + assert!(node.inspect_fleet_action(make_request(234)).await.is_err()); + node.shutdown().await.unwrap(); + assert_eq!(node.stats().active_cells(), 0); + assert_eq!(node.stats().worker_jobs(), 0); + assert_eq!(node.stats().retained_bytes(), 0); + assert_eq!(node.stats().resident_bytes(), 0); + assert_eq!(node.stats().file_descriptors(), 0); + assert_eq!(node.stats().local_disk_reserved_bytes(), 0); + movement.shutdown().await; +} diff --git a/crates/cellule-host/tests/node/fleet_receivers/suffix.rs b/crates/cellule-host/tests/node/fleet_receivers/suffix.rs new file mode 100644 index 00000000..9b0ff078 --- /dev/null +++ b/crates/cellule-host/tests/node/fleet_receivers/suffix.rs @@ -0,0 +1,315 @@ +//! Fresh serving proof against a genuinely sealed native follower suffix. +use super::*; +use bytes::Bytes; +use cellule_runtime::identity::NodeId; +use cellule_runtime::node::log_recovery::{NodeLogRecovery, RecoveryCell, RecoveryCoordinator}; +use cellule_runtime::node::log_transport::{ + AppendRequest, LocalFollowerTransport, NodeLogTransport, +}; +use cellule_runtime::recovery::manifest::RecoveryManifestStore; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn sealed_suffix_inspection_requires_original_manifest_after_advancing() { + inspect_suffix(false).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn sealed_suffix_inspection_refuses_corrupt_original_manifest() { + inspect_suffix(true).await; +} + +async fn inspect_suffix(corrupt: bool) { + let movement = Movement::new(128 << 20).await; + start_suffix_recovery(&movement).await; + let result = apply(&movement.receiver, movement.action(MovementAction::Recover)).await; + assert!( + result.committed && result.execution_error.is_none(), + "{result:?}" + ); + let FleetOutcome::Recovered(recovered) = &result.outcome.outcome else { + panic!("not recovered: {result:?}") + }; + assert!(recovered.recovery.basis().control().recovery.is_some()); + movement.event(AttemptEvent::Recovered(recovered.clone())); + let cleaned = apply(&movement.receiver, movement.action(MovementAction::Cancel)).await; + assert!(matches!( + cleaned.outcome.outcome, + FleetOutcome::ReceiverCleaned + )); + movement.event(AttemptEvent::ReceiverCleaned); + let current = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + let handle = movement + .receiver + .runtime() + .local_handle(movement.inputs.catalog.clone(), ¤t) + .await + .unwrap() + .unwrap(); + assert_eq!(counter(&handle).await, 43); + let now = clock(); + handle + .execute( + MutationIdentity { + request_id: RequestId::from_bytes([231; 16]), + issued_at_ms: now, + expires_at_ms: now + 10_000, + }, + Digest::from_bytes([231; 32]), + now, + 64, + 64, + |transaction| { + transaction.execute("UPDATE counter SET value = 99", [])?; + Ok(HandlerOutcome::Success(vec![99])) + }, + ) + .await + .unwrap(); + movement.inspect(232).await; + let overlay = recovered + .recovery + .basis() + .control() + .recovery + .as_ref() + .unwrap(); + let layout = movement.inputs.authority.layout(); + let path = layout.node_log_recovery_path( + overlay.leader_session.as_bytes(), + overlay.log_epoch, + overlay.manifest_digest.as_bytes(), + ); + let (original, _) = layout.store().get_with_etag(&path).await.unwrap(); + layout.store().delete(&path).await.unwrap(); + if corrupt { + layout + .store() + .create_strict(&path, Bytes::from_static(b"corrupt-manifest")) + .await + .unwrap(); + } + assert_eq!(counter(&handle).await, 99); + let error = movement + .receiver + .inspect_fleet_action(movement.inspection(233)) + .await + .unwrap_err(); + if corrupt { + assert!(matches!(error.as_ref(), Error::Node(_)), "{error:?}"); + layout.store().delete(&path).await.unwrap(); + } else { + assert!(matches!(error.as_ref(), Error::Storage(_)), "{error:?}"); + } + layout.store().create_strict(&path, original).await.unwrap(); + assert!(matches!( + movement.inspect(234).await.outcome().outcome, + FleetOutcome::Recovered(_) + )); + movement.shutdown().await; +} + +async fn start_suffix_recovery(movement: &Movement) { + let prepared = movement.prepare().await; + let FleetOutcome::Reserved(reservation) = prepared.outcome.outcome else { + panic!("not prepared") + }; + movement.event(AttemptEvent::Reserved(reservation)); + movement.event(AttemptEvent::BeginRelease); + let observed = movement + .inputs + .authority + .load(movement.spec.target.cell_id()) + .await + .unwrap() + .unwrap(); + let predecessor = observed.value().ltx_root().unwrap(); + let tail = movement.source._root.path().join("follower-tail.sqlite"); + let writable = movement + .inputs + .replica + .open_root(&predecessor) + .await + .unwrap() + .paged() + .prepare_writable(&tail) + .await + .unwrap(); + let mut writer = writable.open_writable(&tail).unwrap(); + writer.transaction(|transaction| { + transaction.execute("UPDATE counter SET value = value + 1", [])?; + transaction.execute("UPDATE sys_meta SET commit_sequence = commit_sequence + 1, logical_time_ms = logical_time_ms + 1 WHERE singleton = 1", [])?; + Ok(()) + }).unwrap(); + let capture = writer.capture().unwrap(); + let mut frames = Vec::new(); + for (index, segment) in capture.segments.iter().enumerate() { + frames.push( + cellule_runtime::ltx::encode_node_frame( + cellule_runtime::ltx::NodeFrameScope { + leader_session: *movement.spec.source.as_bytes(), + log_epoch: 1, + node_sequence: index as u64 + 1, + application: *movement.spec.target.application().as_bytes(), + cell: *movement.spec.target.cell_id().as_bytes(), + incarnation: *movement.spec.incarnation.as_bytes(), + cell_epoch: observed.value().epoch, + commit_sequence: predecessor.commit_sequence + 1, + }, + segment.info().clone(), + Bytes::from(std::fs::read(segment.path()).unwrap()), + movement.inputs.replica.limits(), + ) + .unwrap() + .encoded() + .clone(), + ); + } + writer.close().unwrap(); + let member = SessionId::from_bytes([230; 16]); + let member_node = NodeId::from_bytes(*member.as_bytes()); + let store = cellule_runtime::FollowerStore::open( + movement.source._root.path().join("follower"), + movement.inputs.replica.limits(), + DiskBudget::new(8 << 30), + ) + .unwrap(); + let transport: Arc = + Arc::new(LocalFollowerTransport::new(member_node, store)); + transport + .append( + member_node, + AppendRequest { + leader_session: movement.spec.source, + log_epoch: 1, + frames, + covered_through: 0, + }, + ) + .await + .unwrap(); + movement.source.lease.fence(); + assert!(matches!( + movement.source.handle.query(1, 1, |_| Ok(Vec::new())).await, + Err(Error::Fenced) | Err(Error::CellDraining) + )); + let now = clock(); + let image = Digest::from_bytes([221; 32]); + let release = Digest::from_bytes([222; 32]); + let directory = cellule_runtime::node::NodeDirectory::new( + movement.inputs.authority.layout().clone(), + scope().fleet, + image, + release, + ); + let key = ed25519_dalek::SigningKey::from_bytes(&[223; 32]); + let signed = |node, session, issued, expires| { + cellule_runtime::node::NodeAdvertisement::sign( + node, + session, + "https://suffix.internal:8789".into(), + scope().fleet, + Digest::from_bytes([224; 32]), + image, + release, + &key, + 1, + issued, + expires, + vec![ + movement + .source + .node + .application() + .registry() + .module_digests()[0], + ], + vec![1], + cellule_runtime::node::NodeFailureDomain::default(), + cellule_runtime::node::NodeCapacity { + free_memory_bytes: 128 << 20, + free_disk_bytes: 8 << 30, + follower_free_bytes: 8 << 30, + follower_retained_bytes: 0, + job_credits: 1, + log_protocol: cellule_runtime::node::NODE_LOG_PROTOCOL_VERSION, + }, + ) + .unwrap() + }; + let leader = directory + .create( + signed( + movement.spec.source_node, + movement.spec.source, + now - 10_000, + now - 1, + ), + now - 10_000, + ) + .await + .unwrap(); + directory + .create( + signed(member_node, member, now - 10_000, now + 10_000), + now - 9_999, + ) + .await + .unwrap(); + let enrolled = directory + .recruit_log(&leader, 1, 1, 2, now - 9_998) + .await + .unwrap(); + directory + .activate_log(&enrolled, now - 9_997) + .await + .unwrap(); + directory + .create( + signed( + movement.spec.destination_node, + movement.spec.destination, + now, + now + 10_000, + ), + now, + ) + .await + .unwrap(); + let fenced = directory + .claim_expired(movement.spec.source, movement.spec.destination, now) + .await + .unwrap(); + let manifests = RecoveryManifestStore::new( + movement.inputs.authority.layout().clone(), + movement.inputs.replica.limits(), + ); + let recovery = + NodeLogRecovery::from_fenced(transport, &fenced, movement.inputs.replica.limits()).unwrap(); + let completed = RecoveryCoordinator::new(recovery, manifests.clone()) + .recover_and_seal( + &directory, + fenced, + vec![RecoveryCell { + application: movement.spec.target.application(), + authority: movement.inputs.authority.clone(), + observed, + }], + now, + ) + .await + .unwrap(); + assert_eq!(completed.controls.len(), 1); + assert!(completed.controls[0].value().recovery.is_some()); + *movement.recovery_inputs.lock().unwrap() = Some(FleetRecoveryInputs { + takeover: completed.takeover, + manifests, + }); + movement.event(AttemptEvent::OutcomeUnknown); + movement.event(AttemptEvent::BeginRecover); +} diff --git a/crates/cellule-host/tests/node/fleet_snapshots.rs b/crates/cellule-host/tests/node/fleet_snapshots.rs new file mode 100644 index 00000000..c39f6e3b --- /dev/null +++ b/crates/cellule-host/tests/node/fleet_snapshots.rs @@ -0,0 +1,186 @@ +//! Native page reads share the original fleet bank and exact journal barrier. +use super::fleet_actions::{clock, fixture}; +use super::*; +use cellule_host::fleet::{FleetSnapshotNativePage, FleetSnapshotSubject}; +use cellule_runtime::fleet::operations::MAX_RECORD_BYTES; + +async fn until(check: impl Fn() -> bool) { + tokio::time::timeout(Duration::from_secs(3), async { + while !check() { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn lost_snapshot_waiter_retains_original_job_and_native_page_until_join() { + let test = fixture().await; + let cellule_runtime::fleet::operations::FleetActionKind::Movement { attempt, .. } = + test.action.kind() + else { + panic!("movement fixture required"); + }; + let cell = attempt.spec().target.cell_id(); + let original = test.authority.load(cell).await.unwrap().unwrap(); + let request = test + .journal + .snapshot_request(FleetSnapshotSubject::Cells(None), 220); + test.journal.block_inspections.store(true, Ordering::SeqCst); + let node = test.node.clone(); + let issued = request.clone(); + let waiter = tokio::spawn(async move { node.fleet_snapshot(issued).await }); + test.journal.inspection_entered.notified().await; + waiter.abort(); + assert!(matches!(waiter.await, Err(error) if error.is_cancelled())); + assert_eq!(test.journal.snapshot_calls.load(Ordering::SeqCst), 1); + let retained = test.node.stats().retained_bytes(); + let node = test.node.clone(); + let replay = request.clone(); + let mut retry = Box::pin(node.fleet_snapshot(replay)); + // Register this waiter on the still-blocked original job before releasing + // it. A yield does not guarantee that a spawned retry has reached submit; + // a late retry can correctly recapture after the original job was joined. + assert!(futures_util::poll!(retry.as_mut()).is_pending()); + assert_eq!(test.node.stats().retained_bytes(), retained); + assert_eq!(test.journal.snapshot_calls.load(Ordering::SeqCst), 1); + test.journal + .block_inspections + .store(false, Ordering::SeqCst); + test.journal.inspection_resume.add_permits(1); + let result = retry.await.unwrap(); + result.validate(&request, clock()).unwrap(); + assert_eq!(result.request(), &request); + assert_eq!(test.journal.snapshot_calls.load(Ordering::SeqCst), 2); + assert!(!result.bindings().managed_startup); + let FleetSnapshotNativePage::Cells(page) = result.page() else { + panic!("original actor page required"); + }; + assert_eq!(page.session(), request.session()); + assert_eq!(page.owned_cells(), 1); + assert_eq!(page.transitioning_cells(), 0); + assert_eq!(page.entries().len(), 1); + assert_eq!(page.entries()[0].cell(), cell); + let unchanged = test.authority.load(cell).await.unwrap().unwrap(); + assert_eq!(unchanged.value(), original.value()); + assert!(test.node.stats().retained_bytes() >= 3 * MAX_RECORD_BYTES as usize + (1 << 20)); + drop(result); + test.node.shutdown().await.unwrap(); + assert_eq!(test.node.stats().retained_bytes(), 0); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn changed_registry_after_native_capture_fails_without_empty_or_restamped_evidence() { + let test = fixture().await; + let request = test + .journal + .snapshot_request(FleetSnapshotSubject::Cells(None), 221); + test.journal + .pause_snapshot_post + .store(true, Ordering::SeqCst); + let node = test.node.clone(); + let issued = request.clone(); + let capture = tokio::spawn(async move { node.fleet_snapshot(issued).await }); + test.journal.snapshot_post_entered.notified().await; + assert!(!capture.is_finished()); + assert!(test.node.stats().retained_bytes() >= 3 * MAX_RECORD_BYTES as usize + (1 << 20)); + test.journal.advance_snapshot_registry(); + test.journal.snapshot_post_resume.add_permits(1); + let error = capture.await.unwrap().err().unwrap(); + assert!(matches!( + error.as_ref(), + Error::Facility { + name: "fleet-action-journal", + .. + } + )); + assert_eq!(test.journal.snapshot_calls.load(Ordering::SeqCst), 2); + assert!(test.node.stats().retained_bytes() < 1 << 20); + assert!(test.node.fleet_snapshot(request).await.is_err()); + test.node.shutdown().await.unwrap(); + assert_eq!(test.node.stats().retained_bytes(), 0); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn two_snapshot_reads_use_the_same_bound_as_effects_and_drain_retains_their_joins() { + let test = fixture().await; + test.journal.block_inspections.store(true, Ordering::SeqCst); + let mut callers = Vec::new(); + for nonce in [222, 223] { + let node = test.node.clone(); + let request = test + .journal + .snapshot_request(FleetSnapshotSubject::Host, nonce); + callers.push(tokio::spawn( + async move { node.fleet_snapshot(request).await }, + )); + } + until(|| test.journal.snapshot_calls.load(Ordering::SeqCst) == 2).await; + let error = test + .node + .apply_fleet_action(test.action.clone(), clock()) + .await + .err() + .unwrap(); + assert!(matches!( + error.as_ref(), + Error::Capacity("fleet action receipt bound") + )); + for caller in callers { + caller.abort(); + assert!(matches!(caller.await, Err(error) if error.is_cancelled())); + } + let node = test.node.clone(); + let drain = tokio::spawn(async move { node.shutdown().await }); + until(|| test.node.state() == NodeState::Draining).await; + assert!(!drain.is_finished()); + test.journal + .block_inspections + .store(false, Ordering::SeqCst); + test.journal.inspection_resume.add_permits(2); + tokio::time::timeout(Duration::from_secs(3), drain) + .await + .unwrap() + .unwrap() + .unwrap(); + assert_eq!(test.journal.snapshot_calls.load(Ordering::SeqCst), 4); + assert_eq!(test.node.stats().retained_bytes(), 0); +} + +#[tokio::test] +async fn unbound_native_owner_and_foreign_endpoint_never_supply_empty_role_coverage() { + let test = fixture().await; + let request = test + .journal + .snapshot_request(FleetSnapshotSubject::Readers(None), 224); + let result = test.node.fleet_snapshot(request.clone()).await.unwrap(); + result.validate(&request, clock()).unwrap(); + assert!(matches!(result.page(), FleetSnapshotNativePage::Unbound)); + assert!(!result.bindings().readers); + let foreign = cellule_host::fleet::FleetSnapshotRequest::new( + request.expected().clone(), + request.nonce(), + request.node(), + SessionId::from_bytes([225; 16]), + request.subject().clone(), + request.limit(), + request.issued_at_ms(), + request.deadline_ms(), + ) + .unwrap(); + let calls = test.journal.snapshot_calls.load(Ordering::SeqCst); + assert!(matches!( + test.node + .fleet_snapshot(foreign) + .await + .err() + .unwrap() + .as_ref(), + Error::FleetOperation(source) if matches!(source.as_ref(), cellule_runtime::fleet::operations::OperationError::Fenced) + )); + assert_eq!(test.journal.snapshot_calls.load(Ordering::SeqCst), calls); + drop(result); + test.node.shutdown().await.unwrap(); + assert_eq!(test.node.stats().retained_bytes(), 0); +} diff --git a/crates/cellule-host/tests/node/inventory.rs b/crates/cellule-host/tests/node/inventory.rs new file mode 100644 index 00000000..32c2c69d --- /dev/null +++ b/crates/cellule-host/tests/node/inventory.rs @@ -0,0 +1,291 @@ +//! Public reader inventory with real restored views and shared host admission. + +use super::*; +use cellule_host::read_replicas::ReaderInventoryCursor; +use cellule_runtime::peer::PeerReplicaResolver; +use cellule_runtime::{ + CellRuntime, + cell::catalog::{CatalogEntry, CellCatalog}, + control::{Owner, authority::CellAuthority}, + identity::{CellTarget, IncarnationId, NamespaceId, NodeId, TenantId}, + ltx::{CellReplica, CellStorageLayout}, + node::{NodeAdvertisement, NodeCapacity, NodeDirectory, NodeFailureDomain, NodeMode}, +}; +use cellule_store::Store; +use ed25519_dalek::SigningKey; +use object_store::{memory::InMemory, path::Path}; + +pub(super) fn clock() -> i64 { + i64::try_from( + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_millis(), + ) + .unwrap() +} + +pub(super) fn advertisement(id: u8, code: Digest, now: i64) -> NodeAdvertisement { + NodeAdvertisement::sign( + NodeId::from_bytes([id; 16]), + SessionId::from_bytes([id; 16]), + "https://node.internal:8789".into(), + Digest::from_bytes([2; 32]), + Digest::from_bytes([3; 32]), + Digest::from_bytes([4; 32]), + Digest::from_bytes([5; 32]), + &SigningKey::from_bytes(&[7; 32]), + 1, + now, + now + 30_000, + vec![code], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + free_memory_bytes: 1 << 30, + free_disk_bytes: 1 << 30, + follower_free_bytes: 1 << 30, + follower_retained_bytes: 0, + job_credits: 3, + log_protocol: 1, + }, + ) + .unwrap() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn reader_inventory_pages_track_real_views_cordon_and_canonical_shutdown() { + let application = application(); + let code = application.registry().module_digests()[0]; + let namespace = NamespaceId::from_bytes([2; 16]); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("reader-inventory"), + [3; 16], + ); + let directory = NodeDirectory::new( + layout.clone(), + Digest::from_bytes([2; 32]), + Digest::from_bytes([4; 32]), + Digest::from_bytes([5; 32]), + ); + let now = clock(); + for id in [1, 2] { + directory + .create(advertisement(id, code, now), now) + .await + .unwrap(); + } + let limits = ReplicaLimits { + max_database_bytes: 64 << 20, + max_capture_bytes: 16 << 20, + ..ReplicaLimits::default() + }; + let reader_root = tempfile::tempdir().unwrap(); + let node = CellNodeBuilder::new(application) + .with_runtime( + SqlWorkerPool::new(1, 8) + .unwrap() + .with_native_memory_limit(128 << 20) + .unwrap(), + 16 << 20, + ) + .with_replica_host(ReplicaHost::default()) + .with_session(SessionId::from_bytes([2; 16])) + .build() + .unwrap(); + node.install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + let manager = node + .install_read_replicas( + layout.clone(), + directory, + reader_root.path().to_owned(), + limits, + ) + .unwrap(); + node.install_node_lease(NodeLeaseGuard::new(now, now + 60_000).unwrap()) + .unwrap(); + let retained = node.stats().retained_bytes(); + assert!( + manager + .fleet_reader_enrollments_page(None, 128, now) + .unwrap() + .is_none() + ); + assert_eq!(node.stats().retained_bytes(), retained); + for limit in [0, 129, usize::MAX] { + assert!(manager.fleet_readers_page(None, limit, now).await.is_err()); + assert!( + manager + .fleet_reader_enrollments_page(None, limit, now) + .is_err() + ); + } + assert!(manager.fleet_reader_enrollments_page(None, 1, -1).is_err()); + assert!(manager.fleet_readers_page(None, 1, -1).await.is_err()); + assert!(ReaderInventoryCursor::from_bytes(&[0; 63]).is_err()); + let empty = manager.fleet_readers_page(None, 128, now).await.unwrap(); + assert_eq!(empty.total_views(), 0); + assert!(empty.entries().is_empty()); + assert!(empty.next().is_none()); + drop(empty); + let source_root = tempfile::tempdir().unwrap(); + let source = CellRuntime::new( + SqlWorkerPool::new(1, 3).unwrap(), + 16 << 20, + SessionId::from_bytes([1; 16]), + ) + .unwrap(); + let catalog = CellCatalog::new(layout.clone(), TenantId::from_bytes([1; 16])); + let authority = CellAuthority::new(layout.clone()); + let mut handles = Vec::new(); + let mut targets = Vec::new(); + for id in 1..=3_u8 { + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + ApplicationId::from_bytes([3; 16]), + namespace, + &[id], + ) + .unwrap(); + let incarnation = IncarnationId::from_bytes([id; 16]); + let proof = catalog + .provision(CatalogEntry::new(&target, CatalogRole::Sql, code, 1).unwrap()) + .await + .unwrap(); + let observed = authority + .create_initial( + &proof, + incarnation, + Owner { + session: SessionId::from_bytes([1; 16]), + endpoint: "https://node.internal:8789".into(), + }, + ) + .await + .unwrap(); + let replica = CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + limits, + ) + .unwrap(); + handles.push( + source + .bootstrap( + proof, + replica, + authority.clone(), + observed, + source_root.path().join(format!("{id}.sqlite")), + |transaction| { + transaction.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES (17)", + )?; + Ok(()) + }, + ) + .await + .unwrap(), + ); + manager.set_target(&target, 0, 1).await.unwrap().unwrap(); + let receipt = manager + .activate(target.clone(), SessionId::from_bytes([1; 16])) + .await + .unwrap(); + assert_eq!(receipt.cell, target.cell_id()); + assert_eq!(receipt.incarnation, incarnation); + targets.push(target); + } + let mut retained_peer_views = Vec::new(); + for target in &targets { + retained_peer_views.push(manager.resolve(target.clone()).await.unwrap()); + } + let resident_before = node.stats().resident_bytes(); + let before = node.stats().retained_bytes(); + let first = manager.fleet_readers_page(None, 1, clock()).await.unwrap(); + assert_eq!(first.session(), SessionId::from_bytes([2; 16])); + assert_eq!(first.total_views(), 3); + assert_eq!(first.entries().len(), 1); + assert_eq!(node.stats().retained_bytes(), before + (1 << 20)); + let cursor = ReaderInventoryCursor::from_bytes(&first.next().unwrap().to_bytes()).unwrap(); + let next = manager + .fleet_readers_page(Some(cursor), 128, clock()) + .await + .unwrap(); + assert_eq!(next.topology(), first.topology()); + assert_eq!(next.entries().len(), 2); + assert!(next.next().is_none()); + assert!( + next.entries()[0].receipt().cell.as_bytes() > first.entries()[0].receipt().cell.as_bytes() + ); + let removed = first.entries()[0].receipt().cell; + drop(first); + drop(next); + assert_eq!(node.stats().retained_bytes(), before); + manager.remove(removed).await.unwrap(); + assert!(node.stats().resident_bytes() < resident_before); + for peer in &retained_peer_views { + if peer.receipt().await.cell == removed { + assert!(matches!(peer.readiness().await, Err(Error::Fenced))); + } + } + assert!( + manager + .fleet_readers_page(Some(cursor), 1, clock()) + .await + .is_err() + ); + assert_eq!(node.stats().retained_bytes(), before); + let before_cordon = manager.fleet_readers_page(None, 1, clock()).await.unwrap(); + let cursor = before_cordon.next().unwrap(); + let topology = before_cordon.topology(); + drop(before_cordon); + node.runtime().stop_acquiring().unwrap(); + assert!( + manager + .fleet_readers_page(Some(cursor), 1, clock()) + .await + .is_err() + ); + let cordoned = manager + .fleet_readers_page(None, 128, clock()) + .await + .unwrap(); + assert_eq!(cordoned.mode(), NodeMode::Cordoned); + assert_ne!(cordoned.topology(), topology); + assert_eq!(cordoned.total_views(), 2); + drop(cordoned); + let removed_target = targets + .into_iter() + .find(|target| target.cell_id() == removed) + .unwrap(); + assert!(matches!( + manager + .activate(removed_target, SessionId::from_bytes([1; 16])) + .await, + Err(Error::CellDraining) + )); + manager.shutdown().await.unwrap(); + let closed = manager + .fleet_readers_page(None, 128, clock()) + .await + .unwrap(); + assert!(closed.closed()); + assert_eq!(closed.total_views(), 0); + drop(closed); + node.shutdown().await.unwrap(); + assert_eq!(node.state(), NodeState::Stopped); + assert_eq!(node.stats().retained_bytes(), 0); + assert_eq!(node.stats().resident_bytes(), 0); + for peer in &retained_peer_views { + assert!(peer.readiness().await.is_err()); + } + drop(retained_peer_views); + for handle in handles { + handle.drain().await.unwrap(); + } + source.shutdown().await.unwrap(); +} diff --git a/crates/cellule-host/tests/node/lifecycle.rs b/crates/cellule-host/tests/node/lifecycle.rs index 798cdcde..32b92a07 100644 --- a/crates/cellule-host/tests/node/lifecycle.rs +++ b/crates/cellule-host/tests/node/lifecycle.rs @@ -2,6 +2,91 @@ use super::*; +mod retained; + +#[tokio::test] +async fn operational_pressure_recovers_acquisition_without_clearing_a_cordon() { + use cellule_runtime::fleet::pressure::{PressureSample, PressureState}; + use cellule_runtime::node::NodeMode; + + let node = CellNodeBuilder::new(application()) + .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) + .with_replica_host(ReplicaHost::default()) + .with_session(SessionId::from_bytes([96; 16])) + .build() + .unwrap(); + node.install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + node.install_node_lease_for_startup(NodeLeaseGuard::new(0, 60_000).unwrap()) + .unwrap(); + node.start().unwrap(); + let at_ms = i64::try_from( + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_millis(), + ) + .unwrap(); + let sample = |offset, used| PressureSample { + at_ms: at_ms + offset, + memory_used_permille: used, + disk_used_permille: used, + jobs_used_permille: 0, + stale: false, + }; + node.runtime() + .observe_pressure(sample(0, 900)) + .await + .unwrap(); + assert_eq!( + node.runtime() + .observe_pressure(sample(1_000, 900)) + .await + .unwrap(), + PressureState::Critical + ); + assert!(!node.runtime().is_acquiring()); + node.runtime() + .observe_pressure(sample(2_000, 0)) + .await + .unwrap(); + assert_eq!( + node.runtime() + .observe_pressure(sample(3_000, 0)) + .await + .unwrap(), + PressureState::Normal + ); + assert!(node.runtime().is_acquiring()); + node.runtime().stop_acquiring().unwrap(); + node.runtime() + .observe_pressure(sample(4_000, 900)) + .await + .unwrap(); + node.runtime() + .observe_pressure(sample(5_000, 900)) + .await + .unwrap(); + node.runtime() + .observe_pressure(sample(6_000, 0)) + .await + .unwrap(); + assert_eq!( + node.runtime() + .observe_pressure(sample(7_000, 0)) + .await + .unwrap(), + PressureState::Normal + ); + let observed = node.runtime().operational_sample().unwrap().unwrap(); + assert_eq!(observed.mode, NodeMode::Cordoned); + assert_eq!(observed.observed_at_ms, at_ms + 7_000); + assert!(!node.runtime().is_acquiring()); + assert!(node.is_ready()); + node.drain().await.unwrap(); + assert_eq!(node.state(), NodeState::Stopped); +} + #[tokio::test] async fn node_shutdown_is_idempotent_and_returns_stopped_state() { let pool = SqlWorkerPool::new(1, 1).unwrap(); diff --git a/crates/cellule-host/tests/node/lifecycle/retained.rs b/crates/cellule-host/tests/node/lifecycle/retained.rs new file mode 100644 index 00000000..f3dec104 --- /dev/null +++ b/crates/cellule-host/tests/node/lifecycle/retained.rs @@ -0,0 +1,269 @@ +//! Original host drain ownership across lost management waiters. +use super::*; + +struct DrainProbe { + completed: Arc, + abandoned: Arc, +} +impl Drop for DrainProbe { + fn drop(&mut self) { + if !self.completed.load(Ordering::Acquire) { + self.abandoned.store(true, Ordering::Release); + } + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn accepted_facility_drain_continues_after_its_only_waiter_is_cancelled() { + let node = Arc::new( + CellNodeBuilder::new(application()) + .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) + .with_replica_host(ReplicaHost::default()) + .with_session(SessionId::from_bytes([35; 16])) + .build() + .unwrap(), + ); + let lease_shutdown = CancellationToken::new(); + let tasks = node + .install_task_group(CancellationToken::new(), lease_shutdown.clone()) + .unwrap(); + let withdrawn = Arc::new(AtomicBool::new(false)); + let lease_withdrawn = Arc::clone(&withdrawn); + tasks + .spawn_lease_maintenance(async move { + lease_shutdown.cancelled().await; + lease_withdrawn.store(true, Ordering::Release); + Ok::<(), Error>(()) + }) + .unwrap(); + let started = Arc::new(tokio::sync::Notify::new()); + let resume = Arc::new(tokio::sync::Semaphore::new(0)); + let completed = Arc::new(AtomicBool::new(false)); + let abandoned = Arc::new(AtomicBool::new(false)); + let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let entered = Arc::clone(&started); + let gate = Arc::clone(&resume); + let finished = Arc::clone(&completed); + let cancelled = Arc::clone(&abandoned); + let invoked = Arc::clone(&calls); + node.install_facility( + CellNodeFacility::new("accepted-drain", move || { + let entered = Arc::clone(&entered); + let gate = Arc::clone(&gate); + let finished = Arc::clone(&finished); + let cancelled = Arc::clone(&cancelled); + let invoked = Arc::clone(&invoked); + async move { + let _probe = DrainProbe { + completed: Arc::clone(&finished), + abandoned: cancelled, + }; + invoked.fetch_add(1, Ordering::AcqRel); + entered.notify_one(); + gate.acquire().await.unwrap().forget(); + finished.store(true, Ordering::Release); + Ok(()) + } + }) + .unwrap(), + ) + .unwrap(); + node.install_node_lease_for_startup(NodeLeaseGuard::new(0, 60_000).unwrap()) + .unwrap(); + node.start().unwrap(); + assert!(node.is_ready()); + + let caller = Arc::clone(&node); + let waiter = tokio::spawn(async move { caller.shutdown().await }); + started.notified().await; + assert_eq!(node.state(), NodeState::Draining); + let running = node.drain_observation().unwrap().unwrap(); + assert_eq!(running.serial, 1); + assert_eq!(running.phase, NodeDrainPhase::Running); + assert!(running.result.is_none()); + assert!(!withdrawn.load(Ordering::Acquire)); + waiter.abort(); + assert!(waiter.await.unwrap_err().is_cancelled()); + let was_abandoned = abandoned.load(Ordering::Acquire); + // The extra permit lets the old implementation retry during test cleanup. + // The fixed implementation must finish its first callback without that retry. + resume.add_permits(2); + let autonomous_stop = tokio::time::timeout(Duration::from_secs(1), async { + while node.state() != NodeState::Stopped { + tokio::task::yield_now().await; + } + }) + .await + .is_ok(); + node.shutdown().await.unwrap(); + + assert!( + !was_abandoned, + "caller cancellation dropped accepted facility work" + ); + assert!( + autonomous_stop, + "accepted host drain did not continue independently" + ); + assert_eq!(calls.load(Ordering::Acquire), 1); + assert!(completed.load(Ordering::Acquire)); + assert!(!abandoned.load(Ordering::Acquire)); + assert!(withdrawn.load(Ordering::Acquire)); + assert_eq!(node.stats().active_cells(), 0); + assert_eq!(node.stats().retained_bytes(), 0); + assert_eq!(node.stats().worker_jobs(), 0); + assert_eq!(node.stats().file_descriptors(), 0); + let joined = node.drain_observation().unwrap().unwrap(); + assert_eq!(joined.serial, 1); + assert_eq!(joined.phase, NodeDrainPhase::Joined); + assert!(joined.result.unwrap().is_ok()); + assert!(joined.first_failure.is_none()); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn cancelled_shutdown_keeps_scale_down_queued_on_the_original_lane() { + let node = starting_node(36); + let entered = Arc::new(tokio::sync::Notify::new()); + let gate = Arc::new(tokio::sync::Semaphore::new(0)); + let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let started = Arc::clone(&entered); + let resume = Arc::clone(&gate); + let invoked = Arc::clone(&calls); + node.install_facility( + CellNodeFacility::new("closing-gate", move || { + let started = Arc::clone(&started); + let resume = Arc::clone(&resume); + let invoked = Arc::clone(&invoked); + async move { + invoked.fetch_add(1, Ordering::AcqRel); + started.notify_one(); + resume.acquire().await.unwrap().forget(); + Ok(()) + } + }) + .unwrap(), + ) + .unwrap(); + node.start().unwrap(); + let caller = Arc::clone(&node); + let waiter = tokio::spawn(async move { caller.shutdown().await }); + entered.notified().await; + waiter.abort(); + assert!(waiter.await.unwrap_err().is_cancelled()); + + let mut scale_down = + Box::pin(node.drain_for_scale_down(Instant::now() + Duration::from_millis(20))); + assert!(futures_util::poll!(&mut scale_down).is_pending()); + tokio::time::sleep(Duration::from_millis(30)).await; + assert!(futures_util::poll!(&mut scale_down).is_pending()); + assert_eq!(calls.load(Ordering::Acquire), 1); + let original = node.drain_observation().unwrap().unwrap(); + assert_eq!(original.serial, 1); + assert_eq!(original.phase, NodeDrainPhase::Running); + assert!(original.result.is_none()); + gate.add_permits(1); + let settled = tokio::time::timeout(Duration::from_secs(1), scale_down) + .await + .unwrap() + .unwrap(); + assert!(settled.ready_to_stop()); + assert_eq!(node.state(), NodeState::Stopped); + assert_eq!(calls.load(Ordering::Acquire), 1); + let joined = node.drain_observation().unwrap().unwrap(); + assert_eq!(joined.serial, 1); + assert_eq!(joined.phase, NodeDrainPhase::Joined); + assert!(joined.result.unwrap().is_ok()); + assert_eq!(node.stats().active_cells(), 0); + assert_eq!(node.stats().retained_bytes(), 0); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn host_deadline_retry_retains_original_failure_history_after_joined_stop() { + let node = starting_node(37); + let gate = Arc::new(tokio::sync::Semaphore::new(0)); + let resume = Arc::clone(&gate); + node.install_facility( + CellNodeFacility::new("retryable-drain", move || { + let resume = Arc::clone(&resume); + async move { + resume.acquire().await.unwrap().forget(); + Ok(()) + } + }) + .unwrap(), + ) + .unwrap(); + node.start().unwrap(); + let failed = node + .shutdown_until(Instant::now() + Duration::from_millis(10)) + .await + .unwrap_err(); + let first = node.drain_observation().unwrap().unwrap(); + assert_eq!(first.serial, 1); + assert_eq!(first.phase, NodeDrainPhase::Joined); + let original = first.result.unwrap().unwrap_err(); + assert!(Arc::ptr_eq( + &original, + first.first_failure.as_ref().unwrap() + )); + assert!(Arc::ptr_eq( + &original, + first.latest_failure.as_ref().unwrap() + )); + assert!(std::ptr::eq( + error_source::(&failed), + error_source::(original.as_ref()), + )); + assert_eq!( + error_source::(original.as_ref()).kind(), + std::io::ErrorKind::TimedOut, + ); + assert_eq!(node.state(), NodeState::Draining); + gate.add_permits(1); + node.shutdown().await.unwrap(); + let completed = node.drain_observation().unwrap().unwrap(); + assert_eq!(completed.serial, 2); + assert_eq!(completed.phase, NodeDrainPhase::Joined); + assert!(completed.result.unwrap().is_ok()); + assert!(Arc::ptr_eq( + &original, + completed.first_failure.as_ref().unwrap() + )); + assert!(Arc::ptr_eq( + &original, + completed.latest_failure.as_ref().unwrap() + )); + assert_eq!(node.state(), NodeState::Stopped); + assert_eq!(node.stats().retained_bytes(), 0); + node.shutdown().await.unwrap(); + assert_eq!(node.drain_observation().unwrap().unwrap().serial, 2); +} + +fn starting_node(session: u8) -> Arc { + let node = Arc::new( + CellNodeBuilder::new(application()) + .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) + .with_replica_host(ReplicaHost::default()) + .with_session(SessionId::from_bytes([session; 16])) + .build() + .unwrap(), + ); + node.install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + node.install_node_lease_for_startup(NodeLeaseGuard::new(0, 60_000).unwrap()) + .unwrap(); + assert!(node.drain_observation().unwrap().is_none()); + node +} + +fn error_source<'a, E: std::error::Error + 'static>( + error: &'a (dyn std::error::Error + 'static), +) -> &'a E { + let mut source = error; + loop { + if let Some(error) = source.downcast_ref::() { + return error; + } + source = source.source().expect("original error must be retained"); + } +} diff --git a/crates/cellule-host/tests/node/movement.rs b/crates/cellule-host/tests/node/movement.rs new file mode 100644 index 00000000..b2df57fd --- /dev/null +++ b/crates/cellule-host/tests/node/movement.rs @@ -0,0 +1,236 @@ +//! Canonical source evidence consumed by a leased receiver's prepared resources. + +use super::*; +use cellule_runtime::cell::actor::CellInventoryEntry; +use cellule_runtime::cell::catalog::{CatalogEntry, CellCatalog}; +use cellule_runtime::cell::executor::{HandlerOutcome, MutationIdentity, Resolution}; +use cellule_runtime::control::{Owner, authority::CellAuthority}; +use cellule_runtime::fleet::operations::{AttemptId, MoveAttemptSpec, OperationId}; +use cellule_runtime::identity::{ + CellTarget, IncarnationId, NamespaceId, NodeId, RequestId, TenantId, +}; +use cellule_runtime::ltx::{CellReplica, CellStorageLayout}; +use cellule_store::Store; +use object_store::{memory::InMemory, path::Path}; + +fn clock() -> i64 { + i64::try_from( + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_millis(), + ) + .unwrap() +} + +fn node(session: SessionId) -> CellNode { + let node = CellNodeBuilder::new(application()) + .with_runtime( + SqlWorkerPool::new(1, 8) + .unwrap() + .with_native_memory_limit(128 << 20) + .unwrap(), + 64 << 20, + ) + .with_replica_host(ReplicaHost::default().with_local_disk_budget(DiskBudget::new(8 << 30))) + .with_session(session) + .build() + .unwrap(); + node.install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + let now = clock(); + node.install_node_lease(NodeLeaseGuard::new(now, now + 60_000).unwrap()) + .unwrap(); + node +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn leased_host_release_returns_exact_evidence_for_prepared_receiver_activation() { + let source_session = SessionId::from_bytes([176; 16]); + let receiver_session = SessionId::from_bytes([177; 16]); + let source = node(source_session); + let receiver = node(receiver_session); + let root = tempfile::tempdir().unwrap(); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("host-movement"), + [3; 16], + ); + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + ApplicationId::from_bytes([3; 16]), + NamespaceId::from_bytes([2; 16]), + b"host-movement", + ) + .unwrap(); + let incarnation = IncarnationId::from_bytes([178; 16]); + let limits = ReplicaLimits { + max_database_bytes: 64 << 20, + max_capture_bytes: 16 << 20, + ..ReplicaLimits::default() + }; + let replica = CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + limits, + ) + .unwrap(); + let code = source.application().registry().module_digests()[0]; + let catalog = CellCatalog::new(layout.clone(), target.tenant()); + let proof = catalog + .provision(CatalogEntry::new(&target, CatalogRole::Sql, code, 1).unwrap()) + .await + .unwrap(); + let authority = CellAuthority::new(layout); + let initial = authority + .create_initial( + &proof, + incarnation, + Owner { + session: source_session, + endpoint: "https://source.internal:8789".into(), + }, + ) + .await + .unwrap(); + let handle = source + .runtime() + .bootstrap( + proof.clone(), + replica.clone(), + authority.clone(), + initial, + root.path().join("source.sqlite"), + |transaction| { + transaction.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES (0)", + )?; + Ok(()) + }, + ) + .await + .unwrap(); + let now = clock(); + let identity = MutationIdentity { + request_id: RequestId::from_bytes([179; 16]), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }; + let digest = Digest::from_bytes([179; 32]); + let acknowledged = handle + .execute(identity, digest, now, 64, 64, |transaction| { + transaction.execute("UPDATE counter SET value = 37", [])?; + Ok(HandlerOutcome::Success(vec![37])) + }) + .await + .unwrap(); + let observation = tokio::time::timeout(Duration::from_secs(5), async { + loop { + let page = source.runtime().fleet_cells_page(None, 128).await.unwrap(); + if let Some(CellInventoryEntry::Owned(owner)) = page.entries().first() + && owner.cost.is_some() + && owner.stable_observations == 2 + { + return (**owner).clone(); + } + drop(page); + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + let spec = MoveAttemptSpec { + id: AttemptId { + operation: OperationId::from_bytes([180; 16]).unwrap(), + sequence: 1, + }, + target: target.clone(), + incarnation, + source_node: NodeId::from_bytes([176; 16]), + source: source_session, + generation: observation.generation, + source_epoch: observation.position.unwrap().epoch, + destination_node: NodeId::from_bytes([177; 16]), + destination: receiver_session, + cost: observation.cost.unwrap(), + snapshot_digest: Digest::from_bytes([181; 32]), + deadline_ms: clock() + 60_000, + }; + let prepared = receiver + .runtime() + .prepare_receiver( + spec.clone(), + proof.clone(), + replica, + root.path().join("receiver.sqlite"), + spec.deadline_ms, + clock(), + ) + .unwrap(); + assert_eq!( + receiver.stats().local_disk_reserved_bytes(), + spec.cost.disk_bytes + ); + // Cordon retains source release eligibility and the leased canonical path. + source.begin_scale_down().unwrap(); + let released = source + .release_idle_cell_at( + target.cell_id(), + source_session, + spec.generation, + incarnation, + spec.source_epoch, + ) + .await + .unwrap(); + assert_eq!(released.epoch, spec.source_epoch); + assert_eq!(released.incarnation, incarnation); + assert!(released.root.commit_sequence >= acknowledged.commit_sequence()); + assert_eq!(source.stats().active_cells(), 0); + let idle = authority.load(target.cell_id()).await.unwrap().unwrap(); + assert_eq!(Some(&released.root), idle.value().root.as_ref()); + let activated = receiver + .runtime() + .activate_prepared_receiver( + &prepared, + authority.clone(), + idle, + Owner { + session: receiver_session, + endpoint: "https://receiver.internal:8789".into(), + }, + clock(), + ) + .await + .unwrap(); + assert_eq!( + activated + .resolve(identity, digest, clock(), 64) + .await + .unwrap(), + Resolution::Committed(acknowledged) + ); + assert_eq!( + activated + .query(64, 64, |connection| { + Ok(connection + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))? + .to_be_bytes() + .to_vec()) + }) + .await + .unwrap(), + 37_i64.to_be_bytes() + ); + let serving = authority.load(target.cell_id()).await.unwrap().unwrap(); + assert_eq!( + serving.value().owner.as_ref().unwrap().session, + receiver_session + ); + assert_eq!(Some(&released.root), serving.value().root.as_ref()); + source.shutdown().await.unwrap(); + receiver.shutdown().await.unwrap(); + assert_eq!(receiver.stats().active_cells(), 0); + assert_eq!(receiver.stats().local_disk_reserved_bytes(), 0); +} diff --git a/crates/cellule-host/tests/node/reader_closure.rs b/crates/cellule-host/tests/node/reader_closure.rs new file mode 100644 index 00000000..c06870fc --- /dev/null +++ b/crates/cellule-host/tests/node/reader_closure.rs @@ -0,0 +1,865 @@ +//! Public host reader ownership through cancelled removal and drain deadlines. + +use super::*; +use cellule_host::read_replicas::ReadReplicaManager; +use cellule_runtime::{ + CellRuntime, + cell::{ + actor::CellHandle, + catalog::{CatalogEntry, CellCatalog}, + }, + client::CellReadReplica, + control::{Owner, authority::CellAuthority}, + identity::{CellTarget, IncarnationId, NamespaceId, TenantId}, + ltx::{CellReplica, CellStorageLayout}, + node::NodeDirectory, + peer::PeerReplicaResolver, + primitives::sql::{SqlBatch, SqlStatement, SqlValue}, + registry::{OperationDescriptor, Query, QueryContext}, +}; +use cellule_store::Store; +use object_store::{memory::InMemory, path::Path}; +use std::{ + collections::BTreeMap, + sync::{OnceLock, atomic::AtomicU64}, +}; +use tokio::sync::{Notify, oneshot}; + +type QueryGate = (Arc, oneshot::Receiver<()>); +static QUERIES: Mutex> = Mutex::new(BTreeMap::new()); + +struct Pause { + id: u64, + entered: Arc, + release: Option>, +} +impl Pause { + fn new() -> Self { + static NEXT: AtomicU64 = AtomicU64::new(1); + let id = NEXT.fetch_add(1, Ordering::Relaxed); + let entered = Arc::new(Notify::new()); + let (release, receive) = oneshot::channel(); + assert!( + QUERIES + .lock() + .unwrap() + .insert(id, (entered.clone(), receive)) + .is_none() + ); + Self { + id, + entered, + release: Some(release), + } + } + async fn entered(&self) { + tokio::time::timeout(Duration::from_secs(3), self.entered.notified()) + .await + .unwrap(); + } + fn release(mut self) { + self.release.take().unwrap().send(()).unwrap(); + } +} +impl Drop for Pause { + fn drop(&mut self) { + QUERIES.lock().unwrap().remove(&self.id); + } +} + +struct ReadCounter; +impl Query for ReadCounter { + const MODULE: &'static str = Module::NAME; + const ID: u32 = 1; + const CODEC_VERSION: u32 = 1; + type Input = u64; + type Output = i64; + fn execute(context: &mut QueryContext<'_>, token: u64) -> cellule_runtime::Result { + if token != 0 { + let (entered, receive) = QUERIES.lock().unwrap().remove(&token).unwrap(); + entered.notify_one(); + receive + .blocking_recv() + .map_err(|_| Error::Command("reader query gate dropped"))?; + } + let sets = context.sql(&SqlBatch { + statements: vec![SqlStatement { + sql: "SELECT value FROM counter".into(), + parameters: vec![], + }], + })?; + match sets + .first() + .and_then(|set| set.rows.first()) + .and_then(|row| row.first()) + { + Some(SqlValue::Integer(value)) => Ok(*value), + _ => Err(Error::Command("reader counter is missing")), + } + } +} +struct ReaderModule; +impl CellModule for ReaderModule { + const NAME: &'static str = Module::NAME; + fn descriptor(&self) -> &'static ModuleDescriptor { + static DESCRIPTOR: OnceLock = OnceLock::new(); + DESCRIPTOR.get_or_init(|| ModuleDescriptor { + queries: &[OperationDescriptor { + id: 1, + codec_version: 1, + schema_min: 1, + schema_max: 1, + input_limit: 8, + output_limit: 8, + }], + ..*Module.descriptor() + }) + } + fn register(self, registry: &mut RegistryBuilder) -> cellule_runtime::Result<()> { + registry.bind_query::() + } +} + +struct Fixture { + _root: tempfile::TempDir, + node: Arc, + manager: ReadReplicaManager, + peers: Vec, + source: CellRuntime, + handles: Vec, +} +async fn fixture() -> Fixture { + fixture_with_store(Store::new(Arc::new(InMemory::new()))).await +} + +async fn fixture_with_store(store: Store) -> Fixture { + let mut app = cellule_app::ApplicationBuilder::new( + "host-test", + BuildDescriptor { + source_revision: "reader-close".into(), + cargo_lock_digest: Digest::from_bytes([7; 32]), + }, + ) + .unwrap(); + app.register(ReaderModule).unwrap(); + let namespace = NamespaceId::from_bytes([2; 16]); + app.cell_type( + cellule_app::CellType::new("host-test", "host-test", namespace, CatalogRole::Sql, 1) + .unwrap(), + ) + .unwrap(); + let app = Arc::new(app.finish().unwrap()); + let code = app.registry().module_digests()[0]; + let layout = CellStorageLayout::new(store, Path::from("reader-closure"), [3; 16]); + let directory = NodeDirectory::new( + layout.clone(), + Digest::from_bytes([2; 32]), + Digest::from_bytes([4; 32]), + Digest::from_bytes([5; 32]), + ); + let now = super::inventory::clock(); + for id in [1, 2] { + directory + .create(super::inventory::advertisement(id, code, now), now) + .await + .unwrap(); + } + let root = tempfile::tempdir().unwrap(); + let limits = ReplicaLimits { + max_database_bytes: 64 << 20, + max_capture_bytes: 16 << 20, + ..ReplicaLimits::default() + }; + let node = Arc::new( + CellNodeBuilder::new(app) + .with_runtime( + SqlWorkerPool::new(2, 8) + .unwrap() + .with_native_memory_limit(128 << 20) + .unwrap(), + 16 << 20, + ) + .with_replica_host( + ReplicaHost::default().with_local_disk_budget(DiskBudget::new(8 << 30)), + ) + .with_session(SessionId::from_bytes([2; 16])) + .build() + .unwrap(), + ); + node.install_task_group(CancellationToken::new(), CancellationToken::new()) + .unwrap(); + let manager = node + .install_read_replicas( + layout.clone(), + directory, + root.path().join("readers"), + limits, + ) + .unwrap(); + node.install_node_lease(NodeLeaseGuard::new(now, now + 60_000).unwrap()) + .unwrap(); + let source = CellRuntime::new_with_replica_host( + SqlWorkerPool::new(2, 8).unwrap(), + 16 << 20, + SessionId::from_bytes([1; 16]), + ReplicaHost::default().with_local_disk_budget(DiskBudget::new(8 << 30)), + ) + .unwrap(); + let catalog = CellCatalog::new(layout.clone(), TenantId::from_bytes([1; 16])); + let authority = CellAuthority::new(layout.clone()); + let mut handles = Vec::new(); + let mut peers = Vec::new(); + for id in 1..=2 { + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + ApplicationId::from_bytes([3; 16]), + namespace, + &[id], + ) + .unwrap(); + let incarnation = IncarnationId::from_bytes([id; 16]); + let proof = catalog + .provision(CatalogEntry::new(&target, CatalogRole::Sql, code, 1).unwrap()) + .await + .unwrap(); + let observed = authority + .create_initial( + &proof, + incarnation, + Owner { + session: SessionId::from_bytes([1; 16]), + endpoint: "https://node.internal:8789".into(), + }, + ) + .await + .unwrap(); + let replica = CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + limits, + ) + .unwrap(); + handles.push( + source + .bootstrap( + proof, + replica, + authority.clone(), + observed, + root.path().join(format!("source-{id}.sqlite")), + |tx| { + tx.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES (17)", + )?; + Ok(()) + }, + ) + .await + .unwrap(), + ); + manager.set_target(&target, 0, 1).await.unwrap().unwrap(); + manager + .activate(target.clone(), SessionId::from_bytes([1; 16])) + .await + .unwrap(); + peers.push(manager.resolve(target).await.unwrap()); + } + Fixture { + _root: root, + node, + manager, + peers, + source, + handles, + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn prepared_opening_is_joined_on_manager_closure_after_native_work_starts() { + let runtime_slot = Arc::new(Mutex::new(None::)); + let slot = runtime_slot.clone(); + let entered = Arc::new(Notify::new()); + let signal = entered.clone(); + let (release, receive) = std::sync::mpsc::channel(); + let gate = Mutex::new(Some(receive)); + let armed = Arc::new(AtomicBool::new(false)); + let once = armed.clone(); + let fixture = fixture_with_store( + Store::new(Arc::new(InMemory::new())).with_read_request_observer(Arc::new(move |kind| { + // Pause the VFS page fault in an actual admitted native open. The + // source publisher and async preparation use different SQL ledgers. + if kind == cellule_store::StorageReadKind::Range + && slot + .lock() + .unwrap() + .as_ref() + .is_some_and(|runtime| runtime.stats().worker_jobs() == 1) + && !once.swap(true, Ordering::AcqRel) + { + let receive = gate.lock().unwrap().take().unwrap(); + signal.notify_one(); + // Dropping the test's sender also releases a failed assertion. + let _ = receive.recv(); + } + })), + ) + .await; + *runtime_slot.lock().unwrap() = Some(fixture.node.runtime()); + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + ApplicationId::from_bytes([3; 16]), + NamespaceId::from_bytes([2; 16]), + &[1], + ) + .unwrap(); + fixture.manager.remove(target.cell_id()).await.unwrap(); + let now = super::inventory::clock(); + fixture.handles[0] + .execute( + cellule_runtime::MutationIdentity { + request_id: cellule_runtime::identity::RequestId::from_bytes([213; 16]), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }, + Digest::from_bytes([214; 32]), + now, + 64, + 64, + |tx| { + // A changed schema page cannot hit the old view's page cache. + tx.execute_batch("CREATE TABLE extra(value INTEGER)")?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await + .unwrap(); + let source = fixture + .manager + .prepare_source(target, SessionId::from_bytes([1; 16])) + .await + .unwrap(); + let manager = fixture.manager.clone(); + let opening = tokio::spawn(async move { manager.activate_source(source).await }); + let entered = tokio::time::timeout(Duration::from_secs(3), entered.notified()).await; + let mut closing = Box::pin(fixture.manager.shutdown()); + let pending = futures_util::poll!(closing.as_mut()).is_pending(); + let before = fixture.node.stats(); + let retained_open = !opening.is_finished(); + // Release every accepted job before checking any fixture assertion. + let _ = release.send(()); + let result = opening.await.unwrap(); + closing.await.unwrap(); + fixture.node.shutdown().await.unwrap(); + runtime_slot.lock().unwrap().take(); + for handle in fixture.handles { + handle.drain().await.unwrap(); + } + fixture.source.shutdown().await.unwrap(); + assert!(entered.is_ok() && armed.load(Ordering::Acquire) && pending && retained_open); + assert_eq!(before.worker_jobs(), 1); + assert!(matches!(result, Err(Error::RuntimeClosed))); + assert_eq!(fixture.node.stats().resident_bytes(), 0); + assert_eq!(fixture.node.stats().worker_jobs(), 0); + assert_eq!(fixture.node.stats().file_descriptors(), 0); + assert_eq!(fixture.node.stats().local_disk_reserved_bytes(), 0); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn prepared_host_activation_pins_the_enrollment_root_and_never_refreshes_an_existing_view() { + let fixture = fixture().await; + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + ApplicationId::from_bytes([3; 16]), + NamespaceId::from_bytes([2; 16]), + &[1], + ) + .unwrap(); + fixture.manager.remove(target.cell_id()).await.unwrap(); + let source = fixture + .manager + .prepare_source(target.clone(), SessionId::from_bytes([1; 16])) + .await + .unwrap(); + assert_eq!(source.target(), &target); + let before = fixture.node.stats(); + let page = fixture + .manager + .fleet_readers_page(None, 128, super::inventory::clock()) + .await + .unwrap(); + assert_eq!(page.total_views(), 1); + drop(page); + let now = super::inventory::clock(); + fixture.handles[0] + .execute( + cellule_runtime::MutationIdentity { + request_id: cellule_runtime::identity::RequestId::from_bytes([211; 16]), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }, + Digest::from_bytes([212; 32]), + now, + 64, + 64, + |tx| { + tx.execute("UPDATE counter SET value=18", [])?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await + .unwrap(); + assert_eq!( + fixture.node.stats().resident_bytes(), + before.resident_bytes() + ); + let receipt = fixture + .manager + .activate_source(source.clone()) + .await + .unwrap(); + assert_eq!(receipt.commit_sequence, source.root().commit_sequence); + let peer = fixture.manager.resolve(target.clone()).await.unwrap(); + assert_eq!(peer.query::(None, 0).await.unwrap().output, 17); + assert!(matches!( + fixture.manager.activate_source(source.clone()).await, + Err(Error::Control("read view is already installed")) + )); + assert_eq!(peer.query::(None, 0).await.unwrap().output, 17); + let refreshed = fixture + .manager + .activate(target.clone(), SessionId::from_bytes([1; 16])) + .await + .unwrap(); + assert!(refreshed.commit_sequence > receipt.commit_sequence); + assert_eq!(peer.query::(None, 0).await.unwrap().output, 18); + assert!(matches!( + fixture + .manager + .prepare_source(target.clone(), SessionId::from_bytes([3; 16])) + .await, + Err(Error::Fenced) + )); + fixture.node.runtime().node_admission().cordon().unwrap(); + fixture.manager.remove(target.cell_id()).await.unwrap(); + assert!(matches!( + fixture.manager.activate_source(source).await, + Err(Error::CellDraining) + )); + fixture.node.shutdown().await.unwrap(); + let stats = fixture.node.stats(); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.retained_bytes(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.file_descriptors(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); + for handle in fixture.handles { + handle.drain().await.unwrap(); + } + fixture.source.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn cancelled_reader_removal_and_shutdown_keep_owned_views_until_native_queries_join() { + for shutdown in [false, true] { + let fixture = fixture().await; + let pause = Pause::new(); + let token = pause.id; + let peer = fixture.peers[0].clone(); + let query = tokio::spawn(async move { peer.query::(None, token).await }); + pause.entered().await; + let cell = fixture.peers[0].receipt().await.cell; + let first_pending = if shutdown { + let close = fixture.manager.shutdown(); + tokio::pin!(close); + futures_util::poll!(close.as_mut()).is_pending() + } else { + let close = fixture.manager.remove(cell); + tokio::pin!(close); + futures_util::poll!(close.as_mut()).is_pending() + }; + let page = fixture + .manager + .fleet_readers_page(None, 128, super::inventory::clock()) + .await + .unwrap(); + let retained_count = page.total_views(); + let terminal = page.closed(); + let joining = *page + .entries() + .iter() + .find(|entry| entry.receipt().cell == cell) + .unwrap(); + drop(page); + let deadline_failed = if shutdown { + fixture + .node + .drain_until(Some(Instant::now() + Duration::from_millis(25))) + .await + .is_err() + } else { + false + }; + let before = fixture.node.stats(); + let state = fixture.node.state(); + // A red assertion cannot strand a native SQL callback or accepted drain. + pause.release(); + assert!(matches!(query.await.unwrap(), Err(Error::Fenced))); + if !shutdown { + fixture.manager.remove(cell).await.unwrap(); + } + fixture.node.shutdown().await.unwrap(); + assert!(first_pending); + assert_eq!(retained_count, 2); + assert_eq!(terminal, shutdown); + assert!(joining.admission_closed()); + assert!(!joining.snapshot_attached()); + assert!(joining.retained_lifetimes() > 0); + assert!(!joining.locally_joined()); + assert_eq!(before.worker_jobs(), 1); + assert!(before.resident_bytes() > 0); + if shutdown { + assert!(deadline_failed); + assert_eq!(state, NodeState::Draining); + } + assert_eq!(fixture.node.state(), NodeState::Stopped); + let stats = fixture.node.stats(); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.retained_bytes(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.file_descriptors(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); + for peer in &fixture.peers { + assert!(peer.readiness().await.is_err()); + let observation = peer.lifecycle_observation().await; + assert!(observation.locally_joined()); + assert_eq!(observation.retained_lifetimes(), 0); + } + for handle in fixture.handles { + handle.drain().await.unwrap(); + } + fixture.source.shutdown().await.unwrap(); + } +} + +#[tokio::test] +async fn reader_lifetime_observation_keeps_cancelled_native_queries_visible_after_detachment() { + let fixture = fixture().await; + let pause = Pause::new(); + let token = pause.id; + let retained_peer = fixture.peers[0].clone(); + let querying = retained_peer.clone(); + let query = tokio::spawn(async move { querying.query::(None, token).await }); + pause.entered().await; + let open = retained_peer.lifecycle_observation().await; + query.abort(); + let cancelled = query.await.unwrap_err().is_cancelled(); + let mut close = Box::pin(retained_peer.close_and_join()); + let pending = futures_util::poll!(close.as_mut()).is_pending(); + let joining = retained_peer.lifecycle_observation().await; + let page = fixture + .manager + .fleet_readers_page(None, 128, super::inventory::clock()) + .await + .unwrap(); + let managed = *page + .entries() + .iter() + .find(|entry| entry.receipt().cell == open.receipt().cell) + .unwrap(); + drop(page); + let resources = fixture.node.stats(); + // Release the native callback before assertions so a red observation cannot + // strand its original SQL job, lifetime or shutdown join. + pause.release(); + let joined_receipt = close.await; + let joined = retained_peer.lifecycle_observation().await; + let refused = retained_peer.query::(None, 0).await; + fixture.node.shutdown().await.unwrap(); + let after_shutdown = retained_peer.lifecycle_observation().await; + for handle in fixture.handles { + handle.drain().await.unwrap(); + } + fixture.source.shutdown().await.unwrap(); + assert!(cancelled && pending); + assert!(!open.admission_closed() && open.snapshot_attached()); + assert!(open.retained_lifetimes() >= 2 && !open.locally_joined()); + assert!(joining.admission_closed() && !joining.snapshot_attached()); + assert!(joining.retained_lifetimes() > 0 && !joining.locally_joined()); + assert_eq!(managed, joining); + assert_eq!(resources.worker_jobs(), 1); + assert!(resources.resident_bytes() > 0); + assert_eq!(joined_receipt, open.receipt()); + assert!(joined.locally_joined()); + assert_eq!(joined.retained_lifetimes(), 0); + assert!(matches!(refused, Err(Error::Fenced))); + assert_eq!(after_shutdown, joined); + let resources = fixture.node.stats(); + assert_eq!(resources.active_cells(), 0); + assert_eq!(resources.retained_bytes(), 0); + assert_eq!(resources.resident_bytes(), 0); + assert_eq!(resources.worker_jobs(), 0); + assert_eq!(resources.file_descriptors(), 0); + assert_eq!(resources.local_disk_reserved_bytes(), 0); +} + +#[tokio::test] +async fn reader_inventory_continuation_rejects_peer_closure_outside_the_returned_page() { + let fixture = fixture().await; + let mut peers = Vec::new(); + for peer in &fixture.peers { + peers.push((peer.receipt().await.cell, peer)); + } + peers.sort_by_key(|(cell, _)| *cell.as_bytes()); + let (cell, peer) = peers.last().unwrap(); + let page = fixture + .manager + .fleet_readers_page(None, 1, super::inventory::clock()) + .await + .unwrap(); + let cursor = page.next().unwrap(); + let original_topology = page.topology(); + assert_ne!(page.entries()[0].receipt().cell, *cell); + drop(page); + peer.close(); + let closed_rejected = fixture + .manager + .fleet_readers_page(Some(cursor), 1, super::inventory::clock()) + .await + .is_err(); + let closed = fixture + .manager + .fleet_readers_page(None, 1, super::inventory::clock()) + .await + .unwrap(); + let closed_topology = closed.topology(); + let cursor = closed.next().unwrap(); + drop(closed); + let attached = peer.lifecycle_observation().await; + peer.close_and_join().await; + let detached_rejected = fixture + .manager + .fleet_readers_page(Some(cursor), 1, super::inventory::clock()) + .await + .is_err(); + let joined = fixture + .manager + .fleet_readers_page(None, 1, super::inventory::clock()) + .await + .unwrap(); + let joined_topology = joined.topology(); + let cursor = joined.next().unwrap(); + drop(joined); + let tail = fixture + .manager + .fleet_readers_page(Some(cursor), 1, super::inventory::clock()) + .await + .unwrap(); + let total = tail.total_views(); + let observation = tail.entries()[0]; + let terminal = tail.next().is_none(); + drop(tail); + fixture.node.shutdown().await.unwrap(); + for handle in fixture.handles { + handle.drain().await.unwrap(); + } + fixture.source.shutdown().await.unwrap(); + assert!(closed_rejected && detached_rejected); + assert_ne!(original_topology, closed_topology); + assert_ne!(closed_topology, joined_topology); + assert!(attached.admission_closed() && attached.snapshot_attached()); + assert!(!attached.locally_joined()); + assert!(observation.locally_joined() && terminal); + assert_eq!(total, 2); + let stats = fixture.node.stats(); + assert_eq!(stats.retained_bytes(), 0); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.file_descriptors(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); +} + +#[tokio::test] +async fn reader_inventory_continuation_tracks_native_work_outside_the_returned_page() { + let fixture = fixture().await; + let mut peers = Vec::new(); + for peer in &fixture.peers { + peers.push((peer.receipt().await.cell, peer.clone())); + } + peers.sort_by_key(|(cell, _)| *cell.as_bytes()); + let (_, peer) = peers.last().unwrap(); + let page = fixture + .manager + .fleet_readers_page(None, 1, super::inventory::clock()) + .await + .unwrap(); + let original = page.topology(); + let original_cursor = page.next().unwrap(); + drop(page); + let pause = Pause::new(); + let token = pause.id; + let querying = peer.clone(); + let query = tokio::spawn(async move { querying.query::(None, token).await }); + pause.entered().await; + let start_rejected = fixture + .manager + .fleet_readers_page(Some(original_cursor), 1, super::inventory::clock()) + .await + .is_err(); + let busy = fixture + .manager + .fleet_readers_page(None, 1, super::inventory::clock()) + .await + .unwrap(); + let busy_topology = busy.topology(); + let busy_cursor = busy.next().unwrap(); + drop(busy); + let tail = fixture + .manager + .fleet_readers_page(Some(busy_cursor), 1, super::inventory::clock()) + .await; + let busy_observation = tail.as_ref().ok().map(|page| page.entries()[0]); + drop(tail); + // Release the original native callback before assertions or query joining. + pause.release(); + let result = query.await.unwrap(); + let quiesced = tokio::time::timeout(Duration::from_secs(3), async { + loop { + let observation = peer.lifecycle_observation().await; + if observation.retained_lifetimes() == 1 { + break observation; + } + tokio::task::yield_now().await; + } + }) + .await; + let end_rejected = fixture + .manager + .fleet_readers_page(Some(busy_cursor), 1, super::inventory::clock()) + .await + .is_err(); + let idle = fixture + .manager + .fleet_readers_page(None, 1, super::inventory::clock()) + .await + .unwrap(); + let idle_topology = idle.topology(); + drop(idle); + fixture.node.shutdown().await.unwrap(); + for handle in fixture.handles { + handle.drain().await.unwrap(); + } + fixture.source.shutdown().await.unwrap(); + assert!(start_rejected && end_rejected); + assert_ne!(original, busy_topology); + assert_eq!(original, idle_topology); // Matching intervals do not prove atomicity. + assert_eq!(result.unwrap().output, 17); + let busy_observation = busy_observation.unwrap(); + assert!(busy_observation.retained_lifetimes() > 1); + let idle_observation = quiesced.unwrap(); + assert!(!idle_observation.admission_closed() && idle_observation.snapshot_attached()); + assert_eq!(busy_observation.receipt(), idle_observation.receipt()); + let stats = fixture.node.stats(); + assert_eq!(stats.retained_bytes(), 0); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.file_descriptors(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); +} + +#[tokio::test] +async fn reader_inventory_continuation_rejects_peer_refresh_outside_the_returned_page() { + let fixture = fixture().await; + let mut peers = Vec::new(); + for (index, peer) in fixture.peers.iter().enumerate() { + peers.push((peer.receipt().await.cell, index)); + } + peers.sort_by_key(|(cell, _)| *cell.as_bytes()); + let (_, index) = *peers.last().unwrap(); + let peer = &fixture.peers[index]; + let page = fixture + .manager + .fleet_readers_page(None, 1, super::inventory::clock()) + .await + .unwrap(); + let original = page.topology(); + let cursor = page.next().unwrap(); + drop(page); + let before = peer.receipt().await; + let now = super::inventory::clock(); + fixture.handles[index] + .execute( + cellule_runtime::MutationIdentity { + request_id: cellule_runtime::identity::RequestId::from_bytes([213; 16]), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }, + Digest::from_bytes([214; 32]), + now, + 64, + 64, + |tx| { + tx.execute("UPDATE counter SET value=19", [])?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await + .unwrap(); + let unpublished_here = fixture + .manager + .fleet_readers_page(Some(cursor), 1, super::inventory::clock()) + .await + .unwrap(); + let still_original = unpublished_here.topology(); + drop(unpublished_here); + let after = peer + .refresh(&fixture._root.path().join("external-refresh.sqlite")) + .await + .unwrap(); + let refused = fixture + .manager + .fleet_readers_page(Some(cursor), 1, super::inventory::clock()) + .await + .is_err(); + let fresh = fixture + .manager + .fleet_readers_page(None, 1, super::inventory::clock()) + .await + .unwrap(); + let fresh_topology = fresh.topology(); + let fresh_cursor = fresh.next().unwrap(); + drop(fresh); + let tail = fixture + .manager + .fleet_readers_page(Some(fresh_cursor), 1, super::inventory::clock()) + .await + .unwrap(); + let position = tail.entries()[0].receipt(); + drop(tail); + let value = peer.query::(None, 0).await.unwrap().output; + fixture.node.shutdown().await.unwrap(); + for handle in fixture.handles { + handle.drain().await.unwrap(); + } + fixture.source.shutdown().await.unwrap(); + assert!(refused); + assert_eq!(original, still_original); + assert_ne!(original, fresh_topology); + assert_eq!(before.cell, after.cell); + assert_eq!(before.incarnation, after.incarnation); + assert!(after.commit_sequence > before.commit_sequence); + assert_eq!(position, after); + assert_eq!(value, 19); + let stats = fixture.node.stats(); + assert_eq!(stats.retained_bytes(), 0); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.file_descriptors(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); +} diff --git a/crates/cellule-host/tests/node/tasks.rs b/crates/cellule-host/tests/node/tasks.rs index 07559a50..095d92df 100644 --- a/crates/cellule-host/tests/node/tasks.rs +++ b/crates/cellule-host/tests/node/tasks.rs @@ -6,10 +6,226 @@ use cellule_runtime::follower::FollowerStore; use cellule_runtime::identity::NodeId; use cellule_runtime::ltx::DiskBudget; use cellule_runtime::node::durability::NodeLogAuthority; -use cellule_runtime::node::log::NodeLogRotationBarrier; +use cellule_runtime::node::log::NodeLogRetirementObservation; use cellule_runtime::node::log_transport::{LocalFollowerTransport, NodeLogTransport}; use std::sync::atomic::AtomicUsize; +fn error_source<'a, E: std::error::Error + 'static>( + error: &'a (dyn std::error::Error + 'static), +) -> &'a E { + let mut source = error; + loop { + if let Some(error) = source.downcast_ref::() { + return error; + } + source = source.source().expect("original error must be retained"); + } +} + +#[tokio::test] +async fn repeated_shutdown_retains_task_failure_and_does_not_withdraw_lease() { + let node = CellNodeBuilder::new(application()) + .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) + .with_replica_host(ReplicaHost::default()) + .with_session(SessionId::from_bytes([34; 16])) + .build() + .unwrap(); + let node_shutdown = CancellationToken::new(); + let tasks = node + .install_task_group(CancellationToken::new(), node_shutdown.clone()) + .unwrap(); + let withdrawn = Arc::new(AtomicBool::new(false)); + let lease_withdrawn = Arc::clone(&withdrawn); + tasks + .spawn_lease_maintenance(async move { + node_shutdown.cancelled().await; + lease_withdrawn.store(true, Ordering::Release); + Ok::<(), Error>(()) + }) + .unwrap(); + let sibling_joined = Arc::new(AtomicBool::new(false)); + let sibling_finished = Arc::clone(&sibling_joined); + let cancellation = tasks.cancellation_token(); + tasks + .spawn(async move { + cancellation.cancelled().await; + sibling_finished.store(true, Ordering::Release); + Ok::<(), Error>(()) + }) + .unwrap(); + tasks + .spawn(async { Err::<(), _>(std::io::Error::other("original work failure")) }) + .unwrap(); + + let first = node.shutdown().await.unwrap_err(); + let second = node.shutdown().await; + let state = node.state(); + let lease_was_withdrawn = withdrawn.load(Ordering::Acquire); + // Explicitly join the lease task even when an assertion detects the old + // false-success behavior. Node shutdown itself must keep it live. + let cleanup = tasks.drain().await; + + let second = second.expect_err("a consumed join must not erase the original failure"); + assert_eq!(state, NodeState::Draining); + assert!(!lease_was_withdrawn); + assert!(sibling_joined.load(Ordering::Acquire)); + assert!(withdrawn.load(Ordering::Acquire)); + let first_source = error_source::(&first); + assert_eq!(first_source.to_string(), "original work failure"); + assert!(std::ptr::eq( + first_source, + error_source::(&second) + )); + assert!(std::ptr::eq( + first_source, + error_source::(cleanup.unwrap_err().as_ref()) + )); +} + +#[tokio::test] +async fn cancelled_task_drain_waiter_retains_the_original_job_and_failure() { + let tasks = CellNodeTaskGroup::new(CancellationToken::new(), CancellationToken::new()); + let started = Arc::new(tokio::sync::Notify::new()); + let gate = Arc::new(tokio::sync::Notify::new()); + let finished = Arc::new(AtomicBool::new(false)); + let task_started = Arc::clone(&started); + let task_gate = Arc::clone(&gate); + let task_finished = Arc::clone(&finished); + tasks + .spawn(async move { + task_started.notify_one(); + task_gate.notified().await; + task_finished.store(true, Ordering::Release); + Err::<(), _>(std::io::Error::other( + "accepted task completed after cancellation", + )) + }) + .unwrap(); + started.notified().await; + + let mut waiter = Box::pin(tasks.drain()); + assert!(futures_util::poll!(&mut waiter).is_pending()); + drop(waiter); + assert!(!finished.load(Ordering::Acquire)); + gate.notify_one(); + let first = tasks.drain().await.unwrap_err(); + assert!(finished.load(Ordering::Acquire)); + let second = tasks.drain().await.unwrap_err(); + let source = error_source::(first.as_ref()); + assert_eq!( + source.to_string(), + "accepted task completed after cancellation" + ); + assert!(std::ptr::eq( + source, + error_source::(second.as_ref()) + )); +} + +#[tokio::test] +async fn concurrent_task_drains_share_the_original_join_and_failure() { + let tasks = CellNodeTaskGroup::new(CancellationToken::new(), CancellationToken::new()); + let started = Arc::new(tokio::sync::Notify::new()); + let gate = Arc::new(tokio::sync::Notify::new()); + let completed = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let task_started = Arc::clone(&started); + let task_gate = Arc::clone(&gate); + let task_completed = Arc::clone(&completed); + tasks + .spawn(async move { + task_started.notify_one(); + task_gate.notified().await; + task_completed.fetch_add(1, Ordering::AcqRel); + Err::<(), _>(std::io::Error::other("one shared failure")) + }) + .unwrap(); + started.notified().await; + + let mut waiters = Box::pin(async { tokio::join!(tasks.drain(), tasks.drain()) }); + assert!(futures_util::poll!(&mut waiters).is_pending()); + assert_eq!(completed.load(Ordering::Acquire), 0); + gate.notify_one(); + let (first, second) = waiters.await; + let first = first.unwrap_err(); + let second = second.unwrap_err(); + assert_eq!(completed.load(Ordering::Acquire), 1); + assert!(std::ptr::eq( + error_source::(first.as_ref()), + error_source::(second.as_ref()) + )); +} + +#[tokio::test] +async fn task_deadline_retains_and_joins_original_cancellation() { + struct DropProbe(Arc); + impl Drop for DropProbe { + fn drop(&mut self) { + self.0.store(true, Ordering::Release); + } + } + + let tasks = CellNodeTaskGroup::new(CancellationToken::new(), CancellationToken::new()); + let started = Arc::new(tokio::sync::Notify::new()); + let dropped = Arc::new(AtomicBool::new(false)); + let task_started = Arc::clone(&started); + let task_dropped = Arc::clone(&dropped); + tasks + .spawn(async move { + let _probe = DropProbe(task_dropped); + task_started.notify_one(); + std::future::pending::<()>().await; + Ok::<(), Error>(()) + }) + .unwrap(); + started.notified().await; + + let timeout = tasks + .drain_until(Some(Instant::now() + Duration::from_millis(10))) + .await + .unwrap_err(); + assert_eq!( + error_source::(timeout.as_ref()).kind(), + std::io::ErrorKind::TimedOut + ); + let first = tasks.drain().await.unwrap_err(); + assert!(dropped.load(Ordering::Acquire)); + let source = error_source::(first.as_ref()); + assert!(source.is_cancelled()); + let second = tasks.drain().await.unwrap_err(); + assert!(std::ptr::eq( + source, + error_source::(second.as_ref()) + )); +} + +#[tokio::test] +async fn task_panic_is_retained_across_every_drain_and_siblings_join() { + let tasks = CellNodeTaskGroup::new(CancellationToken::new(), CancellationToken::new()); + let finished = Arc::new(AtomicBool::new(false)); + let task_finished = Arc::clone(&finished); + let cancellation = tasks.cancellation_token(); + tasks + .spawn(async move { + cancellation.cancelled().await; + task_finished.store(true, Ordering::Release); + Ok::<(), Error>(()) + }) + .unwrap(); + tasks + .spawn_boxed(async { panic!("original supervised panic") }) + .unwrap(); + + let first = tasks.drain().await.unwrap_err(); + assert!(finished.load(Ordering::Acquire)); + let source = error_source::(first.as_ref()); + assert!(source.is_panic()); + let second = tasks.drain().await.unwrap_err(); + assert!(std::ptr::eq( + source, + error_source::(second.as_ref()) + )); +} + struct NoopNodeDurabilityProvider; impl NodeDurabilityProvider for NoopNodeDurabilityProvider { @@ -54,7 +270,7 @@ impl NodeLogAuthority for TrackingAuthority { fn close<'a>( &'a self, - _barrier: &'a NodeLogRotationBarrier, + _barrier: &'a NodeLogRetirementObservation, ) -> futures_util::future::BoxFuture<'a, cellule_runtime::Result<()>> { self.0.fetch_add(1, Ordering::SeqCst); Box::pin(async { Ok(()) }) @@ -149,7 +365,7 @@ async fn provider_membership_change_rotates_the_host_owned_log_epoch() { let node = CellNodeBuilder::new(application()) .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([40; 16])) + .with_session(SessionId::from_bytes([41; 16])) .build() .unwrap(); node.install_task_group(CancellationToken::new(), CancellationToken::new()) diff --git a/crates/cellule-ltx/api-prelude.txt b/crates/cellule-ltx/api-prelude.txt index 97057ed9..8005f529 100644 --- a/crates/cellule-ltx/api-prelude.txt +++ b/crates/cellule-ltx/api-prelude.txt @@ -36,6 +36,9 @@ ReadOnlyRoot RecoveryOverlay Result RootObjectRef +RootPreparation +RootPreparationFuture +RootPreparationMetadata RootRef ScratchMonitor SegmentInfo diff --git a/crates/cellule-ltx/docs/publication.md b/crates/cellule-ltx/docs/publication.md index 1a8c95ac..fd358cd5 100644 --- a/crates/cellule-ltx/docs/publication.md +++ b/crates/cellule-ltx/docs/publication.md @@ -23,6 +23,7 @@ sequenceDiagram | `prepare_bundle` | Selects this Cell's exact rows from a shared bundle. | | `prepare_compaction` | Rewrites representation without changing logical state. | | `prepare_after_compaction` | Appends to a private compaction while retaining its original authority predecessor. | +| `with_root_metadata` | Joins caller-supplied verified derivation metadata with immutable uploads; both must succeed before `PreparedRoot` returns. | | Runtime CAS | Names the authoritative owner and exact root. | | `Db::prune_captured` | Removes only the successfully published batch. | @@ -31,9 +32,37 @@ retry pins and rechecks the selected capture bytes, so path replacement cannot change an in-flight proposal. The host owns request admission, deadlines, and reconciliation after ambiguous results. +`RootPreparation` identifies a native verified derivation while uploads may still +be running. Its construction is private and it grants no uploaded-root, restore, +serving, authority or acknowledgement rights. A `RootPreparationMetadata` future +runs inline under the enclosing preparation owner and shares its native origin +I/O admission; cancellation drops it and releases its permit. It adds no task or +scheduler. Its source error is retained as `RootPreparation`, +classified Ambiguous for caller reconciliation. The runtime recovers its original +typed storage error and uses the existing publisher retry policy. Failed work +can leave proposal metadata or immutable objects; neither selects authority. + A representation-only compaction can remain private while its successor append uploads. `prepare_after_compaction` verifies that the compaction preserves the predecessor's position, commit sequence, Cell and incarnation. The runtime selects the append's schema and can choose the final root with one CAS against the original authority record. Every immutable dependency still finishes uploading before the successor proposal is returned. + +The verified compaction composition sets the original predecessor in the native +factory before derivation metadata runs. Metadata and the complete proposal name +the same input. That private preparation context is removed from the immutable +read view; subsequent preparations cannot inherit an earlier rebase. + +## Verify current origin dependencies + +`CellReplica::reachable_objects` authenticates the exact root's complete current +origin graph, including root metadata, descriptor pages, directory coverage and +all body/index bytes. A process cache cannot certify availability. +`reachable_objects_bounded` uses the same walk with a caller-selected maximum +inventory count. Zero or excess objects refuse with +`LimitKind::RootInventoryObjects`; callers never receive a truncated inventory. +Directory digests obey that bound while descriptor work keeps the existing fixed +root/segment ceilings. The caller owns memory admission, bounded Store stream +chunks and the enclosing deadline. This graph proof grants no selected authority, +retention pin or current serving. diff --git a/crates/cellule-ltx/docs/safety.md b/crates/cellule-ltx/docs/safety.md index 7898dd4e..67a15bbc 100644 --- a/crates/cellule-ltx/docs/safety.md +++ b/crates/cellule-ltx/docs/safety.md @@ -27,6 +27,35 @@ service authenticates the authority record that selects a root. admission bound local scratch, retained cuts, and remote I/O. The host owns scheduling and cancellation; see [cellule-host](../../cellule-host/docs/README.md). +## Prepared disk credit + +An operation can reserve its conservative disk envelope before another node +releases ownership. Convert that `DiskReservation` with `into_budget`, then +give the resulting budget to the operation's normal `Host`. Restore, checksum, +SQLite, and capture reservations consume the already admitted envelope. The +parent node budget remains charged for the full envelope during preparation. + +```rust +fn prepared_host( + host: cellule_ltx::Host, + node_budget: &cellule_ltx::DiskBudget, + bytes: u64, +) -> cellule_ltx::Result<(cellule_ltx::Host, cellule_ltx::DiskBudget)> { + let credit = node_budget.try_reserve(bytes)?.into_budget(); + Ok((host.with_local_disk_budget(credit.clone()), credit)) +} +``` + +After preparation work has joined, call `finish_preparation` to return unused +credit. Live child reservations stay charged; future growth uses ordinary +parent admission and the original operation ceiling. Dropping the initiating +future or budget clone does not release credit held by accepted work. Scope +budgets inherit parent admission and reject installation of another aggregate +hook, which could count the same bytes twice or overwrite node accounting. + +This is a local resource token. It supplies no ownership, controller permit, +authentication, or proof that remote work was cancelled. + ```sh cargo test -p cellule-ltx --no-default-features --locked cargo test -p cellule-ltx --features replica --locked diff --git a/crates/cellule-ltx/src/cell_layout.rs b/crates/cellule-ltx/src/cell_layout.rs index 55ebb125..0321e1c9 100644 --- a/crates/cellule-ltx/src/cell_layout.rs +++ b/crates/cellule-ltx/src/cell_layout.rs @@ -110,6 +110,55 @@ impl CellStorageLayout { self.application_path(&format!("cells/{}/control.json", encode_hex(cell))) } + /// Retained runtime owner observation for one Cell incarnation and epoch. + /// This metadata path supplies no authority or immutable-root retention pin. + #[must_use] + pub fn owner_observation_path( + &self, + cell: &[u8; 32], + incarnation: &[u8; 16], + epoch: u64, + ) -> Path { + self.application_path(&format!( + "cells/{}/owner-history/v1/{}/{epoch:016x}.json", + encode_hex(cell), + encode_hex(incarnation), + )) + } + + /// Runtime acquisition input/materialization metadata for one owner epoch. + /// This path grants no authority, native serving or immutable-root pin. + #[must_use] + pub fn acquisition_record_path( + &self, + cell: &[u8; 32], + incarnation: &[u8; 16], + epoch: u64, + ) -> Path { + self.application_path(&format!( + "cells/{}/acquisitions/v1/{}/{epoch:016x}.bin", + encode_hex(cell), + encode_hex(incarnation), + )) + } + + /// Runtime-retained verified preparation links for an exact immutable root. + /// This metadata path grants no authority or immutable-root retention pin. + #[must_use] + pub fn root_lineage_path( + &self, + cell: &[u8; 32], + incarnation: &[u8; 16], + root: &[u8; 32], + ) -> Path { + self.application_path(&format!( + "cells/{}/root-lineage/v1/{}/{}.bin", + encode_hex(cell), + encode_hex(incarnation), + encode_hex(root), + )) + } + /// Returns the advisory desired read-replica count for one Cell. #[must_use] pub fn read_policy_path(&self, cell: &[u8; 32]) -> Path { @@ -285,6 +334,24 @@ mod tests { layout.control_path(&[0xcd; 32]).as_ref(), "tenant-root/cells/v1/apps/abababababababababababababababab/cells/cdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcd/control.json" ); + assert_eq!( + layout + .owner_observation_path(&[0xcd; 32], &[0xef; 16], 10) + .as_ref(), + "tenant-root/cells/v1/apps/abababababababababababababababab/cells/cdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcd/owner-history/v1/efefefefefefefefefefefefefefefef/000000000000000a.json" + ); + assert_eq!( + layout + .acquisition_record_path(&[0xcd; 32], &[0xef; 16], 10) + .as_ref(), + "tenant-root/cells/v1/apps/abababababababababababababababab/cells/cdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcd/acquisitions/v1/efefefefefefefefefefefefefefefef/000000000000000a.bin" + ); + assert_eq!( + layout + .root_lineage_path(&[0xcd; 32], &[0xef; 16], &[0xab; 32]) + .as_ref(), + "tenant-root/cells/v1/apps/abababababababababababababababab/cells/cdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcd/root-lineage/v1/efefefefefefefefefefefefefefefef/abababababababababababababababababababababababababababababababab.bin" + ); assert_eq!( layout.node_directory_path().as_ref(), "tenant-root/cells/v1/nodes" diff --git a/crates/cellule-ltx/src/environment/host/budget.rs b/crates/cellule-ltx/src/environment/host/budget.rs index 8c034f27..2ff24818 100644 --- a/crates/cellule-ltx/src/environment/host/budget.rs +++ b/crates/cellule-ltx/src/environment/host/budget.rs @@ -19,6 +19,14 @@ pub(crate) struct DiskBudgetInner { used: AtomicU64, has_admissions: AtomicBool, admissions: Mutex, + scope: Option>, +} + +struct DiskScope { + // The parent owns the real node charge. Child reservations divide that + // credit without charging the parent a second time during preparation. + backing: DiskReservation, + preparing: bool, } impl fmt::Debug for DiskBudget { @@ -41,6 +49,7 @@ impl DiskBudget { used: AtomicU64::new(0), has_admissions: AtomicBool::new(false), admissions: Mutex::new(Vec::new()), + scope: None, }), } } @@ -51,6 +60,11 @@ impl DiskBudget { /// may observe the same process-wide budget; dead hooks are removed before /// the new hook is registered. pub fn install_admission(&self, admission: Arc) -> crate::Result<()> { + if self.inner.scope.is_some() { + return Err(crate::LtxError::InvalidState( + "reserved disk budget already inherits parent admission", + )); + } let mut current = self .inner .admissions @@ -81,9 +95,9 @@ impl DiskBudget { /// Reserves bytes without waiting or overcommitting the configured capacity. pub fn try_reserve(&self, bytes: u64) -> crate::Result { self.add(bytes)?; - if let Err(error) = self.reconcile_admissions(self.used()) { + if let Err(error) = self.reconcile_admissions() { let _ = self.remove(bytes); - let _ = self.reconcile_admissions(self.used()); + let _ = self.reconcile_admissions(); return Err(error); } Ok(DiskReservation { @@ -110,7 +124,46 @@ impl DiskBudget { self.capacity().saturating_sub(self.used()) } + /// Releases unused preparation credit while retaining every child charge. + /// + /// This budget must come from [`DiskReservation::into_budget`]. Before this + /// call, its parent retains the full prepared envelope. Afterward the parent + /// tracks actual child bytes; later growth reserves additional parent bytes + /// through ordinary admission. The child's ceiling remains unchanged. + /// + /// Call only after the work relying on the prepared envelope has joined. + /// This operation is idempotent and does not cancel any child work. + pub fn finish_preparation(&self) -> crate::Result<()> { + let mut scope = self + .inner + .scope + .as_ref() + .ok_or(crate::LtxError::InvalidState( + "disk budget has no preparation credit", + ))? + .lock() + .map_err(|_| crate::LtxError::InvalidState("disk scope lock poisoned"))?; + scope.backing.resize(self.used())?; + scope.preparing = false; + Ok(()) + } + fn add(&self, bytes: u64) -> crate::Result<()> { + if let Some(scope) = &self.inner.scope { + let scope = scope + .lock() + .map_err(|_| crate::LtxError::InvalidState("disk scope lock poisoned"))?; + let next = self + .used() + .checked_add(bytes) + .filter(|next| *next <= self.capacity()) + .ok_or(crate::LtxError::Limit(crate::LimitKind::LocalDiskBytes))?; + if !scope.preparing { + scope.backing.resize(next)?; + } + self.inner.used.store(next, Ordering::Release); + return Ok(()); + } self.inner .used .try_update(Ordering::AcqRel, Ordering::Acquire, |used| { @@ -121,11 +174,13 @@ impl DiskBudget { .map_err(|_| crate::LtxError::Limit(crate::LimitKind::LocalDiskBytes)) } - fn reconcile_admissions(&self, bytes: u64) -> crate::Result<()> { + fn reconcile_admissions(&self) -> crate::Result<()> { let Some(mut admissions) = self.live_admissions()? else { return Ok(()); }; - Self::reconcile_admissions_locked(&mut admissions, bytes) + // Capture usage after acquiring the hook lane: a delayed earlier + // caller must not overwrite a newer aggregate with a stale sample. + Self::reconcile_admissions_locked(&mut admissions, self.used()) } fn live_admissions(&self) -> crate::Result>> { @@ -166,6 +221,22 @@ impl DiskBudget { } fn remove(&self, bytes: u64) -> crate::Result<()> { + if let Some(scope) = &self.inner.scope { + let scope = scope + .lock() + .map_err(|_| crate::LtxError::InvalidState("disk scope lock poisoned"))?; + let next = self + .used() + .checked_sub(bytes) + .ok_or(crate::LtxError::InvalidState( + "local disk reservation underflow", + ))?; + if !scope.preparing { + scope.backing.resize(next)?; + } + self.inner.used.store(next, Ordering::Release); + return Ok(()); + } self.inner .used .try_update(Ordering::AcqRel, Ordering::Acquire, |used| { @@ -192,6 +263,31 @@ impl fmt::Debug for DiskReservation { } impl DiskReservation { + /// Transfers this admitted envelope into a bounded budget for one operation. + /// + /// Its child reservations use the already charged bytes. All clones and + /// child tokens retain the parent credit until they close, including work + /// that outlives its initiating future. Parent capacity is never released + /// and reacquired during this conversion. + /// + /// Once preparation has joined, [`DiskBudget::finish_preparation`] returns + /// unused bytes and leaves lasting child charges on ordinary admission. + #[must_use] + pub fn into_budget(self) -> DiskBudget { + DiskBudget { + inner: Arc::new(DiskBudgetInner { + capacity: self.bytes(), + used: AtomicU64::new(0), + has_admissions: AtomicBool::new(false), + admissions: Mutex::new(Vec::new()), + scope: Some(Mutex::new(DiskScope { + backing: self, + preparing: true, + })), + }), + } + } + /// Adds bytes to this reservation without exceeding the shared budget. pub fn try_grow(&self, bytes: u64) -> crate::Result<()> { let mut held = match self.bytes.lock() { @@ -202,9 +298,9 @@ impl DiskReservation { .checked_add(bytes) .ok_or(crate::LtxError::Limit(crate::LimitKind::LocalDiskBytes))?; self.budget.add(bytes)?; - if let Err(error) = self.budget.reconcile_admissions(self.budget.used()) { + if let Err(error) = self.budget.reconcile_admissions() { let _ = self.budget.remove(bytes); - let _ = self.budget.reconcile_admissions(self.budget.used()); + let _ = self.budget.reconcile_admissions(); return Err(error); } *held = next; @@ -221,9 +317,9 @@ impl DiskReservation { if bytes > current { let added = bytes - current; self.budget.add(added)?; - if let Err(error) = self.budget.reconcile_admissions(self.budget.used()) { + if let Err(error) = self.budget.reconcile_admissions() { let _ = self.budget.remove(added); - let _ = self.budget.reconcile_admissions(self.budget.used()); + let _ = self.budget.reconcile_admissions(); return Err(error); } *held = bytes; @@ -235,10 +331,10 @@ impl DiskReservation { *held = current; return Err(error); } - if let Err(error) = self.budget.reconcile_admissions(self.budget.used()) { + if let Err(error) = self.budget.reconcile_admissions() { self.budget.add(released)?; *held = current; - let _ = self.budget.reconcile_admissions(self.budget.used()); + let _ = self.budget.reconcile_admissions(); return Err(error); } Ok(()) @@ -252,7 +348,7 @@ impl DiskReservation { let released = *held; *held = 0; let _ = self.budget.remove(released); - let _ = self.budget.reconcile_admissions(self.budget.used()); + let _ = self.budget.reconcile_admissions(); } /// Returns the bytes this reservation still holds. diff --git a/crates/cellule-ltx/src/environment/tests.rs b/crates/cellule-ltx/src/environment/tests.rs index 4809cb2a..d6d9a329 100644 --- a/crates/cellule-ltx/src/environment/tests.rs +++ b/crates/cellule-ltx/src/environment/tests.rs @@ -121,6 +121,231 @@ fn disk_budget_reservations_resize_and_release_exact_bytes() { assert_eq!(budget.available(), 10); } +#[test] +fn prepared_disk_budget_divides_parent_credit_without_double_charging() { + let parent = DiskBudget::new(10); + let child = parent.try_reserve(10).unwrap().into_budget(); + assert_eq!(parent.used(), 10); + assert_eq!(child.capacity(), 10); + assert_eq!(child.used(), 0); + let first = child.try_reserve(6).unwrap(); + let second = child.try_reserve(4).unwrap(); + assert_eq!(parent.used(), 10); + assert_eq!(child.used(), 10); + assert!(child.try_reserve(1).is_err()); + assert!(parent.try_reserve(1).is_err()); + first.resize(3).unwrap(); + assert_eq!(child.used(), 7); + assert_eq!(parent.used(), 10); + drop(second); + drop(child); + // The live child token retains the prepared envelope after its caller exits. + assert_eq!(parent.used(), 10); + drop(first); + assert_eq!(parent.used(), 0); +} + +#[test] +fn prepared_disk_budget_releases_unused_credit_and_accounts_later_growth() { + let parent = DiskBudget::new(10); + let child = parent.try_reserve(10).unwrap().into_budget(); + let file = child.try_reserve(6).unwrap(); + child.finish_preparation().unwrap(); + assert_eq!(parent.used(), 6); + child.finish_preparation().unwrap(); + let competing = parent.try_reserve(4).unwrap(); + assert!(file.try_grow(1).is_err()); + assert_eq!(file.bytes(), 6); + assert_eq!(child.used(), 6); + assert_eq!(parent.used(), 10); + drop(competing); + file.try_grow(4).unwrap(); + assert_eq!(parent.used(), 10); + assert_eq!(child.used(), 10); + assert!(file.try_grow(1).is_err()); + file.resize(2).unwrap(); + assert_eq!(parent.used(), 2); + assert_eq!(child.used(), 2); + drop(file); + assert_eq!(parent.used(), 0); + assert_eq!(child.used(), 0); + assert!(parent.finish_preparation().is_err()); +} + +#[test] +fn prepared_disk_budget_retains_parent_admission_and_refuses_rebinding() { + #[derive(Default)] + struct Recorded(std::sync::atomic::AtomicU64); + impl DiskBudgetAdmission for Recorded { + fn reconcile(&self, bytes: u64) -> crate::Result<()> { + self.0.store(bytes, Ordering::SeqCst); + Ok(()) + } + } + let parent = DiskBudget::new(10); + let admission = Arc::new(Recorded::default()); + parent.install_admission(admission.clone()).unwrap(); + let child = parent.try_reserve(10).unwrap().into_budget(); + let file = child.try_reserve(3).unwrap(); + assert_eq!(admission.0.load(Ordering::SeqCst), 10); + assert!(child.install_admission(admission.clone()).is_err()); + child.finish_preparation().unwrap(); + assert_eq!(admission.0.load(Ordering::SeqCst), 3); + file.resize(7).unwrap(); + assert_eq!(admission.0.load(Ordering::SeqCst), 7); + drop(file); + assert_eq!(admission.0.load(Ordering::SeqCst), 0); +} + +#[test] +fn prepared_disk_budget_nested_envelopes_keep_each_parent_bound() { + let parent = DiskBudget::new(12); + let child = parent.try_reserve(10).unwrap().into_budget(); + let grandchild = child.try_reserve(8).unwrap().into_budget(); + let file = grandchild.try_reserve(4).unwrap(); + grandchild.finish_preparation().unwrap(); + assert_eq!(grandchild.used(), 4); + assert_eq!(child.used(), 4); + assert_eq!(parent.used(), 10); + child.finish_preparation().unwrap(); + assert_eq!(parent.used(), 4); + file.resize(8).unwrap(); + assert_eq!(parent.used(), 8); + assert!(file.resize(9).is_err()); + assert_eq!(parent.used(), 8); + drop(file); + assert_eq!(parent.used(), 0); +} + +#[test] +fn prepared_disk_budget_concurrent_children_keep_exact_parent_accounting() { + let parent = DiskBudget::new(16); + let child = parent.try_reserve(16).unwrap().into_budget(); + let admitted = Arc::new(std::sync::Barrier::new(9)); + let release = Arc::new(std::sync::Barrier::new(9)); + std::thread::scope(|threads| { + for _ in 0..8 { + let child = child.clone(); + let admitted = admitted.clone(); + let release = release.clone(); + threads.spawn(move || { + let file = child.try_reserve(1).unwrap(); + admitted.wait(); + release.wait(); + file.resize(2).unwrap(); + }); + } + admitted.wait(); + assert_eq!(child.used(), 8); + assert_eq!(parent.used(), 16); + child.finish_preparation().unwrap(); + assert_eq!(parent.used(), 8); + release.wait(); + }); + assert_eq!(child.used(), 0); + assert_eq!(parent.used(), 0); +} + +#[test] +fn prepared_disk_budget_zero_credit_and_overflow_fail_without_parent_leaks() { + let parent = DiskBudget::new(u64::MAX); + let zero = parent.try_reserve(0).unwrap().into_budget(); + assert!(zero.try_reserve(1).is_err()); + zero.finish_preparation().unwrap(); + assert_eq!(parent.used(), 0); + let child = parent.try_reserve(u64::MAX).unwrap().into_budget(); + let file = child.try_reserve(u64::MAX).unwrap(); + assert!(file.try_grow(1).is_err()); + assert_eq!(parent.used(), u64::MAX); + child.finish_preparation().unwrap(); + drop(file); + assert_eq!(parent.used(), 0); +} + +#[test] +fn prepared_disk_budget_failed_parent_hook_preserves_credit_and_source_error() { + struct Refusing(AtomicBool); + impl DiskBudgetAdmission for Refusing { + fn reconcile(&self, _bytes: u64) -> crate::Result<()> { + if self.0.load(Ordering::SeqCst) { + Err(crate::LtxError::Io(io::Error::new( + io::ErrorKind::StorageFull, + "injected parent admission failure", + ))) + } else { + Ok(()) + } + } + } + let parent = DiskBudget::new(10); + let admission = Arc::new(Refusing(AtomicBool::new(false))); + parent.install_admission(admission.clone()).unwrap(); + let child = parent.try_reserve(10).unwrap().into_budget(); + let file = child.try_reserve(3).unwrap(); + admission.0.store(true, Ordering::SeqCst); + assert!(matches!( + child.finish_preparation(), + Err(crate::LtxError::Io(error)) if error.kind() == io::ErrorKind::StorageFull + )); + assert_eq!(parent.used(), 10); + assert_eq!(child.used(), 3); + assert_eq!(file.bytes(), 3); + admission.0.store(false, Ordering::SeqCst); + child.finish_preparation().unwrap(); + assert_eq!(parent.used(), 3); + admission.0.store(true, Ordering::SeqCst); + assert!(matches!( + file.resize(4), + Err(crate::LtxError::Io(error)) if error.kind() == io::ErrorKind::StorageFull + )); + assert_eq!(parent.used(), 3); + assert_eq!(child.used(), 3); + assert_eq!(file.bytes(), 3); + admission.0.store(false, Ordering::SeqCst); + drop(file); + assert_eq!(parent.used(), 0); +} + +#[test] +fn prepared_disk_budget_competing_scopes_reconcile_the_latest_parent_usage() { + #[derive(Default)] + struct Recorded(std::sync::atomic::AtomicU64); + impl DiskBudgetAdmission for Recorded { + fn reconcile(&self, bytes: u64) -> crate::Result<()> { + self.0.store(bytes, Ordering::SeqCst); + Ok(()) + } + } + let parent = DiskBudget::new(16); + let admission = Arc::new(Recorded::default()); + parent.install_admission(admission.clone()).unwrap(); + for _ in 0..32 { + let admitted = Arc::new(std::sync::Barrier::new(9)); + let release = Arc::new(std::sync::Barrier::new(9)); + std::thread::scope(|threads| { + for _ in 0..8 { + let parent = parent.clone(); + let admitted = admitted.clone(); + let release = release.clone(); + threads.spawn(move || { + let child = parent.try_reserve(2).unwrap().into_budget(); + let file = child.try_reserve(1).unwrap(); + child.finish_preparation().unwrap(); + admitted.wait(); + release.wait(); + drop(file); + }); + } + admitted.wait(); + assert_eq!(parent.used(), 8); + assert_eq!(admission.0.load(Ordering::SeqCst), 8); + release.wait(); + }); + assert_eq!(parent.used(), 0); + assert_eq!(admission.0.load(Ordering::SeqCst), 0); + } +} + #[cfg(feature = "replica")] #[test] fn directory_cache_survives_restart_and_evicts_by_bytes() { diff --git a/crates/cellule-ltx/src/error.rs b/crates/cellule-ltx/src/error.rs index f24ef5f1..91e0d6e7 100644 --- a/crates/cellule-ltx/src/error.rs +++ b/crates/cellule-ltx/src/error.rs @@ -29,6 +29,8 @@ pub enum LimitKind { CellBundleBytes, /// The Cell root exceeded its byte budget. CellRootBytes, + /// The exact-root dependency inventory exceeded its object-count budget. + RootInventoryObjects, /// The Cell root exceeded its segment-count budget. CellRootSegments, /// One Cell scale-load batch exceeded its byte budget. @@ -96,6 +98,7 @@ impl LimitKind { Self::CapturedLtxBytes => "captured LTX bytes", Self::CellBundleBytes => "Cell bundle bytes", Self::CellRootBytes => "Cell root bytes", + Self::RootInventoryObjects => "root inventory objects", Self::CellRootSegments => "Cell root segments", Self::CellScaleBytes => "Cell scale bytes", Self::CellScaleChecksumLength => "Cell scale checksum length", @@ -176,6 +179,14 @@ pub enum LtxError { #[cfg(feature = "replica")] #[error("object-store replication failure: {0}")] Storage(#[from] cellule_store::StorageError), + /// Caller-owned preparation metadata failed; immutable data may be uploaded. + #[cfg(feature = "replica")] + #[error("root preparation metadata failed: {source}")] + RootPreparation { + /// Original caller/provider error; no preparation completion is granted. + #[source] + source: Box, + }, /// Replica metadata was not valid JSON. #[cfg(feature = "replica")] #[error("invalid replica metadata: {0}")] @@ -243,7 +254,7 @@ impl LtxError { #[cfg(feature = "replica")] Self::Json(_) => FailureClass::Permanent, #[cfg(feature = "replica")] - Self::Task(_) => FailureClass::Ambiguous, + Self::Task(_) | Self::RootPreparation { .. } => FailureClass::Ambiguous, Self::ChecksumMismatch | Self::LTXCorrupted | Self::LTXMissing diff --git a/crates/cellule-ltx/src/lib.rs b/crates/cellule-ltx/src/lib.rs index f402756f..b562be61 100644 --- a/crates/cellule-ltx/src/lib.rs +++ b/crates/cellule-ltx/src/lib.rs @@ -78,7 +78,8 @@ pub use node_frame::{NodeFrameScope, VerifiedNodeFrame, encode_node_frame, inspe #[cfg(feature = "replica")] pub use replica::{ CellPagedDatabase, CellReplica, CellWritableDatabase, PreparedRoot, PublicationCost, - ReadOnlyRoot, RecoveryOverlay, RootObjectRef, RootRef, VerifiedRoot, + ReadOnlyRoot, RecoveryOverlay, RootObjectRef, RootPreparation, RootPreparationFuture, + RootPreparationMetadata, RootRef, VerifiedRoot, }; #[cfg(feature = "replica")] pub use writable_vfs::Hydration; diff --git a/crates/cellule-ltx/src/replica/directory/mod.rs b/crates/cellule-ltx/src/replica/directory/mod.rs index 7e236131..4f72248e 100644 --- a/crates/cellule-ltx/src/replica/directory/mod.rs +++ b/crates/cellule-ltx/src/replica/directory/mod.rs @@ -300,6 +300,7 @@ pub(super) async fn reachable_digests( root: [u8; 32], height: u32, expected_root: Aggregate, + max_objects: Option, ) -> Result> { if height > 3 || verification.database_pages == 0 { return Err(LtxError::LTXCorrupted); @@ -317,6 +318,9 @@ pub(super) async fn reachable_digests( if (remaining == 0) != (header.kind == 0) { return Err(LtxError::LTXCorrupted); } + if max_objects.is_some_and(|limit| digests.len() == limit) { + return Err(LtxError::Limit(crate::LimitKind::RootInventoryObjects)); + } digests.push(digest); if header.kind == 0 { let (aggregate, entries) = verify_leaf( diff --git a/crates/cellule-ltx/src/replica/mod.rs b/crates/cellule-ltx/src/replica/mod.rs index a93f43bd..28caceae 100644 --- a/crates/cellule-ltx/src/replica/mod.rs +++ b/crates/cellule-ltx/src/replica/mod.rs @@ -17,7 +17,9 @@ mod cache; mod compaction; pub(crate) mod directory; mod merge; +mod preparation; mod prepare; +pub use preparation::{RootPreparation, RootPreparationFuture, RootPreparationMetadata}; mod read_only; mod restore; pub(crate) mod root; @@ -211,6 +213,15 @@ impl RecoveryOverlay { } impl PreparedRoot { + /// Returns the exact native derivation after all immutable uploads completed. + #[must_use] + pub fn preparation(&self) -> RootPreparation { + RootPreparation { + root: self.root(), + predecessor: self.predecessor, + } + } + /// Returns the exact root that was prepared. #[must_use] pub fn root(&self) -> RootRef { @@ -695,6 +706,10 @@ pub struct CellReplica { limits: Limits, host: Host, cost: Arc, + root_metadata: Option>, + // Set only by verified representation-only compaction composition. It + // follows this one preparation and never survives in a prepared read view. + preparation_predecessor: Option, } impl CellReplica { @@ -715,6 +730,8 @@ impl CellReplica { limits: limits.validate()?, host: Host::default(), cost: Arc::new(PublicationLedger::default()), + root_metadata: None, + preparation_predecessor: None, }) } @@ -724,6 +741,13 @@ impl CellReplica { self.limits } + /// Returns the fixed Cell and incarnation binding of every immutable operation. + /// This scope provides no authority or selected root. + #[must_use] + pub const fn scope(&self) -> ([u8; 32], [u8; 16]) { + (self.cell, self.incarnation) + } + /// Returns the cumulative immutable publication cost this replica paid. /// /// The ledger covers every object the replica uploaded: segment bodies and @@ -745,6 +769,14 @@ impl CellReplica { self.cost.take() } + /// Joins caller-owned verified-derivation metadata with root uploads. + /// This replaces the one metadata facility; it supplies no authority policy. + #[must_use] + pub fn with_root_metadata(mut self, metadata: Arc) -> Self { + self.root_metadata = Some(metadata); + self + } + /// Selects the caller's bounded I/O and blocking execution facilities. #[must_use] pub fn with_host(mut self, host: Host) -> Self { diff --git a/crates/cellule-ltx/src/replica/preparation.rs b/crates/cellule-ltx/src/replica/preparation.rs new file mode 100644 index 00000000..ef2366c8 --- /dev/null +++ b/crates/cellule-ltx/src/replica/preparation.rs @@ -0,0 +1,47 @@ +//! Native verified derivation observed before immutable upload completion. +use super::*; +use std::{future::Future, pin::Pin}; + +/// Verified native root derivation whose immutable uploads may still be running. +/// +/// Construction is private. This value allows retention of preparation metadata; +/// it grants no uploaded-root, authority, restore, serving or acknowledgement +/// rights. Only the later `PreparedRoot` proves all immutable uploads finished. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct RootPreparation { + pub(super) root: RootRef, + pub(super) predecessor: Option, +} +impl RootPreparation { + /// Exact native root identity, without upload-completion rights. + #[must_use] + pub const fn root(&self) -> RootRef { + self.root + } + /// Exact verified input, if any. + #[must_use] + pub const fn predecessor(&self) -> Option { + self.predecessor + } +} + +/// Caller-owned metadata work joined with the same immutable preparation. +pub type RootPreparationFuture<'a> = Pin< + Box< + dyn Future>> + + Send + + 'a, + >, +>; + +/// Caller-supplied storage of verified native preparation metadata. +/// +/// This future may overlap immutable root-object uploads. Both must finish +/// successfully before a `PreparedRoot` escapes. The caller owns admission and +/// its finite work lifetime; the future shares the replica's origin I/O permit. +/// This starts no task, scheduler or authority effect. +/// Errors preserve their source and require caller reconciliation/classification. +pub trait RootPreparationMetadata: Send + Sync { + /// Retains one exact verified derivation, without selecting Cell authority. + fn retain(&self, preparation: RootPreparation) -> RootPreparationFuture<'_>; +} diff --git a/crates/cellule-ltx/src/replica/prepare.rs b/crates/cellule-ltx/src/replica/prepare.rs index 02288307..4a44d8d9 100644 --- a/crates/cellule-ltx/src/replica/prepare.rs +++ b/crates/cellule-ltx/src/replica/prepare.rs @@ -34,14 +34,14 @@ impl CellReplica { "append requires a representation-only compaction", )); } - let mut successor = self - .prepare(Some(&root), cuts, commit_sequence, schema) - .await?; // The private compaction authenticates identical logical state. The - // final root replaces that original state directly, after every new - // dependency has uploaded through the normal preparation path. - successor.predecessor = Some(predecessor); - Ok(successor) + // native factory must name the final authority predecessor before + // exposing derivation metadata, not rebase it after retention. + let mut replica = self.clone(); + replica.preparation_predecessor = Some(predecessor); + replica + .prepare(Some(&root), cuts, commit_sequence, schema) + .await } /// Verifies and uploads a new immutable root without changing authority. @@ -630,9 +630,6 @@ impl CellReplica { let bytes = encode_root(&document)?; let digest = *blake3::hash(&bytes).as_bytes(); root_objects.push((digest, bytes)); - // The document and its immutable segment pages can be uploaded in - // parallel. The root digest remains private until all uploads finish. - self.put_objects(CellObjectKind::Root, root_objects).await?; let root = RootRef { cell: self.cell, incarnation: self.incarnation, @@ -646,14 +643,36 @@ impl CellReplica { .without_recovery() .without_dirty() .without_scratch(); + // Check the full native graph before exposing even its derivation. The + // verified view carries no preparation callback beyond this owned call. + let mut view_replica = self.clone().with_host(host); + view_replica.root_metadata = None; + view_replica.preparation_predecessor = None; + let verified = VerifiedRoot::from_graph(view_replica, root, &document, descriptors)?; + let predecessor = self.preparation_predecessor.or_else(|| base.copied()); + let preparation = RootPreparation { root, predecessor }; + let metadata = async { + if let Some(metadata) = &self.root_metadata { + // Metadata shares native origin admission. It owns no second + // queue and releases this permit on completion or cancellation. + let _permit = self.host.io_permit().await?; + metadata + .retain(preparation) + .await + .map_err(|source| LtxError::RootPreparation { source })?; + } + Ok::<_, LtxError>(()) + }; + // Independent immutable objects and derivation metadata can overlap. + // Neither the complete proposal nor authority rights escape on failure. + futures_util::future::try_join( + self.put_objects(CellObjectKind::Root, root_objects), + metadata, + ) + .await?; Ok(PreparedRoot { - predecessor: base.copied(), - verified: VerifiedRoot::from_graph( - self.clone().with_host(host), - root, - &document, - descriptors, - )?, + predecessor, + verified, }) } diff --git a/crates/cellule-ltx/src/replica/verify.rs b/crates/cellule-ltx/src/replica/verify.rs index 2da73532..6fb16087 100644 --- a/crates/cellule-ltx/src/replica/verify.rs +++ b/crates/cellule-ltx/src/replica/verify.rs @@ -23,6 +23,31 @@ impl CellReplica { /// Callers may use this bounded inventory for backup pinning and reachability /// collection. A missing or corrupt dependency fails the traversal closed. pub async fn reachable_objects(&self, root: &RootRef) -> Result> { + self.reachable_objects_inner(root, None).await + } + + /// Verifies the same complete origin graph with an explicit inventory bound. + /// + /// Zero or excess objects refuse with `RootInventoryObjects`; no partial + /// inventory is returned. The bound covers retained directory digests and + /// the final distinct inventory. Descriptor work retains the existing fixed + /// root/segment limits. The caller owns memory admission and the deadline. + pub async fn reachable_objects_bounded( + &self, + root: &RootRef, + max_objects: usize, + ) -> Result> { + if max_objects == 0 { + return Err(LtxError::Limit(crate::LimitKind::RootInventoryObjects)); + } + self.reachable_objects_inner(root, Some(max_objects)).await + } + + async fn reachable_objects_inner( + &self, + root: &RootRef, + max_objects: Option, + ) -> Result> { // Inventory must prove origin presence even for metadata uploaded here. let graph = self.load_graph_with_cache(root, false).await?; let extents = object_extents(&graph.descriptors)?; @@ -41,6 +66,7 @@ impl CellReplica { graph.document.directory_digest, graph.document.directory_height, graph.aggregate, + max_objects, ) .await?; @@ -100,6 +126,9 @@ impl CellReplica { digest, kind: CellObjectKind::Directory, })); + if max_objects.is_some_and(|limit| objects.len() > limit) { + return Err(LtxError::Limit(crate::LimitKind::RootInventoryObjects)); + } Ok(objects.into_iter().collect()) } diff --git a/crates/cellule-ltx/tests/cell/roots.rs b/crates/cellule-ltx/tests/cell/roots.rs index d1b986dc..2ffe77da 100644 --- a/crates/cellule-ltx/tests/cell/roots.rs +++ b/crates/cellule-ltx/tests/cell/roots.rs @@ -34,5 +34,6 @@ fn replica(store: Store, cell: [u8; 32], incarnation: [u8; 16]) -> CellReplica { mod compaction; mod directory; mod lifecycle; +mod preparation; mod prepare_cost; mod sparse; diff --git a/crates/cellule-ltx/tests/cell/roots/lifecycle.rs b/crates/cellule-ltx/tests/cell/roots/lifecycle.rs index 4bf50d73..ef52631a 100644 --- a/crates/cellule-ltx/tests/cell/roots/lifecycle.rs +++ b/crates/cellule-ltx/tests/cell/roots/lifecycle.rs @@ -379,6 +379,21 @@ async fn exact_root_inventory_verifies_every_remote_dependency() { writer.close().unwrap(); let objects = replica.reachable_objects(&root).await.unwrap(); + assert_eq!( + replica + .reachable_objects_bounded(&root, objects.len()) + .await + .unwrap(), + objects + ); + for limit in [0, 1, objects.len() - 1] { + assert!(matches!( + replica.reachable_objects_bounded(&root, limit).await, + Err(cellule_ltx::LtxError::Limit( + cellule_ltx::LimitKind::RootInventoryObjects + )) + )); + } assert!(objects.windows(2).all(|pair| pair[0] < pair[1])); for kind in [ CellObjectKind::Ltx, diff --git a/crates/cellule-ltx/tests/cell/roots/preparation.rs b/crates/cellule-ltx/tests/cell/roots/preparation.rs new file mode 100644 index 00000000..0897bba5 --- /dev/null +++ b/crates/cellule-ltx/tests/cell/roots/preparation.rs @@ -0,0 +1,160 @@ +use super::*; +use cellule_ltx::{ + FailureClass, LtxError, RootPreparation, RootPreparationFuture, RootPreparationMetadata, +}; +use std::sync::{ + Mutex, + atomic::{AtomicBool, AtomicUsize}, +}; +use tokio::sync::Notify; + +#[derive(Default)] +struct Metadata { + observed: Mutex>, + pause: AtomicBool, + active: AtomicUsize, + entered: Notify, + resume: Notify, +} +struct Active<'a>(&'a AtomicUsize); +impl Drop for Active<'_> { + fn drop(&mut self) { + self.0.fetch_sub(1, Ordering::SeqCst); + } +} +impl RootPreparationMetadata for Metadata { + fn retain(&self, preparation: RootPreparation) -> RootPreparationFuture<'_> { + Box::pin(async move { + self.active.fetch_add(1, Ordering::SeqCst); + let _active = Active(&self.active); + self.observed.lock().unwrap().push(preparation); + if self.pause.load(Ordering::SeqCst) { + self.entered.notify_one(); + self.resume.notified().await; + } + Ok(()) + }) + } +} + +#[tokio::test] +async fn preparation_metadata_has_exact_native_inputs_and_no_detached_lifetime() { + let directory = tempfile::tempdir().unwrap(); + let mut database = + Db::open(&directory.path().join("native.sqlite"), Limits::default()).unwrap(); + database + .transaction(|transaction| { + transaction + .execute_batch("CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES(7)") + }) + .unwrap(); + let metadata = Arc::new(Metadata::default()); + let slots = Arc::new(tokio::sync::Semaphore::new(1)); + let replica = replica(Store::new(Arc::new(InMemory::new())), [1; 32], [2; 16]) + .with_host(Host::default().with_io_slots(slots.clone())) + .with_root_metadata(metadata.clone()); + let cuts = database.capture().unwrap(); + metadata.pause.store(true, Ordering::SeqCst); + { + let preparation = replica.prepare(None, &cuts, 1, 1); + tokio::pin!(preparation); + tokio::time::timeout(std::time::Duration::from_secs(5), async { + tokio::select! { + result = &mut preparation => panic!("paused preparation escaped: {}", result.is_ok()), + _ = metadata.entered.notified() => {} + } + }).await.unwrap(); + assert_eq!(metadata.active.load(Ordering::SeqCst), 1); + assert_eq!(slots.available_permits(), 0); + } + assert_eq!( + metadata.active.load(Ordering::SeqCst), + 0, + "dropping caller future must drop inline metadata work" + ); + assert_eq!(slots.available_permits(), 1); + metadata.pause.store(false, Ordering::SeqCst); + let first = replica.prepare(None, &cuts, 1, 1).await.unwrap(); + assert_eq!(metadata.observed.lock().unwrap()[0], first.preparation()); + assert_eq!(first.preparation().root(), first.root()); + assert_eq!(first.preparation().predecessor(), None); + database + .transaction(|transaction| { + transaction.execute("UPDATE counter SET value = 8", [])?; + Ok(()) + }) + .unwrap(); + let second = replica + .prepare(Some(&first.root()), &database.capture().unwrap(), 2, 1) + .await + .unwrap(); + assert_eq!( + metadata.observed.lock().unwrap().last().copied(), + Some(second.preparation()) + ); + assert_eq!(second.preparation().predecessor(), Some(first.root())); + let compacted = replica + .prepare_compaction(&second.root(), 0..2, 1, directory.path()) + .await + .unwrap(); + assert_ne!(compacted.root(), second.root()); + database + .transaction(|transaction| { + transaction.execute("UPDATE counter SET value = 9", [])?; + Ok(()) + }) + .unwrap(); + let third = replica + .prepare_after_compaction(&compacted, &database.capture().unwrap(), 3, 1) + .await + .unwrap(); + assert_eq!(third.predecessor(), Some(second.root())); + assert_eq!( + metadata.observed.lock().unwrap().last().copied(), + Some(third.preparation()), + "native metadata must observe the final rebased predecessor" + ); + assert_eq!(metadata.active.load(Ordering::SeqCst), 0); + assert_eq!(slots.available_permits(), 1); + database.close().unwrap(); +} + +struct FailedMetadata; +impl RootPreparationMetadata for FailedMetadata { + fn retain(&self, _preparation: RootPreparation) -> RootPreparationFuture<'_> { + Box::pin(async { + Err(Box::new(std::io::Error::new( + std::io::ErrorKind::PermissionDenied, + "original metadata error", + )) as Box) + }) + } +} + +#[tokio::test] +async fn metadata_error_preserves_source_and_refuses_ready_root() { + let directory = tempfile::tempdir().unwrap(); + let mut database = + Db::open(&directory.path().join("native.sqlite"), Limits::default()).unwrap(); + database + .transaction(|transaction| transaction.execute_batch("CREATE TABLE events(value INTEGER)")) + .unwrap(); + let replica = replica(Store::new(Arc::new(InMemory::new())), [1; 32], [2; 16]) + .with_root_metadata(Arc::new(FailedMetadata)); + let error = match replica + .prepare(None, &database.capture().unwrap(), 1, 1) + .await + { + Ok(_) => panic!("failed metadata returned a Ready proposal"), + Err(error) => error, + }; + assert_eq!(error.classify(), FailureClass::Ambiguous); + let LtxError::RootPreparation { source } = error else { + panic!("lost original metadata source") + }; + assert_eq!( + source.downcast::().unwrap().kind(), + std::io::ErrorKind::PermissionDenied + ); + database.close().unwrap(); +} diff --git a/crates/cellule-ltx/tests/host/hooks/activation.rs b/crates/cellule-ltx/tests/host/hooks/activation.rs index 5f6d8273..c2d47da5 100644 --- a/crates/cellule-ltx/tests/host/hooks/activation.rs +++ b/crates/cellule-ltx/tests/host/hooks/activation.rs @@ -222,6 +222,92 @@ async fn writable_activation_dispatches_filesystem_work_with_one_job_slot() { writer.close().unwrap(); } +#[tokio::test] +async fn prepared_disk_credit_reaches_real_sparse_activation_and_sqlite() { + let (directory, _, host, mut writer) = fixture(); + let expected: Vec = writer + .query_with(|connection| connection.query_row("SELECT v FROM t", [], |row| row.get(0))) + .unwrap(); + let parent = cellule_ltx::DiskBudget::new(3 << 30); + let prepared = parent.try_reserve(parent.capacity()).unwrap().into_budget(); + let host = host.with_local_disk_budget(prepared.clone()); + let paged = prepared_root(host, &mut writer, Store::new(Arc::new(InMemory::new()))).await; + let destination = directory.path().join("prepared-credit.sqlite"); + let writable = paged.prepare_writable(&destination).await.unwrap(); + assert_eq!(parent.available(), 0); + let task_parent = parent.clone(); + let task_prepared = prepared.clone(); + tokio::task::spawn_blocking(move || { + let mut restored = writable.open_writable(&destination).unwrap(); + let count: i64 = restored + .query_with(|connection| { + connection.query_row("SELECT count(*) FROM t", [], |row| row.get(0)) + }) + .unwrap(); + assert_eq!(count, 1); + let actual: Vec = restored + .query_with(|connection| connection.query_row("SELECT v FROM t", [], |row| row.get(0))) + .unwrap(); + assert_eq!(actual, expected); + assert_eq!(task_parent.used(), task_parent.capacity()); + // Sparse pages acquire lasting disk tokens when SQLite hydrates them. + assert!(task_prepared.used() > 0); + task_prepared.finish_preparation().unwrap(); + assert_eq!(task_parent.used(), task_prepared.used()); + assert!(task_parent.available() > 0); + restored + .transaction(|transaction| transaction.execute("INSERT INTO t VALUES(2)", [])) + .unwrap(); + let captured = restored.capture().unwrap(); + assert!(!captured.segments.is_empty()); + assert_eq!(task_parent.used(), task_prepared.used()); + restored.close().unwrap(); + }) + .await + .unwrap(); + writer.close().unwrap(); + assert_eq!(prepared.used(), 0); + assert_eq!(parent.used(), 0); +} + +#[tokio::test] +async fn canceled_sparse_activation_retains_prepared_disk_credit_until_job_closes() { + let (directory, faults, host, mut writer) = fixture(); + let parent = cellule_ltx::DiskBudget::new(3 << 30); + let prepared = parent.try_reserve(parent.capacity()).unwrap().into_budget(); + let host = host.with_local_disk_budget(prepared.clone()); + let paged = prepared_root(host, &mut writer, Store::new(Arc::new(InMemory::new()))).await; + let destination = directory.path().join("cancel-prepared-credit.sqlite"); + let pause = Arc::new(Pause { + operation: "create", + entered: tokio::sync::Notify::new(), + released: Mutex::new(false), + wake: std::sync::Condvar::new(), + }); + let release = Release(pause.clone()); + *faults.pause.lock().unwrap() = Some(pause.clone()); + let task = tokio::spawn(async move { paged.prepare_writable(&destination).await }); + tokio::time::timeout(Duration::from_secs(5), pause.entered.notified()) + .await + .unwrap(); + task.abort(); + assert!(task.await.err().unwrap().is_cancelled()); + drop(prepared); + assert_eq!(parent.used(), parent.capacity()); + assert!(parent.try_reserve(1).is_err()); + *faults.pause.lock().unwrap() = None; + drop(release); + tokio::time::timeout(Duration::from_secs(5), async { + while parent.used() != 0 { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + assert_eq!(parent.available(), parent.capacity()); + writer.close().unwrap(); +} + #[tokio::test] async fn canceled_activation_retains_admission_until_file_cleanup_finishes() { for operation in ["create", "write_all", "file_len"] { diff --git a/crates/cellule-runtime/api-prelude.txt b/crates/cellule-runtime/api-prelude.txt index 63d39d17..723e8410 100644 --- a/crates/cellule-runtime/api-prelude.txt +++ b/crates/cellule-runtime/api-prelude.txt @@ -43,6 +43,7 @@ Query QueueModule QueueNamespace Receipt +ReadReplicaSource Registry RegistryBuilder Resolution diff --git a/crates/cellule-runtime/docs/README.md b/crates/cellule-runtime/docs/README.md index 0624ca30..2f2b6cd6 100644 --- a/crates/cellule-runtime/docs/README.md +++ b/crates/cellule-runtime/docs/README.md @@ -61,6 +61,24 @@ flowchart LR Cell view. Its view and replacement refresh are charged to the node runtime's memory, descriptor, and disk ledgers. +`close()` fences new reader work. `close_and_join()` also detaches snapshots +from every peer clone and joins accepted query/refresh work, including native +jobs whose waiters were cancelled. Its receipt preserves the last installed +position; it does not establish current authority or fleet retirement. See the +[host reader lifecycle](../../cellule-host/docs/read-replicas.md). + +`lifecycle_observation()` reads the same irreversible admission word and shared +snapshot state. It exposes accepted native lifetimes after caller cancellation +and detachment. Local joining requires closed admission, detached state and zero +lifetimes; this supplies no remote authority, enrollment or replacement-policy +proof. The host's bounded reader pages include these original observations. + +`prepare_source()` observes an opaque exact root, owner/epoch, code/schema and +signed physical boot scope without reserving a view. `open_source()` opens that +pinned root through the same admitted native path, even if the owner publishes +a newer root meanwhile. Installation still checks current authority and the +original signed boot identity. Source metadata grants no admission or readiness. + Those charges are provisional: product routing and measured capacity qualification remain open under [Plan 036](https://github.com/crabbuild/crab/blob/beb439039cb37e750afe6625a2358101c70d1191/advisor-plans/036-cell-read-replicas-and-fenced-promotion.md). diff --git a/crates/cellule-runtime/docs/canonical-ltx-scaling.md b/crates/cellule-runtime/docs/canonical-ltx-scaling.md index 89a05b19..5df1a4e1 100644 --- a/crates/cellule-runtime/docs/canonical-ltx-scaling.md +++ b/crates/cellule-runtime/docs/canonical-ltx-scaling.md @@ -208,7 +208,7 @@ shared authenticated mechanics remain private to Cell roots. | Surface | Current owner and behavior | | --- | --- | | Request entry | [`RepositoryCellRouter::route_target`](https://github.com/crabbuild/crab/blob/beb439039cb37e750afe6625a2358101c70d1191/crates/crab-http-server/src/cells/router.rs) calls `route_existing` twice around an activation lock, then repeats catalog and control loads for activation. | -| Metadata lookup | [`CellCatalog::lookup`](../src/cell/catalog/mod.rs) loads the shard head and every referenced immutable catalog page; [`CellAuthority::load`](../src/control/authority.rs) separately reads exact control. | +| Metadata lookup | [`CellCatalog::lookup`](../src/cell/catalog/mod.rs) loads the shard head and every referenced immutable catalog page; [`CellAuthority::load`](../src/control/authority/mod.rs) separately reads exact control. | | Local residency | [`CellRuntime::resident_handle`](../src/cell/actor/mod.rs) asks the actor for a fully resident owner before remote metadata; [`local_handle`](../src/cell/actor/mod.rs) remains the verified slow-path lookup for sparse or activation callers. Fenced, draining, and non-resident actors miss safely. | | Sparse hydration | [`Db::prepare_hydration` and `install_hydration`](../../cellule-ltx/src/db/mod.rs) bracket asynchronous fetch; a separate hydration effect permits foreground work and retains drain obligations. Cancellation, overwrite and takeover tests cover the split; fleet latency qualification remains. | | Fleet observation | [`NodePublisher`](https://github.com/crabbuild/crab/blob/beb439039cb37e750afe6625a2358101c70d1191/crates/crab-http-server/src/peer.rs) signs short-lived measured capacity and backlog observations; `NodeAdvertisement` carries a versioned placement signature. [`RepositoryCellRouter`](https://github.com/crabbuild/crab/blob/beb439039cb37e750afe6625a2358101c70d1191/crates/crab-http-server/src/cells/router.rs) plans movement from live signed samples and actor-settled candidates, then records confirmed release and receiver activation separately. Advertised disk headroom is clamped by the runtime ledger, server memory resolves nested cgroup-v1/v2 membership, and cold activation sends a bounded direct-node hint before normal authority acquisition. The test-only process race covers one shared-control winner; unified process-wide probe parity and protected multi-process movement proof remain. | diff --git a/crates/cellule-runtime/docs/delivery.md b/crates/cellule-runtime/docs/delivery.md index af441078..49a797b1 100644 --- a/crates/cellule-runtime/docs/delivery.md +++ b/crates/cellule-runtime/docs/delivery.md @@ -87,7 +87,7 @@ Each layer has one owner and one primary evidence surface. | Boundary | Primary source | Evidence | | --- | --- | --- | | Identity and Cell derivation | `src/identity.rs` | `src/identity.rs` tests, `tests/contracts/application.rs`, `tests/runtime/catalog.rs` | -| Control CAS and transitions | `src/control.rs`, `src/control/authority.rs` | authority and actor tests | +| Control CAS and transitions | `src/control/mod.rs`, `src/control/authority/mod.rs` | authority and actor tests | | SQLite command ledger | `src/cell/executor.rs`, `src/cell/schema.rs` | `tests/runtime/lifecycle.rs`, `tests/runtime/migration.rs` | | Fixed SQL workers | `src/cell/worker.rs` | `tests/runtime/workers.rs` | | Publication and exact-root recovery | `src/publication.rs`, `cellule-ltx` | `tests/runtime/publication.rs`, `cellule-ltx/tests/cell/roots.rs` | diff --git a/crates/cellule-runtime/docs/deployment.md b/crates/cellule-runtime/docs/deployment.md index e5a1e046..1f75b787 100644 --- a/crates/cellule-runtime/docs/deployment.md +++ b/crates/cellule-runtime/docs/deployment.md @@ -240,6 +240,144 @@ The service chooses each signed advertisement lifetime, up to 30 seconds. An exp Local admission remains authoritative. An advertisement cannot force a node to accept work after its measured budget is exhausted. +Canonical advertisement decoding checks shape, the identity and any understood +placement signature, and exact JSON bytes. Canonical byte comparison serializes that same verified, +immutable value without repeating signature verification. Storage encoding +still verifies before emission, and directory reads independently enforce +scope and current lease validity. Legacy, schema-2 and schema-3 record bytes +remain unchanged. + +### Operational observations and the reader rollout + +`NodeAdvertisement::with_placement_capacity` continues to emit schema 2. +Upgraded readers also understand schema 3 through +`NodeAdvertisement::with_operational_placement`. Schema 3 signs the node's +`Active`, `Cordoned`, or `Draining` mode, stable pressure tier, and measurement +sequence/time. Pressure and lifecycle mode are separate; measured free capacity +is retained on cordoned nodes. Writer, reader, follower, and recovery-executor +selection reject a schema 3 node whose mode or pressure closes new admission. + +Use `CellRuntime::operational_sample` for the local classifier output. It is +unknown until the first successful observation. Reading it or republishing a +heartbeat does not refresh its time. A changed measurement needs a new +sequence; directory refresh rejects regressing samples and changed capacity +under an unchanged sequence. The transient `Recovering` marker cannot be signed. + +The runtime's shared `NodeAdmission` gate closes new writer and reader +acquisition during pressure and keeps a cordon sticky after pressure recovery. +The host installs it on its follower store before readiness. Direct follower +store integrations must install `runtime.node_admission()` using +`FollowerStore::with_node_admission` before sharing the store. Existing enrolled +tail appends and existing reader refreshes continue under their normal fences. + +Pressure transfers on an Active donor prefer recently used settled Cells using +the actor's `last_used_ms`. Local emergency eviction continues closing oldest +idle Cells under its existing movement budget. Drains and normal balancing +retain their Cell identity order. Recency conveys no reservation: a racing +source close can still invalidate a proposal, and exact action validation must +refuse it and join unused receiver credit. Demand timestamps must be nonnegative +and no later than the planner's observation time. + +Fleet-managed hosts additionally hold this same gate with +`NodeAdmission::hold_startup` before installing a lease. While held, an otherwise +Active node reports Cordoned. `confirm_startup(mode)` removes only that hold; +it preserves a racing cordon/drain and pressure, increments the measurement +sequence, and leaves the original sample time unchanged. The host calls it only +after atomic boot/intent confirmation and required startup checks. These local +methods supply no remote authorization or canonical enrollment proof. See +[fleet boot admission](../../cellule-host/docs/lifecycle.md#fleet-boot-admission). + +`CellRuntime::try_reserve_node_metadata_bytes` charges bounded lifecycle +metadata to the same retained-byte ledger before lease admission or after +fencing. This token authorizes no native work or role. Native byte admission +still checks the node lease; terminal drain closes both allocation paths. +The host uses metadata credit for its retained supervisor observation owner. + +Old strict JSON readers reject the added schema 3 field. Deploy upgraded +readers throughout the fleet while continuing schema 2 output, then enable +schema 3 production through application rollout policy. Complete the +[fleet operations plan](../../../docs/fleet-operations-plan.md) before enabling +automatic movement. The operation records and local gate are foundations; +the full executor, busy-Cell maintenance, and foreign follower evacuation still +require the remaining work packages and qualification. + +### Observe local ownership in bounded pages + +`CellRuntime::fleet_cells_page(cursor, limit)` observes all local owners, +including busy/draining actors and outstanding activation/close/release tasks. +Choose a limit from 1 through 128. Each returned page holds one MiB of native +byte admission until dropped; cancellation retains that admission until the +actor finishes or discards its response. + +The opaque cursor is bound to the runtime session and ownership topology. +Restart a scan when topology changes. SQL activity can change advisory row +details between pages; a complete topology scan never authorizes release. +Actions still need fresh Cell authority, session, and generation checks. + +| Observation | Interpretation | +| --- | --- | +| `Owned` | Verified target and generation, distinct residence/last-use times, available published position, and local blockers. | +| `Transitioning` | A local lifecycle task still owns this Cell's obligation; include it in drain diagnostics. | +| Missing cost | Receiver demand is unknown; block proactive movement until worker/resource accounting establishes a conservative bound. | + +Worker probes measure logical SQLite bytes, persisted-work classes, and commit +sequence through the serialized SQL path. `database_bytes` is the current +logical size; `cost` reserves conservative restore demand derived from the +validated LTX limits. Native-memory cost is admission accounting, not measured +process RSS. A mutation invalidates the sample. A late probe from an older +inventory revision cannot recreate demand or clear newer unknown work. + +`sampled_at_ms` belongs to the worker probe; reading a page does not refresh it. +`stable_observations` counts distinct unchanged worker samples and is capped at +two. Ordinary movement needs two fresh samples. Unknown schema, failed probes, +unpublished sequence mismatch, clock regression, and samples at least 30 +seconds old keep costs unknown. The existing background loop admits at most +32 inventory probes at once through the shared worker-job ledger. + +The counts include obligations outside the returned page. For a bounded status +sample, callers can read them without collecting every row: + +```rust +use cellule_runtime::cell::actor::CellRuntime; + +async fn remaining_local_obligations(runtime: &CellRuntime) -> cellule_runtime::Result { + let page = runtime.fleet_cells_page(None, 128).await?; + Ok(page.owned_cells().saturating_add(page.transitioning_cells())) +} +``` + +Zero local obligations does not prove relocation, reader replacement, foreign +follower safety, or successful host shutdown. The fleet completion contract +requires each of those separate proofs. + +### Observe foreign follower obligations + +`FollowerStore::fleet_lanes_page(cursor, limit, now_ms)` reports persisted +leader/epoch lanes, including cold lanes after store reopen. Pages contain at +most 128 entries and retain one MiB from the existing follower-index memory +budget until dropped. Cancelled calls keep that charge until the blocking scan +finishes. The scan serializes with existing lane creation, append, seal, +retirement, and collection. Commit cordon before using it as a local enrollment +barrier. Changed topology or a newly opened store invalidates its cursor. + +Open and sealed lanes remain obligations. Retired lanes retain their durable +append fence, even after canonical retirement removes their chunks. Quarantine +and malformed filesystem entries require diagnosis; they cannot establish an +empty safe-to-stop node. Observation never seals, retires, or deletes a tail. + +Pair local pages with +`NodeDirectory::follower_logs_page(member, cursor, limit, now_ms)`. That scan +includes current log references in inactive/expired advertisements and +fenced/recovering tombstones. It retains a bounded window and refuses more than +10,000 directory records. Heartbeat and coverage updates preserve topology; +epoch or ensemble changes require restarting pagination. Apply the caller's +deadline and obtain the fleet observer's membership/enrollment barrier: +object-store listing alone is not an atomic proof that no reference exists. + +Recheck each exact epoch through the existing directory authorization and +recovery path before settling it. Zero local writer count, a dead leader, or +zero unretired local lanes alone cannot authorize maintenance shutdown. + ## Size node profiles and admission @@ -575,6 +713,168 @@ Drain ordering against reader admission: - Every phase uses the same absolute shutdown deadline. - A failed drain must not authorize a durability-mode change. +### Observe an original session's terminal fence + +`NodeDirectory::closed_session` returns an opaque `NodeSessionClosure` only +from the exact physical node/session's permanent tombstone with no enrolled +leader log or an exact Retired log. A live claimant and fresh canonical read +are required. Missing/expired advertisements and Open/Recovering/Sealed logs +refuse, including inactive logs awaiting native retirement. The closure retains +original fencing/expiry times, epoch, ensemble, coverage and pinned manifest; +mutable recovery claim renewal does not change this identity. + +This read starts no retirement or process effect. It proves no process joining, +affected Cell relocation, foreign role settlement or planned withdrawal. The +[host failed-boot publication](../../cellule-host/docs/lifecycle.md#publish-an-original-failed-boots-closure) +combines it with complete original enrollment closure and application-authenticated +durable process/accepted external work evidence. Grace retention and the +existing single-writer/recovery protocols still apply. + +### Prepare capacity before releasing a Cell + +`CellRuntime::prepare_receiver` binds an exact movement attempt, catalog proof, +receiver session, destination, cost, and expiry. It reserves an actual Cell +slot, conservative native memory and descriptors, the Cell's affine SQL worker, +and disk credit through the ordinary node ledgers. A partial refusal returns +its tokens and leaves source authority untouched. Cost must cover the incoming +replica's validated LTX bounds. Catalog target and replica Cell/incarnation +must match the immutable attempt before any resource admission. + +```rust +use std::path::PathBuf; +use cellule_runtime::cell::actor::{CellRuntime, PreparedCellReceiver}; +use cellule_runtime::cell::catalog::CatalogProof; +use cellule_runtime::fleet::operations::MoveAttemptSpec; +use cellule_runtime::ltx::CellReplica; + +fn prepare_receive( + runtime: &CellRuntime, + attempt: MoveAttemptSpec, + catalog: CatalogProof, + replica: CellReplica, + destination: PathBuf, + now_ms: i64, +) -> cellule_runtime::Result { + let expires_at_ms = attempt.deadline_ms; + runtime.prepare_receiver(attempt, catalog, replica, destination, expires_at_ms, now_ms) +} +``` + +The runtime retains at most two local receipts. Dropping the opaque reference +does not cancel credit; `prepared_receiver(attempt_id)` recovers a lost reply. +An exact duplicate returns the existing lifecycle without another charge. +Expiry permits explicit cancellation; it does not free credit automatically. + +| API | Contract | +| --- | --- | +| `activate_prepared_receiver` | Checks the exact Idle incarnation and receiver session, and an authority epoch at least as new as the source, then transfers tokens through canonical takeover, root verification, SQLite open, and actor activation. Accepted work survives a dropped waiter. | +| `cancel_prepared_receiver` | Returns unused credit synchronously. Refuses after activation was accepted; inspect authority and actor readiness to reconcile that work. | +| `PreparedCellReceiver::state` | Local lifecycle hint. `Activated` requires fresh authority and actor checks; `Failed` does not prove ownership is absent. | +| `retire_prepared_receiver` | Removes a joined terminal receipt after the application durably records and reconciles its result. Never reuses an attempt identity or closes its serving actor. | +| `shutdown` | Cancels unused preparation and joins accepted acquisition before closing actors and workers, including canonical rollback of a failed claim. Retained opaque references do not keep resources alive. | + +These are trusted local mechanisms. The host fleet executor binds scope/session, +retains accepted work, and journals exact source release and receiver results. +It records the checked Idle acquisition input before its ownership CAS; an +ambiguous basis-write reply prevents takeover. After release, unused-credit +cleanup preserves relocation state and the fleet permit. Receiver readiness +and resource settlement are independent proofs, including when ordinary cold +acquisition wins on the preferred session. + +The application supplies authentication and durable adapter semantics. A cached +action result still needs fresh authority and actor checks before being counted +as currently serving. `CellNode::inspect_fleet_action` captures that check using +`FleetInspectionRequest` and `FleetInspectionObservation`. Bind the complete +request to a new nonce, exact head action, retained registry version, endpoint +and capture deadline. Validate the original interval when consuming the reply; +republication cannot refresh it. The adapter checks current journal authorization +in one read transaction; the host performs actor/authority checks without using +cached Inspect success or starting recovery. Recheck both journal versions when +committing a dependent transition. A failed inspection retains the permit. +After a proven clean release, a fresh inspection may target another established +node or a new boot of the preferred node that acquired the Cell normally. The +reconciler selects a possible serving endpoint from authenticated actor rows +under the current durable roster; native authority and actor checks then prove +the exact successor position. This read creates no effect acceptance. Cleanup +continues to require the original receiver's independent resource proof. +Old inspection readers refuse the additional endpoint shape; deploy and qualify +upgraded readers before enabling this path in a mixed-version fleet. +Recovery across receiver sessions, the recurring reconciler, +busy maintenance, role evacuation, and deployment qualification remain work in +the +[fleet operations plan](../../../docs/fleet-operations-plan.md). + +### Retain failed source recovery evidence + +`AcquisitionObserver` supplies two confirmed recording points on canonical +`acquire_idle_restored_observed` and `takeover_restored_observed`: + +| Recording point | Required behavior | +| --- | --- | +| `before_claim` | Retain the exact input before ownership CAS. A changed predecessor is another input that the recorder explicitly accepts or rejects. A lost write reply aborts before CAS. | +| `before_activation` | Retain that same input and the actual control after optional pinned-overlay publication, before actor admission. Failure uses canonical acquisition rollback and cannot admit a writer. | + +The ordinary methods delegate to these same paths without a recorder. These +trusted callbacks grant no authority and authenticate no caller. Their caller +must own acquisition independently of transport cancellation; the host fleet +executor provides that finite-task ownership. + +`RecoveryBasis` binds an accepted recovery action to its exact failed source, +canonical control, and original capture time. Construction requires existing +`NodeTakeoverProof`. `RecoveryEvidence` checks the exact takeover and optional +canonical recovery-publication transition. Its root is the required recovery +position, not a current-serving witness. Recovered completion also checks the +live successor actor and current authority. Receiver cleanup remains an +independent prerequisite for retiring the charged fleet permit. + +Local host tests exercise source loss, an unobserved release CAS, lost recording +replies, later root advancement, and dropped waiters. Public runtime tests +exercise the callbacks with a pinned follower tail. These tests do not establish +controller restart durability, cross-session receiver failover, or the fleet +plan's process/provider campaign. + +### Retain intent and enrollment before fleet orchestration + +The pure `fleet::operations` registry records define these adapter contracts: + +| Record or gate | Contract | +| --- | --- | +| `NodeIntent::advance_maintenance` | Conditionally advances the retained physical-node row with the head transaction. A different maintenance operation cannot overwrite a cordon. Reboot adoption advances the intent revision. | +| `RegistryVersion` | One revision shared by every intent and enrollment mutation. Record the controlled bootstrap barrier; recheck this same version after collecting observations. A marker alone does not prove coverage. | +| `RegistryVersion::set_scheduling` | Revision-checked stop/resume. A new registry starts stopped; enabling requires bootstrap. Stopping preserves accepted work, charged permits, and retained cordons. | +| `RegistryVersion::authorize_allocation` | Invoke inside the allocation transaction with exact current intent rows. Reject a cordoned receiver, changed boot, stopped policy, or a draining source outside its current evacuation operation. The head reducer still checks fencing and budgets. | +| `EnrollmentRecord` | First Pending acceptance checks exact source/target boot and intent revisions. Retain unknown work after lease expiry. Compare the full spec on duplicate request identities. | +| `IntentPage` and `EnrollmentPage` | At most 128 sorted rows and one MiB per page. Keep older cordons and failed-session obligations; carry the same registry version through every cursor request. | + +Reader records name the exact Cell and published position. Follower records name +the source boot and node-log epoch. Boot enrollment records carry the retained +mode: a reboot may enroll its maintenance lease in Draining mode while writer, +reader, and follower admission remains closed. Boot enrollment does not open +readiness or clear intent. + +Only checked ordinary enrollment, definite refusal, or canonical role retirement +advances an enrollment record. Evidence digests identify the application's +retained proofs; decoding them supplies no authority or authentication. Duplicates +return the original proof and transition time. A timeout is not refusal, and +expiry never retires an unknown obligation. + +The host exposes `FleetJournal` and `FleetEnrollmentJournal` alongside the action +journal. Implement all three against the same transaction domain. Conditional +head publication must retain intent/operation changes and progress atomically; +enrollment acceptance must linearize current intent checks with Pending. These +records and pure tests do not establish complete fleet observation. The host's +[fleet journal example](../../cellule-host/minion/README.md) +implements these contracts in one local SQLite transaction domain, with focused +lost-reply and reconstruction evidence. The host's caller-driven reconciler now +uses the existing planner and reducer for settled movement; its initial tests +use simulated effects against that journal. The `overload` executable additionally +moves two real Cells across three leased nodes, reconstructs its controller +client and checks original receipts, restored state and joined resource ledgers. +Its fixed collector reports incomplete role coverage. Enrollment producers, complete +observation collection, leased-node failure integration and maintenance execution +remain implementation work. Local SQLite evidence +does not qualify a distributed journal provider or process-crash behavior. + ## Back up immutable roots and release metadata @@ -688,3 +988,76 @@ Alert on stalled scheduler progress, repeated owner fencing, publication backlog | Peer mTLS transport and routing | [Peer security](../../cellule-peer-http/docs/security.md) | | Repository-wide doc index | [Technical reference](technical-reference.md) | | Embedding application contract | [Workspace framework integration guide](../../../docs/framework.md) | + + +## Foreground quiescence for planned maintenance + +After retaining and authorizing a maintenance intent, an embedding application +can call `CellRuntime::quiesce_cell_at` with the exact Cell ID, runtime boot, +local generation, incarnation and ownership epoch. A mismatch fails before +closing admission. The transition is sticky for that activation and returns +when the actor installs the gate; it does not wait for accepted work to finish. +An unavailable publisher conservatively refuses the request, requiring a retry. + +| Work | During Cell quiescence | +| --- | --- | +| Previously actor-admitted commands/queries | Finish through the existing serialized worker and publication gates. | +| New foreground commands, queries, migrations and claims | Refused with CellDraining. | +| Native Queue/Effect lease operations; Activity completion/extension | Continue with their existing exact-token, expiry and durability checks. | +| Native lease validation and original outcome resolution | Continue on the current owner. | +| New hydration and compaction | Suppressed; required inventory can refresh. | +| Fencing, final drain and transfer preflight | Close completion admission as well. | + +Only private native registry bindings carry completion admission. Raw handlers, +application bindings and peer requests cannot supply an exemption flag. These +calls use the ordinary request, byte and worker bounds. Duplicate binding errors +leave the original handler intact. Registry descriptors, release bytes, command +IDs and peer formats are unchanged. + +`OwnedCellObservation` reports `quiescing` and optional `maintenance_work` from +the serialized worker inventory. Absence means unknown. The latter reports all +live Effect, Queue and Activity lease classes together; malformed leases block. +Valid expired leases remain unchanged for canonical destination reclamation. +Pending durable messages, Effects, timers and Workflow waits can be carried in +an exact root after claims close. Blob stream/upload/pin coverage remains an +explicit blocker. + +Public behavior cases now cover live Queue/Effect/Activity completion and exact +receiver restoration. Effect expiry preserves the existing retry backoff and +inbox deduplication; Activity expiry preserves its pinned definition and rejects +stale completion tokens. Unclaimed Effects and Activities resume through the +registered native drivers. Restored Workflow waits accept deduplicated signals, +and preserved due timers advance through the normal Tick. These local fixtures +are recorded in the [execution evidence](../../../docs/fleet-operations-progress.md); +provider/process faults, Cron and Blob owners still require qualification. + +A transferable primitive snapshot grants no release authority and does not +establish actor settlement or complete role coverage. Ordinary idle transfer +keeps its conservative readiness checks. + +`CellRuntime::release_maintenance_cell_at` accepts the same exact source identity +and a preflight deadline. The actor owns the request after acceptance, including +when its caller drops the reply waiter. It closes foreground admission, gives a +serialized readiness read a bounded worker slot behind accepted SQL, then closes +native completion admission and joins accepted work/publication. A second fresh +read must prove readiness before the existing canonical release starts. Lease +renewal continues until that confirmation. A live lease found on the second read +restores native completion only and retries within the deadline. + +| Result | Meaning | +| --- | --- | +| `MaintenanceCellRelease::Released(position)` | Canonical release completed with the exact final authority root and epoch. Ordinary exact-root receiver activation can consume it. | +| `MaintenanceCellRelease::Refused { blocker, error }` | This request started no canonical release. Preserve any source error; foreground quiescence stays installed if already accepted, and native completion remains available. | +| Other error or lost reply | Release may be unresolved. Retain fleet permits and inspect retained evidence or use canonical failed-session recovery. | + +The deadline bounds preflight; it does not cancel a confirmed release. A Blob +Cell refuses before closing admission because external stream/upload/pin owners +are not covered. `OwnedCellObservation::maintenance_cost` is a separate peak +receiver envelope derived from validated per-Cell LTX limits. It stays available +while mutations invalidate measured worker samples; it establishes no readiness. +The driver can plan with it only for the exact boot/node of a retained Evacuating +maintenance operation and a fresh authenticated collection barrier. Receiver +preparation precedes source quiescence and rechecks the actual cost. Ordinary +pressure/count moves retain their settled-sample rules. Complete primitive +acceptance, reader/follower producer wiring and role finalization remain required by the +[fleet implementation plan](../../../docs/fleet-operations-plan.md). diff --git a/crates/cellule-runtime/docs/failover-and-followers.md b/crates/cellule-runtime/docs/failover-and-followers.md index 80615372..b175374e 100644 --- a/crates/cellule-runtime/docs/failover-and-followers.md +++ b/crates/cellule-runtime/docs/failover-and-followers.md @@ -387,13 +387,41 @@ race through stale local state. append-fence marker for ten minutes. The server then: - Scans at most 64 lanes per minute. -- Requires the exact node-session record to exist and no longer name that log - epoch. +- Requires the exact node-session record to exist and no longer require local + copies of that epoch, including a canonically recovered Retired tombstone. - Rechecks the unchanged marker and its filesystem timestamp under the lane lock, and only then deletes it and releases disk admission. Missing authority fails closed. +**Recovered failed-owner retirement.** After the canonical recovery coordinator +pins the affected overlays and seals the tombstone, use +`retire_recovered_members` with a `RecoveredNodeLogTransport`. This extends the +same ordinary transport with an explicit recovery request. The receiver +authenticates the live requester and calls +`NodeDirectory::authorize_recovered_log_retire` for its own physical member, +original leader/epoch and pinned manifest. Open/Recovering records, foreign +members, changed manifests and expired requesters are refused. A native seal +alone cannot authorize retirement. + +`FollowerStore::retire_recovered` uses the same lane lock, byte ledger, scanner +and durable retired marker as ordinary retirement. Active lanes require their +original seal watermark to exactly match verified local records. An inactive +enrollment can fence an empty lane; unexpected records still block it. The +caller supplies no truncation watermark. Every original member's retirement +response is joined and retained, including failures and contradictory receipts. +Only complete confirmation can authorize `NodeDirectory::retire_recovered_log`. + +That CAS preserves the original epoch, ensemble and manifest in the permanent +Retired tombstone. Takeover remains valid; original recovery completion can +adopt the terminal record after a lost reply. `retired_recovered_log` rechecks +exact canonical retirement before effects, including after local grace +collection; `None` keeps a matching Sealed epoch outstanding. `Retired` stops +referencing local copies for collection purposes. The existing grace boundary, +exact marker check and external authority check still apply. This supplies a +tail retirement boundary, not failed-process joining, fleet enrollment +publication, replacement policy or permission to stop a physical node. + Deterministic fault coverage includes the two ambiguous recovery boundaries: - A follower may fsync a frame and lose its ACK without authorizing a fleet @@ -913,6 +941,67 @@ publication: - A successful retirement response with any other watermark is a protocol error and blocks the CAS. +**Maintenance member confirmation.** `NodeDurability::shutdown_for_maintenance` +uses the same shipper drain and contiguous object-coverage barrier, then requires +every original member to confirm its exact persisted append fence before calling +the canonical authority close. A lost response leaves the epoch retryable; +healthy siblings are joined before a member failure is returned. + +| API | Evidence and limit | +| --- | --- | +| `shutdown()` | Ordinary authorities use best-effort closure. Managed fleet authorities require all member confirmations and retain retry ownership. | +| `retirement_observation()` | Latest joined responses for the original leader, epoch, complete member set and watermark; each original transport error is retained. Contradictory receipts return a protocol error. | +| `NodeLogRetirementObservation::confirmed()` | Opaque confirmation of every member's append fence. This alone does not establish authority closure or lane deletion. | +| `shutdown_for_maintenance()` | Returns the member proof after canonical authority closure succeeds, and retains it for idempotent calls. Earlier best-effort closure with missing responses cannot be upgraded into proof. | + +Before every member confirms, cancellation creates no complete proof; retry +addresses the same epoch and complete member set. Once every checked response +is joined, the runtime retains that exact observation before awaiting authority +closure. A lost closure reply or cancelled closure waiter reuses those fences +and retries only the original authority callback. It sends no new retirement +RPCs against an epoch whose directory authorization may already have closed. +The shutdown proof is returned only after that callback succeeds; applications +must reconcile their exact original CAS and preserve ownership of accepted work. +Native follower retirement persists its fence before responding, so a lost +member reply still requires reconciliation even when the lane is already retired. +Complete fleet-role observation, journal settlement, +replacement-policy evidence and failed-process closure still belong to the +embedding application's maintenance controller. This API alone does not certify +that a physical node is safe to stop. Unmanaged host rotation retains ordinary +best-effort closure; the managed fleet binding requires complete member fences. + +**Prepared follower enrollment.** A durable fleet producer can separate the +existing directory selection from its conditional authority write: + +| API | Ordering and evidence | +| --- | --- | +| `prepare_log_enrollment` | Read-only selection of the complete ensemble through the canonical selector. Retains original signed physical boots and a provider-assigned epoch. | +| `prepare_log_enrollment_attempt` | Revalidates every original boot, live receiver admission/capacity and the source CAS version. Allows heartbeat updates; never substitutes a new member. | +| `commit_log_enrollment` | Writes the fixed ensemble using only the retained source version. Fleet producers must accept Pending for every original member before dispatch. | +| `inspect_log_enrollment` | Reconciles the original leader boot, epoch and complete physical member set across activation, coverage and heartbeats. Absence, expiry, withdrawal or another epoch leaves the result unknown. | +| `fence_log_enrollment` | Competes against that exact attempt using the same original CAS token. A confirmed no-log successor prevents its delayed write. A newer empty record cannot authorize a rebased fence. | + +Prepared values belong to one directory instance and its clones. Providers must +assign a unique advancing epoch per recruitment request for each leader boot, +including after ambiguous results and closure. Retain the attempt before its +first CAS await. Rebase only before registry acceptance; a timeout cannot +authorize another attempt or another member set. The resulting opaque proof +observes canonical enrollment. Selected follower boot metadata does not prove +current receiver authority, fsync, registry publication or retirement. Transport +construction and fleet producer ownership remain application/host integration +responsibilities; these APIs alone do not install a journal-bound producer. + +The host's [managed follower binding](../../cellule-host/docs/lifecycle.md#managed-follower-enrollment) +owns this Pending-before-CAS protocol in the existing supervisor. Its authority +requires confirmed member retirement during automatic rotation and runtime drain. +`NodeLogAuthority::close` receives the complete joined +`NodeLogRetirementObservation`; ordinary implementations can use its `barrier()`. +`observe_retirement` captures responses before confirmation and close, and +`observe_shutdown_failure` retains failures before observation during managed +shutdown retries. These diagnostics grant no authority. Managed runtime shutdown +keeps retrying its original barrier, including across a cancelled drain waiter; +ordinary authorities retain their best-effort closure behavior. + **Epoch rotation controller.** The long-lived HTTP runtime applies the same barrier when shipping stops or the current epoch reaches `1_000_000` issued node-log frames: @@ -1413,47 +1502,69 @@ incarnation, and Cell epoch: requires the uncovered suffix to begin at the next commit. - Fully rooted groups need no overlay. -The recovery reservation stays owned through immutable pinning. -It writes shared immutable bundles and one small -manifest per affected Cell. +`RecoveryManifestStore::load_manifest` reads every original recovered scope from +the immutable manifest named by the canonical sealed log. It verifies the bounded +body, digest, original leader/epoch, canonical encoding and strictly ordered unique +scopes. The returned `RecoveryManifestInventory` includes every application; the +store's application does not filter it. A reconstructed controller can still read +this inventory after successors materialize roots and clear their overlay pointers. + +```rust,ignore +let manifest_digest = sealed.log().recovery_manifest().ok_or(Error::PendingPublication)?; +let inventory = manifests + .load_manifest(sealed.session(), sealed.log().epoch(), manifest_digest) + .await?; +for original in inventory.cells() { + // Use the original application's store for byte verification, then check + // its current authority and serving state through the ordinary paths. + let store = manifest_store_for(original.application)?; + let overlay = store + .load_overlay(original.cell, original.incarnation, &original.recovery) + .await?; +} +``` + +The inventory verifies metadata, not bundle availability or current successors. +It omits object-covered Cells that needed no overlay. A fleet operation must retain +the complete original writer set before relocation and separately verify every +current successor. An empty suffix has no manifest; it cannot prove an empty +original writer set or authorize node finalization. + +The recovery reservation stays owned through immutable pinning. It writes shared +immutable bundles and one manifest containing all affected scopes. The persisted +version 1 shape is shown below, formatted for reading; canonical bytes are compact +JSON, and integers are decimal strings or fixed-width hexadecimal strings. ```json { "version": 1, - "leader_session": "16-byte-hex", - "log_epoch": 3, - "application": "16-byte-hex", - "cell": "32-byte-hex", - "incarnation": "16-byte-hex", - "cell_epoch": 12, - "predecessor": { - "root": "32-byte-hex", - "txid": 500, - "checksum": 9223372036854776000, - "commit_sequence": 700 - }, - "entries": [ + "leader_session": "01010101010101010101010101010101", + "log_epoch": "3", + "cells": [ { - "node_sequence": 9002, - "bundle": "32-byte-hex", - "offset": 4096, - "length": 8192, - "ltx": { - "min_txid": 501, - "max_txid": 501, - "post_checksum": 9223372036854777000, - "commit_sequence": 701, - "blake3": "32-byte-hex" - } + "application": "03030303030303030303030303030303", + "cell": "0404040404040404040404040404040404040404040404040404040404040404", + "incarnation": "05050505050505050505050505050505", + "cell_epoch": "12", + "first_node_sequence": "9002", + "last_node_sequence": "9002", + "predecessor_digest": "0606060606060606060606060606060606060606060606060606060606060606", + "predecessor_txid": "500", + "predecessor_checksum": "8000000000000000", + "predecessor_commit_sequence": "700", + "final_txid": "501", + "final_checksum": "8000000000000001", + "final_commit_sequence": "701", + "bundle_digest": "0707070707070707070707070707070707070707070707070707070707070707" } ] } ``` -The final implementation uses canonical strict JSON or the existing canonical -binary manifest codec; it must not use floating-point numbers or permissive -unknown fields. The manifest digest covers its canonical bytes. Every bundle -extent is range-readable and independently BLAKE3-bound. +The manifest has at most 1,024 strictly ordered unique scopes and a two-MiB body. +Unknown fields and noncanonical values are rejected. Its digest covers every +canonical byte. The ordinary overlay loader verifies the referenced bundle and +retains its disk reservation until the overlay is dropped. ### Pin every affected Cell diff --git a/crates/cellule-runtime/docs/ltx-performance-audit.md b/crates/cellule-runtime/docs/ltx-performance-audit.md index 813a7e05..4a0ed543 100644 --- a/crates/cellule-runtime/docs/ltx-performance-audit.md +++ b/crates/cellule-runtime/docs/ltx-performance-audit.md @@ -450,7 +450,7 @@ The follow-up recovery audit reproduced a qualification gap shared with the compared main snapshot. The first-fault collector and both receipt selection checks used the startup log's members even though the receipt already contains the active log observed after the follower-only acknowledgement. The -[durability supervisor](../../cellule-host/src/durability.rs) can retire a +[durability supervisor](../../cellule-host/src/durability/mod.rs) can retire a fully covered log and recruit different members before that write. Two public receipt regressions demonstrated rejection of a valid later cohort and acceptance of an inactive acknowledging log. Selection now uses the active @@ -2868,7 +2868,7 @@ router, then [`ReplicaReadRouter::query`](../src/client/routing.rs). `selected` reads control and desired-reader policy concurrently on every selection. `NodeDirectory::select_readers` shares a signed membership scan for -at most one second. The selected [`CellReadReplica`](../src/client/replica.rs) +at most one second. The selected [`CellReadReplica`](../src/client/replica/mod.rs) executes SQL, then reads control and the owner's live enrollment before releasing output. A remote attempt also performs peer authentication. The ordinary owner query remains a sibling with different authority and lifecycle @@ -2975,7 +2975,7 @@ decoding had already occurred. The measured excess is origin traffic; neither **Evidence map.** The public entry is [`VerifiedRoot::open_read_only`](../../cellule-ltx/src/replica/mod.rs), with connection ownership in [`ReadOnlyRoot`](../../cellule-ltx/src/replica/read_only.rs). -[`CellReadReplica::refresh`](../src/client/replica.rs) creates such replacement +[`CellReadReplica::refresh`](../src/client/replica/mod.rs) creates such replacement views after an exact root advances. This diagnostic exercises that LTX opening and SQL path directly; it excludes runtime routing and authority confirmation. diff --git a/crates/cellule-runtime/docs/runtime.md b/crates/cellule-runtime/docs/runtime.md index 774d020c..0cf97e51 100644 --- a/crates/cellule-runtime/docs/runtime.md +++ b/crates/cellule-runtime/docs/runtime.md @@ -548,6 +548,13 @@ The node scheduler: **Release order.** A clean per-Cell drain closes SQLite before releasing control to `Idle`. Releasing control first would allow a successor to open while the previous writer still owns local mutable state. +`release_idle_cell_at` returns the canonical final `PublishedPosition` for its +exact source identity. `Error::CellReleaseRefused` preserves the original error +and proves that this request stopped before canonical deactivation began. +Errors from canonical close, publication or a lost response remain uncertain. +Local pressure eviction may release the same Cell independently; neither its +Idle root nor a missing actor supplies the fleet action's historical proof. + Node shutdown follows this order: ```mermaid diff --git a/crates/cellule-runtime/docs/storage.md b/crates/cellule-runtime/docs/storage.md index 0fde98b7..3fee60a1 100644 --- a/crates/cellule-runtime/docs/storage.md +++ b/crates/cellule-runtime/docs/storage.md @@ -143,6 +143,7 @@ cells/v1/apps//releases/.json cells/v1/apps//catalog/tenants//<00..ff>/head.json cells/v1/apps//catalog/objects/.json cells/v1/apps//cells//control.json +cells/v1/apps//cells//owner-history/v1//.json cells/v1/apps//cells//inc//objects/. cells/v1/apps//pins/.json cells/v1/apps//pins/objects/.json @@ -186,9 +187,212 @@ cells/v1/nodes/.json - **Writes.** Every replacement validates the runtime transition table before calling conditional update. +### Retain original owners before departure + +Before the ordinary release, takeover or tombstone CAS removes or replaces an +owner, `CellAuthority::transition` retains its complete original control at the +typed version 1 owner-history path. The body uses the same canonical 8 KiB control +codec. It includes unpublished and recovering controls, exact roots, code/schema, +and pinned recovery overlays. Same-owner publication and renewal write no history. +The existing control CAS remains the only ownership authority. + +Several departure proposals can observe the same epoch at different revisions. +History advances by ETag CAS; a delayed older proposal cannot overwrite a newer +observation. A failed proposal may leave a retained observation, which supplies no +departure proof. Ambiguous history replies require a confirmed equal or later +original observation before control departure. Storage failures remain errors. + +`owner_observation` reads one exact original epoch. `owner_history` collects every +closed ownership epoch in the current incarnation and appends its current owner, +then rechecks exact current authority. The caller bounds rows and its enclosing +deadline. Missing history returns `Error::OwnerHistoryIncomplete` with the original +Cell, incarnation and first missing epoch. Concurrent authority changes refuse the +read. Ordinary release closes the same epoch; tombstone consumes a final fence +epoch without inventing another owner. + +```rust,no_run +use cellule_runtime::{Result, identity::CellId}; +use cellule_runtime::control::authority::{CellAuthority, CellOwnerHistory}; + +async fn original_owners( + authority: &CellAuthority, + cell: CellId, + row_limit: usize, +) -> Result { + authority.owner_history(cell, row_limit).await +} +``` + +This is retained metadata for one Cell incarnation, not a complete physical-node +inventory or a root retention pin. Fleet collection must traverse authenticated +complete application/tenant catalogs, bind original boot/process joining and +operation scope, durably retain the selected complete set, and freshly verify +successor prefixes and serving. Include object-covered and unpublished writers. +A current-owner filter or recovered-suffix manifest alone omits originals after +takeover. Legacy departures, older binaries and restored controls may lack history; +never interpret that absence as proof that an original boot owned no Cells. +Mixed-binary qualification and a verified earlier inventory are required before +fleet completion can use such scope. The existing immutable-object collector +does not delete these metadata records or pin their historical root graphs. + +### Retain the successful acquisition input + +The canonical Idle acquisition, published takeover and rootless takeover paths +retain a `CellAcquisitionRecord` after the ownership CAS and any recovery-root +publication, before actor admission. It keeps the exact successful CAS input, +including a pinned recovery overlay, and the exact claimed/materialized Control. +Later renewal, publication, release and compaction do not replace this record. + +| Boundary | Contract | +| --- | --- | +| Path | `cells/v1/apps//cells//acquisitions/v1//.bin`; epoch is fixed-width 16-digit hex. | +| Codec | Version 1 domain, two big-endian length-prefixed canonical Controls, at most 8 KiB each; the complete envelope is bounded to 20 KiB. | +| Validation | Rebuild the ordinary Takeover from its input and require exact equality, or validate its one canonical PublishRecovery transition. Reject changed scope, owner, epoch, root, code or schema. | +| Publication | Strict immutable creation. Identical committed bytes can resolve an ambiguous reply; conflicting metadata cannot be overwritten. | +| Admission failure | No actor admission before confirmed retention. Published acquisition follows ordinary rollback. A rootless failure retains the claimed Recovering Control for ordinary bootstrap/takeover recovery. | +| Reader | `CellAuthority::acquisition_record` performs a bounded origin read and verifies Cell, incarnation and epoch. Storage and malformed-record failures remain errors. | + +The record may survive failed activation and proves no restore completion, +current serving, root retention, full acknowledged-prefix coverage or maintenance +settlement. Missing metadata remains `None`: initial bootstrap, direct activation +of an already-claimed Control, older binaries and cancellation before retention +can supply no record. Never infer successful acquisition or an empty writer set +from that absence. The existing object collector neither deletes these records +nor pins their referenced roots. Prefix verification must combine complete +original scope with independently checked ownership lineage, exact dependencies +and current native serving. + +```rust,no_run +use cellule_runtime::{Result, identity::{CellId, IncarnationId}}; +use cellule_runtime::control::authority::{CellAuthority, CellAcquisitionRecord}; + +async fn retained_claim( + authority: &CellAuthority, + cell: CellId, + incarnation: IncarnationId, + epoch: u64, +) -> Result> { + authority.acquisition_record(cell, incarnation, epoch).await +} +``` + ## Store roots as bounded immutable graphs +### Prove an exact root prefix after compaction + +Canonical publication retains verified `PreparedRoot` inputs before the root CAS, +including ordinary append, quiet/foreground compaction, migration and recovery +overlay publication. The version 1 runtime metadata path is +`cells/v1/apps//cells//root-lineage/v1//.bin`. +It adds no field to control JSON or LTX roots. Only native opaque preparations +produce links; failed root CAS leaves a verified proposal, never ownership. + +| Boundary | Contract | +| --- | --- | +| Codec | Fixed-width scoped root references, sorted distinct predecessors and a BLAKE3 envelope checksum; at most 64 predecessors and 8 KiB. | +| Byte-equivalent roots | Several valid preparations may produce identical root bytes. ETag CAS accumulates their original links; a delayed writer cannot erase an earlier input. | +| Publication | Confirm the required link before selecting its root. Lost replies require a confirmed equal/superset record. Existing publisher retry and acquisition rollback own failures; no new task or retry owner. | +| Fresh publication I/O | Strictly create the verified record first; its successful conditional write needs no preceding absence read. On a create conflict, read the verified existing record and perform at most one additive ETag merge. A stale merge cannot erase an earlier input; unresolved errors return to the existing publisher. | +| Preparation ordering | Retention overlaps immutable native uploads under the existing publisher; both finish before a complete proposal escapes. An exact private preparation confirmation avoids a second write before authority CAS. External and rebased proposals use the same canonical retention path before CAS. | +| Identity compaction | Preparing the same exact root introduces no self-link. Traversal also detects repeated roots, so representation cycles cannot loop indefinitely. | +| Prefix | `verify_root_prefix` must reach the exact requested digest, scope, TXID, checksum and sequence. Higher counters alone are insufficient. | +| Bounds | The caller permits at most 10,000 expanded/queued lineage roots. Complete origin inventory also caps at 10,000 objects. One enclosing deadline bounds the work; excess refuses without truncating evidence. | +| Availability | After finding a verified derivation path, authenticate the successor's complete current origin graph through the canonical `reachable_objects_bounded` walk, including every body/extent; metadata caches cannot substitute. | +| Missing data | Missing legacy/manual-publication links yield `RootLineageIncomplete`; a complete path search that cannot reach the prefix yields `RootPrefixUnproven`. Storage, corrupt metadata and missing/corrupt graph dependencies preserve their errors. | + +The runtime wrapper reserves transient metadata in the existing node retained-byte +ledger before I/O: 16 MiB for bounded graph/cache/fetch/decode work, plus 1 KiB +per permitted lineage root and each of 10,000 origin objects. Vector growth and +map overhead are included conservatively. The token spans awaited work and drops +on success, error or caller cancellation. Origin body bytes stream through the +configured shared LTX I/O host; application Store adapters supply bounded chunks. +Direct authority callers own equivalent admission. No new task or scheduler is +created; the existing finite fleet action owner retains accepted work. + +`VerifiedRootPrefix` is an opaque point observation of native verified derivation +and complete successor dependency availability. It is not selected authority, +current actor serving, an immutable-root pin, a complete original physical-boot +inventory, recovered-suffix scope or maintenance settlement. Applications +authenticate canonical backend mappings and protect these runtime metadata writes +with the same storage authorization as authority. Old root objects may be collected +after valid compaction; the retained preparation links remain metadata, and the +verified successor graph must still contain the current state. The existing +immutable-object collector does not delete these lineage metadata records. + +Use `CellRuntime::verify_root_prefix` for the runtime's shared configured LTX I/O +host; standalone callers can use `CellAuthority::verify_root_prefix`. Neither path +starts a scheduler, changes Cell authority or invents a legacy link. + +```rust,no_run +use cellule_runtime::{Result, cell::{actor::CellRuntime, catalog::CatalogProof}}; +use cellule_runtime::control::authority::{CellAuthority, VerifiedRootPrefix}; +use cellule_runtime::ltx::{CellReplica, RootRef}; + +async fn verify_prefix( + runtime: &CellRuntime, + catalog: &CatalogProof, + authority: &CellAuthority, + replica: CellReplica, + original: RootRef, + successor: RootRef, +) -> Result { + runtime.verify_root_prefix(catalog, authority, replica, original, successor, 10_000).await +} +``` + +### Prove an original sealed recovery suffix + +`CellRuntime::verify_recovered_prefix` consumes one exact `PinnedRecoveryCell` +from the original digest-verified manifest. It binds that row to the retained +closed owner, then checks bounded canonical acquisitions through the selected +Serving epoch. An interrupted claim can leave no acquisition record; a later +materialization must retain the identical original overlay. Its original Cell +epoch remains the manifest epoch, rather than the later acquisition input epoch. +All leader/log/manifest, node-sequence, predecessor and final boundaries compare +exactly. A matching endpoint without the original overlay is insufficient. + +| Boundary | Contract | +| --- | --- | +| Materialization | Select the last canonical acquisition with the exact original overlay; require its exact final TXID, checksum and sequence, then verify native derivation to the current successor and every current origin dependency. | +| Bounds | The same caller limit caps both acquisition epochs and lineage traversal, at most 10,000. Excess refuses; storage/corrupt-record errors remain errors. | +| Missing metadata | An absent original owner yields `OwnerHistoryIncomplete`. Without a matching materialization, missing acquisition records yield `AcquisitionHistoryIncomplete`; different recovery inputs refuse. No legacy rows are fabricated. | +| Selected authority | Require Serving at the selected root and recheck owner, incarnation, epoch, state and exact root after origin verification. Lease renewal may continue. | +| Resource ownership | Reuse the root verifier's shared memory reservation and configured LTX I/O host before metadata/origin I/O. The caller supplies the finite deadline; cancellation drops the reservation. | +| Proof scope | `VerifiedRecoveryPrefix` retains the exact required row, materialization epoch and opaque root-prefix proof. It grants no native serving, root pin, authenticated physical boot/backend scope or aggregate settlement. | + +```rust,no_run +use cellule_runtime::{Result, cell::{actor::CellRuntime, catalog::CatalogProof}}; +use cellule_runtime::control::authority::{CellAuthority, VerifiedRecoveryPrefix}; +use cellule_runtime::ltx::{CellReplica, RootRef}; +use cellule_runtime::recovery::manifest::PinnedRecoveryCell; + +async fn verify_suffix( + runtime: &CellRuntime, + catalog: &CatalogProof, + authority: &CellAuthority, + replica: CellReplica, + original: &PinnedRecoveryCell, + successor: RootRef, +) -> Result { + runtime.verify_recovered_prefix(catalog, authority, replica, original, successor, 10_000).await +} +``` + +Fleet movement requires the exact released root or retained recovery +materialization as its prefix. Recovered serving first compares the entire +journal recovery input/result with canonical acquisition metadata. With a pinned +overlay, it repeats the existing read-only provider lookup, loads the original +manifest and verifies the exact original row through `verify_recovered_prefix`. +The host charges an additional 8-MiB transient acquisition/manifest envelope +before I/O and holds it through verification. After complete origin verification, it rechecks +the same native actor through ordinary FIFO admission, selected root/owner/epoch +and native ownership inventory before returning fresh serving evidence. Durable +historical results remain historical; their replay does not refresh this proof. +Complete original-writer/suffix aggregation, physical boot/process scope, reader +and follower replacement policy, and terminal action joining remain separate +requirements before role settlement or finalization. + One root identifies the complete SQLite state at one transaction ID. ```mermaid @@ -472,6 +676,41 @@ absence. - `CellAuthority::create_initial` requires a verified `CatalogProof`. - Readers recompute every Cell ID and enforce ordering across page boundaries. +**Complete operation traversal.** `CellCatalog::scan_all(limit)` captures every +head before returning the first page. It streams through the same verified shard +reader and enforces a nonzero cumulative row bound. A partial, failed or cancelled +scan supplies no receipt. `finish()` requires observed end-of-stream and rechecks +all 256 original heads, including absence, revision, locators and ETag. Changes +fail rather than silently replacing the captured set. The receipt exposes tenant, +application, entry count and each original revision/page-digest list; `revalidate()` +reads the same original adapter again. + +```rust +use cellule_runtime::cell::catalog::{CellCatalog, CatalogScanReceipt}; +use cellule_runtime::identity::CellId; + +async fn collect_cells( + catalog: &CellCatalog, + row_limit: usize, +) -> cellule_runtime::Result<(Vec, CatalogScanReceipt)> { + let mut scan = catalog.scan_all(row_limit).await?; + let mut cells = Vec::new(); + while let Some(page) = scan.next_page().await? { + cells.extend(page.entries().iter().map(|proof| proof.entry().cell())); + } + let receipt = scan.finish().await?; + Ok((cells, receipt)) +} +``` + +The scan retains at most 256 heads of 256 locators and returns at most 256 +entries per page. Callers account for their retained output. The heads and final +checks are sequential observations; they are not a global catalog transaction. +Application authentication, complete application/tenant enumeration, original +process and accepted-work joining, authority/history collection and durable +operation binding remain separate required barriers. A receipt pins no objects, +proves no successor serving and does not establish continuing page availability. + **Tenant scope and retention** - Tenant-scoped catalog heads do not make release or backup management diff --git a/crates/cellule-runtime/docs/vfs-ltx-scale-plan.md b/crates/cellule-runtime/docs/vfs-ltx-scale-plan.md index d28a8972..32e1c9c5 100644 --- a/crates/cellule-runtime/docs/vfs-ltx-scale-plan.md +++ b/crates/cellule-runtime/docs/vfs-ltx-scale-plan.md @@ -48,7 +48,7 @@ replication protocol. | Responsibility | Existing owner | Contract to preserve | | --- | --- | --- | | Product authentication, authorization, and ingress | [HTTP router](https://github.com/crabbuild/crab/blob/beb439039cb37e750afe6625a2358101c70d1191/crates/crab-http-server/src/cells/router.rs) and [peer receiver](https://github.com/crabbuild/crab/blob/beb439039cb37e750afe6625a2358101c70d1191/crates/crab-http-server/src/peer.rs) | Authorize the target and action before dispatch; bound peer hops | -| Cell identity, owner, epoch, lifecycle, root | [Cell authority](../src/control/authority.rs) | Only conditional control writes grant or change ownership | +| Cell identity, owner, epoch, lifecycle, root | [Cell authority](../src/control/authority/mod.rs) | Only conditional control writes grant or change ownership | | Actor admission and one SQL writer | [Cell actor](../src/cell/actor/mod.rs) | A fenced or draining actor refuses queued and new work | | Local SQLite, WAL capture, LTX | [Db](../../cellule-ltx/src/db/mod.rs) | Local commit alone never releases a response | | Sparse exact-root page access | [Writable VFS](../../cellule-ltx/src/writable_vfs/mod.rs) and [paged I/O](../../cellule-ltx/src/paged_io.rs) | Verify inherited pages, reserve disk, and create a fresh local file | diff --git a/crates/cellule-runtime/src/cell/actor/acquire.rs b/crates/cellule-runtime/src/cell/actor/acquire.rs index 2faedc58..901dc592 100644 --- a/crates/cellule-runtime/src/cell/actor/acquire.rs +++ b/crates/cellule-runtime/src/cell/actor/acquire.rs @@ -211,6 +211,11 @@ impl CellRuntime { } } }; + // Keep the actual successful input before initialization can admit + // an actor. Rootless takeover has no acknowledged database prefix. + authority + .retain_acquisition(current.value(), claimed.value()) + .await?; let incarnation = claimed.value().incarnation; let schema = claimed.value().schema; return self @@ -260,6 +265,7 @@ impl CellRuntime { observed, destination, reservation, + None, ) .await } @@ -273,10 +279,33 @@ impl CellRuntime { observed: VersionedControl, destination: PathBuf, owner: Owner, + ) -> crate::Result { + self.acquire_idle_restored_observed( + catalog, + replica, + authority, + observed, + destination, + owner, + None, + ) + .await + } + + /// Acquires an Idle root with confirmed durable recording before CAS and + /// before actor admission. Uses the same canonical acquisition and rollback. + pub async fn acquire_idle_restored_observed( + &self, + catalog: CatalogProof, + replica: cellule_ltx::CellReplica, + authority: CellAuthority, + observed: VersionedControl, + destination: PathBuf, + owner: Owner, + observer: Option>, ) -> crate::Result { self.ensure_acquiring()?; self.check_application_limits(&catalog, replica.limits())?; - let rollback_node_lease = self.inner.node_lease.guard()?; self.claiming_cell(&catalog, &observed, &owner)?; if observed.value().state != crate::control::ControlState::Idle || observed.value().owner.is_some() @@ -290,7 +319,53 @@ impl CellRuntime { .replica_with_directory_cache(replica, &destination) .await?; let reservation = self.inner.pool.reserve_activation()?; + self.acquire_idle_reserved( + catalog, + replica, + authority, + observed, + destination, + owner, + reservation, + None, + observer, + ) + .await + } + + #[expect( + clippy::too_many_arguments, + reason = "canonical idle acquisition consumes its explicit prepaid resources" + )] + pub(super) async fn acquire_idle_reserved( + &self, + catalog: CatalogProof, + replica: cellule_ltx::CellReplica, + authority: CellAuthority, + observed: VersionedControl, + destination: PathBuf, + owner: Owner, + reservation: CellReservation, + job: Option, + observer: Option>, + ) -> crate::Result { + self.ensure_acquiring()?; + self.check_application_limits(&catalog, replica.limits())?; + let rollback_node_lease = self.inner.node_lease.guard()?; + self.claiming_cell(&catalog, &observed, &owner)?; + if observed.value().state != crate::control::ControlState::Idle + || observed.value().owner.is_some() + || observed.value().root.is_none() + { + return Err(Error::Control( + "idle acquisition requires a published idle control", + )); + } let successor = observed.value().takeover(owner)?; + if let Some(observer) = &observer { + observer.before_claim(observed.value()).await?; + } + self.ensure_acquiring()?; let ownership_started = std::time::Instant::now(); let claimed = match authority .transition(&observed, successor.clone(), Transition::Takeover) @@ -315,17 +390,28 @@ impl CellRuntime { let rollback_authority = authority.clone(); let rollback_claim = claimed.clone(); let rollback_replica = replica.clone(); - match self - .activate_restored_reserved( + let activation = async { + authority + .retain_acquisition(observed.value(), claimed.value()) + .await?; + if let Some(observer) = &observer { + observer + .before_activation(observed.value(), claimed.value()) + .await?; + } + self.activate_restored_reserved( catalog, replica, authority, claimed, destination, reservation, + job, ) .await - { + } + .await; + match activation { Ok(handle) => Ok(handle), Err(error) => { match rollback_failed_acquisition( @@ -349,6 +435,38 @@ impl CellRuntime { reason = "the takeover boundary keeps every authority, recovery and activation input explicit" )] pub async fn takeover_restored( + &self, + catalog: CatalogProof, + replica: cellule_ltx::CellReplica, + authority: CellAuthority, + observed: VersionedControl, + takeover: crate::node::NodeTakeoverProof, + recovery_store: crate::recovery::manifest::RecoveryManifestStore, + destination: PathBuf, + owner: Owner, + ) -> crate::Result { + self.takeover_restored_observed( + catalog, + replica, + authority, + observed, + takeover, + recovery_store, + destination, + owner, + None, + ) + .await + } + + /// Takes over a fenced predecessor with durable input and recovery-position + /// recording. Callback failure cannot admit an actor; post-CAS failure uses + /// the ordinary rollback path. A lost waiter must remain caller-owned. + #[expect( + clippy::too_many_arguments, + reason = "the recorder supplements explicit recovery and authority inputs" + )] + pub async fn takeover_restored_observed( &self, catalog: CatalogProof, replica: cellule_ltx::CellReplica, @@ -358,6 +476,7 @@ impl CellRuntime { recovery_store: crate::recovery::manifest::RecoveryManifestStore, destination: PathBuf, owner: Owner, + observer: Option>, ) -> crate::Result { self.ensure_acquiring()?; self.check_application_limits(&catalog, replica.limits())?; @@ -397,6 +516,10 @@ impl CellRuntime { } let reservation = self.inner.pool.reserve_activation()?; let successor = current.value().takeover(owner.clone())?; + if let Some(observer) = &observer { + observer.before_claim(current.value()).await?; + } + self.ensure_acquiring()?; let claimed = match authority .transition(¤t, successor.clone(), Transition::Takeover) .await @@ -438,17 +561,28 @@ impl CellRuntime { } }; let rollback_claim = claimed.clone(); - return match self - .activate_restored_reserved( + let activation = async { + authority + .retain_acquisition(current.value(), claimed.value()) + .await?; + if let Some(observer) = &observer { + observer + .before_activation(current.value(), claimed.value()) + .await?; + } + self.activate_restored_reserved( catalog, replica, authority, claimed, destination, reservation, + None, ) .await - { + } + .await; + return match activation { Ok(handle) => Ok(handle), Err(error) => { match rollback_failed_acquisition( @@ -491,6 +625,7 @@ impl CellRuntime { let successor = observed .value() .publish_recovery(&prepared, observed.value().next_due_ms)?; + authority.retain_root_lineage(&prepared).await?; match authority .transition(&observed, successor.clone(), Transition::PublishRecovery) .await @@ -518,7 +653,9 @@ impl CellRuntime { observed: VersionedControl, destination: PathBuf, reservation: CellReservation, + job: Option, ) -> crate::Result { + self.ensure_acquiring()?; // Acquisition installed one cache owner before recovery. Keep that // replica through root verification, SQLite and publisher activation. let cell = self.activation_cell(&catalog, &observed)?; @@ -556,6 +693,7 @@ impl CellRuntime { .await? } }; + self.ensure_acquiring()?; let current = authority.load(cell).await?.ok_or(Error::Fenced)?; if !current.value().is_same_or_pure_renewal_of(observed.value()) { return Err(Error::Fenced); @@ -571,6 +709,7 @@ impl CellRuntime { schema, root, reservation, + job, })), replica, authority, @@ -638,7 +777,7 @@ impl CellRuntime { Ok(cell) } - fn check_application_limits( + pub(super) fn check_application_limits( &self, catalog: &CatalogProof, limits: cellule_ltx::Limits, @@ -687,12 +826,9 @@ impl CellRuntime { self.inner.node_lease.check() } - fn ensure_acquiring(&self) -> crate::Result<()> { + pub(super) fn ensure_acquiring(&self) -> crate::Result<()> { self.ensure_running()?; - if !self.inner.accepting_cells.load(Ordering::Acquire) { - return Err(Error::CellDraining); - } - Ok(()) + self.inner.node_admission.check_new_role() } async fn activate_inner( @@ -748,14 +884,19 @@ impl CellRuntime { &self, replica: cellule_ltx::CellReplica, destination: &Path, + ) -> crate::Result { + Self::replica_with_host_cache(replica, self.inner.replica_host.clone(), destination).await + } + + pub(super) async fn replica_with_host_cache( + replica: cellule_ltx::CellReplica, + host: cellule_ltx::Host, + destination: &Path, ) -> crate::Result { let scratch_directory = destination .parent() .ok_or(Error::Control("Cell activation destination has no parent"))?; - let host = self - .inner - .replica_host - .clone() + let host = host .with_directory_cache(scratch_directory.join(".cellule-directory-cache")) .await?; Ok(replica.with_host(host)) diff --git a/crates/cellule-runtime/src/cell/actor/acquisition_observer.rs b/crates/cellule-runtime/src/cell/actor/acquisition_observer.rs new file mode 100644 index 00000000..f33e7d19 --- /dev/null +++ b/crates/cellule-runtime/src/cell/actor/acquisition_observer.rs @@ -0,0 +1,29 @@ +use std::{future::Future, pin::Pin}; + +use crate::control::Control; + +/// A durable recording call at a canonical acquisition boundary. +pub type AcquisitionObservation<'a> = Pin> + Send + 'a>>; + +/// Trusted recorder for the exact input and restored position of acquisition. +/// +/// The runtime waits for confirmation before ownership CAS and before actor +/// admission. The adapter must retain immutable original records and reject +/// incompatible repeats. These callbacks grant no ownership or authentication. +/// A lost write reply is an error, not permission to continue. The caller must +/// own this acquisition future independently of transport waiters. +pub trait AcquisitionObserver: Send + Sync + 'static { + /// Records the exact canonical input before its ownership CAS. Takeover may + /// retry a changed predecessor, so an adapter must explicitly accept or + /// reject each input; it must never silently overwrite an earlier basis. + fn before_claim<'a>(&'a self, input: &'a Control) -> AcquisitionObservation<'a>; + + /// Records the exact published recovery/root position before actor activation. + /// `input` is the confirmed input of the successful CAS, not a later reread. + /// Failure follows canonical acquisition rollback and preserves its error. + fn before_activation<'a>( + &'a self, + input: &'a Control, + restored: &'a Control, + ) -> AcquisitionObservation<'a>; +} diff --git a/crates/cellule-runtime/src/cell/actor/admission.rs b/crates/cellule-runtime/src/cell/actor/admission.rs index fab46b15..c6eee594 100644 --- a/crates/cellule-runtime/src/cell/actor/admission.rs +++ b/crates/cellule-runtime/src/cell/actor/admission.rs @@ -58,7 +58,9 @@ pub(super) fn fence_active(active: &mut ActiveCell) { active.coordination.step(CoordinationInput::Fence); fence_admission(&active.admission); if let Some(transfer) = active.transfer.take() { - let _ = transfer.reply.send(Err(Error::Fenced)); + transfer + .reply + .refuse(DrainBlocker::IncompleteObservation, Error::Fenced); } active.inventory_refreshing = false; while let Some(publication) = active.publications.pop_front() { @@ -104,6 +106,7 @@ pub(super) fn new_cell_admission(owner_fence: crate::control::OwnerFence) -> Arc requests: Arc::new(Semaphore::new(CELL_REQUESTS)), bytes: Arc::new(Semaphore::new(CELL_BYTES)), draining: AtomicBool::new(false), + maintenance_quiescing: AtomicBool::new(false), fenced: AtomicBool::new(false), }) } diff --git a/crates/cellule-runtime/src/cell/actor/handle.rs b/crates/cellule-runtime/src/cell/actor/handle.rs index 9ed69594..d9cda592 100644 --- a/crates/cellule-runtime/src/cell/actor/handle.rs +++ b/crates/cellule-runtime/src/cell/actor/handle.rs @@ -17,6 +17,7 @@ use crate::cell::catalog::CatalogProof; use crate::cell::catalog::CatalogRole; use crate::cell::executor::StoredOutcome; use crate::cell::executor::{MutationIdentity, Resolution}; +use crate::coordination::AdmissionKind; use crate::fleet::resource::{ResourceCost, ResourceReservation}; use crate::identity::IncarnationId; use crate::identity::{CellId, Digest}; @@ -44,10 +45,20 @@ pub(super) struct CellAdmission { pub(super) requests: Arc, pub(super) bytes: Arc, pub(super) draining: AtomicBool, + pub(super) maintenance_quiescing: AtomicBool, pub(super) fenced: AtomicBool, } +pub(crate) struct CommandWork { + pub(crate) identity: MutationIdentity, + pub(crate) operation_digest: Digest, + pub(crate) now_ms: i64, + pub(crate) operation_bytes: usize, + pub(crate) max_result_bytes: usize, +} + pub(super) struct WorkAdmission { + pub(super) kind: AdmissionKind, pub(super) _request: OwnedSemaphorePermit, pub(super) _cell_bytes: OwnedSemaphorePermit, pub(super) _node_bytes: ResourceReservation, @@ -155,7 +166,47 @@ impl CellHandle { + Send + 'static, { - let admission = self.reserve_work(operation_bytes, max_result_bytes)?; + self.execute_registered( + false, + CommandWork { + identity, + operation_digest, + now_ms, + operation_bytes, + max_result_bytes, + }, + handler, + ) + .await + } + + pub(crate) async fn execute_registered( + &self, + lease_completion: bool, + request: CommandWork, + handler: F, + ) -> crate::Result + where + F: for<'connection> FnOnce( + &cellule_ltx::rusqlite::Transaction<'connection>, + ) + -> crate::Result + + Send + + 'static, + { + let CommandWork { + identity, + operation_digest, + now_ms, + operation_bytes, + max_result_bytes, + } = request; + let kind = if lease_completion { + AdmissionKind::LeaseCommand + } else { + AdmissionKind::Command + }; + let admission = self.reserve_work_kind(kind, operation_bytes, max_result_bytes)?; let (reply, response) = oneshot::channel(); self.inner .sender @@ -253,7 +304,26 @@ impl CellHandle { where F: FnOnce(&cellule_ltx::rusqlite::Connection) -> crate::Result> + Send + 'static, { - let admission = self.reserve_work(operation_bytes, max_result_bytes)?; + self.query_registered(false, operation_bytes, max_result_bytes, handler) + .await + } + + pub(crate) async fn query_registered( + &self, + lease_validation: bool, + operation_bytes: usize, + max_result_bytes: usize, + handler: F, + ) -> crate::Result> + where + F: FnOnce(&cellule_ltx::rusqlite::Connection) -> crate::Result> + Send + 'static, + { + let kind = if lease_validation { + AdmissionKind::LeaseQuery + } else { + AdmissionKind::Query + }; + let admission = self.reserve_work_kind(kind, operation_bytes, max_result_bytes)?; let (reply, response) = oneshot::channel(); self.inner .sender @@ -300,7 +370,7 @@ impl CellHandle { if self.admission.fenced.load(Ordering::Acquire) { return Ok(Resolution::Unknown); } - let admission = match self.reserve_work(48, max_result_bytes) { + let admission = match self.reserve_work_kind(AdmissionKind::Resolve, 48, max_result_bytes) { Ok(admission) => admission, Err(Error::Fenced) => return Ok(Resolution::Unknown), Err(error) => return Err(error), @@ -341,7 +411,7 @@ impl CellHandle { if self.admission.fenced.load(Ordering::Acquire) { return Ok(Resolution::Unknown); } - let admission = match self.reserve_work(72, max_result_bytes) { + let admission = match self.reserve_work_kind(AdmissionKind::Resolve, 72, max_result_bytes) { Ok(admission) => admission, Err(Error::Fenced) => return Ok(Resolution::Unknown), Err(error) => return Err(error), @@ -444,6 +514,15 @@ impl CellHandle { &self, operation_bytes: usize, max_result_bytes: usize, + ) -> crate::Result { + self.reserve_work_kind(AdmissionKind::Command, operation_bytes, max_result_bytes) + } + + fn reserve_work_kind( + &self, + kind: AdmissionKind, + operation_bytes: usize, + max_result_bytes: usize, ) -> crate::Result { if self.inner.shutting_down.load(Ordering::Acquire) { return Err(Error::RuntimeClosed); @@ -459,10 +538,14 @@ impl CellHandle { if self.admission.fenced.load(Ordering::Acquire) { return Err(Error::Fenced); } - if self.admission.draining.load(Ordering::Acquire) { + if self.admission.draining.load(Ordering::Acquire) + || (self.admission.maintenance_quiescing.load(Ordering::Acquire) + && !kind.allowed_while_quiescing()) + { return Err(Error::CellDraining); } let admission = WorkAdmission { + kind, _request: try_one(self.admission.requests.clone(), "Cell mailbox requests")?, _cell_bytes: try_many( self.admission.bytes.clone(), @@ -481,7 +564,10 @@ impl CellHandle { if self.admission.fenced.load(Ordering::Acquire) { return Err(Error::Fenced); } - if self.admission.draining.load(Ordering::Acquire) { + if self.admission.draining.load(Ordering::Acquire) + || (self.admission.maintenance_quiescing.load(Ordering::Acquire) + && !kind.allowed_while_quiescing()) + { return Err(Error::CellDraining); } self.inner.node_lease.check()?; diff --git a/crates/cellule-runtime/src/cell/actor/inventory/demand.rs b/crates/cellule-runtime/src/cell/actor/inventory/demand.rs new file mode 100644 index 00000000..0810df9e --- /dev/null +++ b/crates/cellule-runtime/src/cell/actor/inventory/demand.rs @@ -0,0 +1,161 @@ +//! Generation-local demand samples. Page reads cannot manufacture stability. + +use crate::cell::worker::{ACTIVE_CELL_PAGE_CACHE_BYTES, WorkerCellInventory}; +use crate::fleet::operations::TransferCost; +use crate::fleet::resource::{ + ACTIVE_CELL_FILE_DESCRIPTORS, ACTIVE_CELL_NATIVE_BYTES, READ_REPLICA_FILE_DESCRIPTORS, + READ_REPLICA_NATIVE_BYTES, +}; +use crate::{Error, Result}; + +const REFRESH_MS: i64 = 15_000; +const FRESH_MS: i64 = 30_000; +const RETRY_MS: i64 = 1_000; +const SECOND_SAMPLE_MS: i64 = 100; + +#[derive(Default)] +pub(in crate::cell::actor) struct CellDemandState { + sample: Option, + stable_observations: u8, + last_measurement_ms: Option, + last_attempt_ms: i64, + refresh_after_ms: i64, +} + +impl CellDemandState { + pub(in crate::cell::actor) fn clear(&mut self) { + self.sample = None; + self.stable_observations = 0; + self.refresh_after_ms = 0; + } + + pub(in crate::cell::actor) fn failed(&mut self, now_ms: i64) { + self.clear(); + self.last_attempt_ms = self.last_attempt_ms.max(now_ms); + self.refresh_after_ms = now_ms.saturating_add(RETRY_MS); + } + + pub(in crate::cell::actor) fn record( + &mut self, + sample: WorkerCellInventory, + limits: cellule_ltx::Limits, + published_sequence: u64, + ) -> Result<()> { + self.last_attempt_ms = self.last_attempt_ms.max(sample.observed_at_ms); + let validation = || { + if sample.observed_at_ms < 0 + || sample.persisted_work.is_unknown() + || self + .last_measurement_ms + .is_some_and(|at| sample.observed_at_ms < at) + { + return Err(Error::Control( + "Cell demand measurement regressed or is unknown", + )); + } + if sample.commit_sequence != published_sequence { + return Err(Error::PendingPublication); + } + transfer_cost(limits, sample.database_bytes).map(|_| ()) + }; + if let Err(error) = validation() { + self.failed(sample.observed_at_ms); + return Err(error); + } + let unchanged = self.sample.is_some_and(|old| { + old.commit_sequence == sample.commit_sequence + && old.database_bytes == sample.database_bytes + && old.persisted_work == sample.persisted_work + && old.transfer_work == sample.transfer_work + && old.maintenance_work == sample.maintenance_work + }); + self.stable_observations = if unchanged { + if self + .sample + .is_some_and(|old| sample.observed_at_ms > old.observed_at_ms) + { + self.stable_observations.saturating_add(1).min(2) + } else { + self.stable_observations + } + } else { + 1 + }; + self.last_measurement_ms = Some(sample.observed_at_ms); + self.sample = Some(sample); + self.refresh_after_ms = + sample + .observed_at_ms + .saturating_add(if self.stable_observations < 2 { + SECOND_SAMPLE_MS + } else { + REFRESH_MS + }); + Ok(()) + } + + pub(in crate::cell::actor) fn should_refresh(&self, now_ms: i64, work_unknown: bool) -> bool { + now_ms >= self.refresh_after_ms + && (work_unknown + || self.sample.is_none() + || self.stable_observations < 2 + || self + .sample + .is_some_and(|s| now_ms.saturating_sub(s.observed_at_ms) >= REFRESH_MS)) + } + + pub(in crate::cell::actor) fn refresh_priority(&self) -> i64 { + self.last_attempt_ms + } + + pub(in crate::cell::actor) fn fresh_sample( + &self, + now_ms: i64, + published_sequence: u64, + ) -> Option { + self.sample.filter(|s| { + now_ms >= s.observed_at_ms + && now_ms.saturating_sub(s.observed_at_ms) < FRESH_MS + && s.commit_sequence == published_sequence + }) + } + + pub(in crate::cell::actor) fn stable_observations(&self) -> u8 { + self.stable_observations + } +} + +pub(in crate::cell::actor) fn transfer_cost( + limits: cellule_ltx::Limits, + database_bytes: u64, +) -> Result { + if database_bytes == 0 + || database_bytes > limits.max_database_bytes + || limits.max_database_bytes < 512 + || limits.max_file_bytes < 128 + || limits.max_plan_bytes < limits.max_file_bytes + { + return Err(Error::Capacity( + "Cell demand exceeds its validated LTX bounds", + )); + } + // Bound full-image restore, retained plan inputs, and the canonical restore's + // 64-MiB headroom. Use the validated database ceiling rather than today's + // sparse-file length, so growth before release does not underreserve disk. + let disk_bytes = limits + .max_database_bytes + .checked_mul(2) + .and_then(|bytes| bytes.checked_add(limits.max_plan_bytes)) + .and_then(|bytes| bytes.checked_add(64 << 20)) + .ok_or(Error::Capacity("Cell transfer disk cost overflow"))?; + Ok(TransferCost { + // Reuse the framework's conservative cache/native accounting. This is + // admission demand, not a measured process RSS guarantee. + memory_bytes: ACTIVE_CELL_NATIVE_BYTES as u64 + + ACTIVE_CELL_PAGE_CACHE_BYTES + + READ_REPLICA_NATIVE_BYTES as u64, + disk_bytes, + file_descriptors: (ACTIVE_CELL_FILE_DESCRIPTORS + READ_REPLICA_FILE_DESCRIPTORS) as u32, + job_credits: 1, + }) +} diff --git a/crates/cellule-runtime/src/cell/actor/inventory/mod.rs b/crates/cellule-runtime/src/cell/actor/inventory/mod.rs new file mode 100644 index 00000000..424be5dd --- /dev/null +++ b/crates/cellule-runtime/src/cell/actor/inventory/mod.rs @@ -0,0 +1,318 @@ +//! Bounded actor-owned advisory inventory. It cannot authorize a Cell release. + +use super::*; +use crate::fleet::operations::{DrainBlocker, MAX_PAGE_ENTRIES, PublishedPosition, TransferCost}; +use crate::identity::IncarnationId; + +mod demand; +pub(super) use demand::{CellDemandState, transfer_cost}; + +/// Continuation of a sorted ownership-topology scan, scoped to this runtime. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct CellInventoryCursor { + topology: Digest, + after: CellId, +} + +impl CellInventoryCursor { + /// Returns the ownership topology fingerprint; it confers no authority. + #[must_use] + pub const fn topology(self) -> Digest { + self.topology + } + /// Returns the last Cell ID emitted by the preceding page. + #[must_use] + pub const fn after(self) -> CellId { + self.after + } + /// Encodes the fixed-width opaque continuation for application transports. + #[must_use] + pub fn to_bytes(self) -> [u8; 64] { + let mut bytes = [0; 64]; + bytes[..32].copy_from_slice(self.topology.as_bytes()); + bytes[32..].copy_from_slice(self.after.as_bytes()); + bytes + } + /// Restores exactly one fixed-width continuation; the actor revalidates it. + pub fn from_bytes(bytes: &[u8]) -> crate::Result { + let bytes: &[u8; 64] = bytes + .try_into() + .map_err(|_| Error::Node("invalid Cell inventory cursor width"))?; + let mut topology = [0; 32]; + topology.copy_from_slice(&bytes[..32]); + let mut after = [0; 32]; + after.copy_from_slice(&bytes[32..]); + Ok(Self { + topology: Digest::from_bytes(topology), + after: CellId::from_bytes(after), + }) + } +} + +/// One generation-bound owner, including busy or draining Cells. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct OwnedCellObservation { + /// Verified catalog target; actor observations do not replace its proof. + pub target: CellTarget, + /// Local generation required by the canonical release path. + pub generation: u64, + /// Authority-pinned incarnation loaded by this activation. + pub incarnation: IncarnationId, + /// Current executable contract identity. + pub code: Digest, + /// Current schema version. + pub schema: u32, + /// Namespace's persistent role. + pub role: CatalogRole, + /// Activation time, unaffected by later use or publication. + pub resident_since_ms: i64, + /// Most recent admitted work time; separate from residence. + pub last_used_ms: i64, + /// Actor's last published authority position, if available locally. + pub position: Option, + /// Conservative receiver cost from a fresh worker sample; required by idle movement. + pub cost: Option, + /// Configured peak receiver envelope for explicit busy maintenance. This is + /// independent of worker settlement and grants no release authority. Blob + /// owners remain unsupported until their external barriers are implemented. + pub maintenance_cost: Option, + /// Logical SQLite size measured by the serialized worker, distinct from restore cost. + pub database_bytes: Option, + /// Time of the worker measurement; reading this page does not refresh it. + pub sampled_at_ms: Option, + /// Consecutive unchanged worker observations, capped at two. + pub stable_observations: u8, + /// First executable work class blocking ordinary idle movement, if known. + pub work_blocker: Option, + /// Sticky foreground closure installed for this exact local activation. + /// Native lease completion and validation remain available until release. + pub quiescing: bool, + /// Fresh primitive readiness from the same serialized worker measurement. + /// Absence is unknown; a transferable value alone proves no actor barrier. + pub maintenance_work: + Option, + /// Local advisory blockers; action acceptance rechecks them. + pub blockers: Vec, +} + +/// All local ownership obligations appear even when metadata is unavailable. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum CellInventoryEntry { + /// A live actor whose catalog and generation are known. + Owned(Box), + /// Activation/close/release is outstanding; do not interpret it as absent. + Transitioning { + /// Stable Cell ID whose lifecycle task is still owned. + cell: CellId, + }, +} + +impl CellInventoryEntry { + /// Returns the stable key used for bounded sorted pagination. + #[must_use] + pub fn cell(&self) -> CellId { + match self { + Self::Owned(owner) => owner.target.cell_id(), + Self::Transitioning { cell } => *cell, + } + } +} + +/// A bounded page holding its native-memory reservation until dropped. +pub struct CellInventoryPage { + session: SessionId, + topology: Digest, + observed_at_ms: i64, + owned_cells: usize, + transitioning_cells: usize, + entries: Vec, + next: Option, + _retained: ResourceReservation, +} + +impl CellInventoryPage { + /// Returns the exact runtime boot identity shared by its owner observations. + #[must_use] + pub const fn session(&self) -> SessionId { + self.session + } + /// Returns the fingerprint shared by every page of a stable topology scan. + #[must_use] + pub const fn topology(&self) -> Digest { + self.topology + } + /// Returns the time this actor page was captured, without refreshing old rows. + #[must_use] + pub const fn observed_at_ms(&self) -> i64 { + self.observed_at_ms + } + /// Counts all active actors, including busy and draining ones. + #[must_use] + pub const fn owned_cells(&self) -> usize { + self.owned_cells + } + /// Counts all outstanding activation/close/release tasks. + #[must_use] + pub const fn transitioning_cells(&self) -> usize { + self.transitioning_cells + } + /// Returns at most 128 entries in ascending Cell-ID order. + #[must_use] + pub fn entries(&self) -> &[CellInventoryEntry] { + &self.entries + } + /// Returns a continuation, or None only after the full topology was scanned. + #[must_use] + pub const fn next(&self) -> Option { + self.next + } +} + +pub(super) fn validate_limit(limit: usize) -> crate::Result<()> { + if limit == 0 || limit > MAX_PAGE_ENTRIES { + return Err(Error::Node("invalid Cell inventory page limit")); + } + Ok(()) +} + +pub(super) fn collect_page( + cells: &HashMap, + transitioning: &HashSet, + next_generation: u64, + session: SessionId, + cursor: Option, + limit: usize, + retained: ResourceReservation, +) -> crate::Result { + validate_limit(limit)?; + // SQL pool admission bounds total ownership to 10,000. Copy fixed IDs only, + // within the page's one-MiB reservation; clone catalog rows for this page. + let key_count = cells + .len() + .checked_add(transitioning.len()) + .filter(|count| *count <= crate::cell::worker::MAX_ACTIVE_CELLS) + .ok_or(Error::Capacity( + "Cell inventory topology exceeds bounded scan", + ))?; + let mut keys = Vec::with_capacity(key_count); + keys.extend( + cells + .keys() + .chain(transitioning.iter()) + .map(|cell| *cell.as_bytes()), + ); + keys.sort_unstable(); + keys.dedup(); + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.actor-inventory.v1\0"); + hash.update(session.as_bytes()); + hash.update(&next_generation.to_be_bytes()); + for key in &keys { + let cell = CellId::from_bytes(*key); + hash.update(key); + hash.update(&cells.get(&cell).map_or(0, |a| a.generation).to_be_bytes()); + hash.update(&[u8::from(transitioning.contains(&cell))]); + } + let topology = Digest::from_bytes(*hash.finalize().as_bytes()); + let first = match cursor { + Some(cursor) => { + if cursor.topology != topology { + return Err(Error::Node("Cell inventory topology changed; restart scan")); + } + keys.binary_search(cursor.after.as_bytes()) + .map_err(|_| Error::Node("Cell inventory cursor key is absent"))? + + 1 + } + None => 0, + }; + let end = first.saturating_add(limit).min(keys.len()); + let mut entries = Vec::with_capacity(end - first); + for key in &keys[first..end] { + let cell = CellId::from_bytes(*key); + let entry = match cells.get(&cell).filter(|_| !transitioning.contains(&cell)) { + Some(active) => CellInventoryEntry::Owned(Box::new(observe_owned(active)?)), + None => CellInventoryEntry::Transitioning { cell }, + }; + entries.push(entry); + } + let next = if end < keys.len() { + Some(CellInventoryCursor { + topology, + after: CellId::from_bytes(keys[end - 1]), + }) + } else { + None + }; + Ok(CellInventoryPage { + session, + topology, + observed_at_ms: unix_millis(), + owned_cells: cells.len(), + transitioning_cells: transitioning.len(), + entries, + next, + _retained: retained, + }) +} + +fn observe_owned(active: &ActiveCell) -> crate::Result { + let mut blockers = Vec::new(); + if active.busy() || active.draining() || !active.queue.is_empty() { + blockers.push(DrainBlocker::BusyExecution); + } + if !active.publications.is_empty() || active.unpublished_node_logs != 0 { + blockers.push(DrainBlocker::PendingPublication); + } + let sample = active + .demand + .fresh_sample(unix_millis(), active.published_sequence); + if sample.is_none() { + blockers.push(DrainBlocker::UnknownInventory); + } + let cost = sample + .map(|sample| demand::transfer_cost(active.resource_limits, sample.database_bytes)) + .transpose()?; + let maintenance_cost = (active.role != CatalogRole::Blob) + .then(|| { + demand::transfer_cost( + active.resource_limits, + active.resource_limits.max_database_bytes, + ) + }) + .transpose()?; + let work_blocker = sample.and_then(|sample| sample.transfer_work.first_blocker()); + if work_blocker.is_some() && !blockers.contains(&DrainBlocker::BusyExecution) { + blockers.push(DrainBlocker::BusyExecution); + } + let position = active.publisher.as_ref().and_then(|publisher| { + let control = publisher.control().value(); + control.root.clone().map(|root| PublishedPosition { + incarnation: control.incarnation, + epoch: control.epoch, + root, + }) + }); + Ok(OwnedCellObservation { + target: active.catalog.target()?, + generation: active.generation, + incarnation: active.incarnation, + code: active.code, + schema: active.schema, + role: active.role, + resident_since_ms: active.resident_since_ms, + last_used_ms: active.last_used_ms, + position, + cost, + maintenance_cost, + database_bytes: sample.map(|sample| sample.database_bytes), + sampled_at_ms: sample.map(|sample| sample.observed_at_ms), + stable_observations: sample.map_or(0, |_| active.demand.stable_observations()), + work_blocker, + quiescing: active.coordination.is_maintenance_quiescing(), + maintenance_work: sample.map(|sample| sample.maintenance_work), + blockers, + }) +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-runtime/src/cell/actor/inventory/tests.rs b/crates/cellule-runtime/src/cell/actor/inventory/tests.rs new file mode 100644 index 00000000..33879b03 --- /dev/null +++ b/crates/cellule-runtime/src/cell/actor/inventory/tests.rs @@ -0,0 +1,469 @@ +use super::*; +use crate::cell::worker::WorkerCellInventory; +use crate::fleet::operations::MAX_PAGE_BYTES; +use crate::primitives::maintenance::{PersistedWorkInventory, TransferWorkInventory}; + +fn retained(ledger: &ResourceLedger) -> ResourceReservation { + ledger + .try_reserve(ResourceCost::zero().with_retained_bytes(MAX_PAGE_BYTES as usize)) + .unwrap() +} + +fn ledger() -> ResourceLedger { + ResourceLedger::new(ResourceCost::zero().with_retained_bytes(2 * MAX_PAGE_BYTES as usize)) +} + +fn session() -> SessionId { + SessionId::from_bytes([7; 16]) +} + +fn sample(at_ms: i64) -> WorkerCellInventory { + WorkerCellInventory { + persisted_work: PersistedWorkInventory::default(), + transfer_work: TransferWorkInventory::default(), + maintenance_work: + crate::primitives::maintenance_readiness::MaintenanceWorkInventory::default(), + database_bytes: 4096, + commit_sequence: 1, + observed_at_ms: at_ms, + } +} + +#[tokio::test] +async fn stale_actor_probe_cannot_clear_newer_mutation_markers_or_replace_newer_demand() { + use crate::cell::catalog::{CatalogEntry, CellCatalog}; + use crate::control::Owner; + use crate::identity::{ApplicationId, IncarnationId, NamespaceId, TenantId}; + use cellule_ltx::{CellReplica, CellStorageLayout}; + use object_store::{memory::InMemory, path::Path}; + + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + ApplicationId::from_bytes([3; 16]), + NamespaceId::from_bytes([6; 16]), + b"inventory-revision", + ) + .unwrap(); + let layout = CellStorageLayout::new( + cellule_store::Store::new(Arc::new(InMemory::new())), + Path::from("root"), + [3; 16], + ); + let catalog = CellCatalog::new(layout.clone(), target.tenant()) + .provision( + CatalogEntry::new( + &target, + CatalogRole::Application, + Digest::from_bytes([5; 32]), + 1, + ) + .unwrap(), + ) + .await + .unwrap(); + let authority = CellAuthority::new(layout.clone()); + let incarnation = IncarnationId::from_bytes([2; 16]); + let control = authority + .create_initial( + &catalog, + incarnation, + Owner { + session: session(), + endpoint: "https://node.internal:8789".into(), + }, + ) + .await + .unwrap(); + let root = tempfile::tempdir().unwrap(); + let publisher = CellPublisher::new( + CellReplica::new( + layout, + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + cellule_ltx::Limits::default(), + ) + .unwrap(), + authority, + control, + root.path().to_owned(), + ); + let connection = cellule_ltx::rusqlite::Connection::open_in_memory().unwrap(); + let now = std::time::Instant::now(); + let active = ActiveCell { + generation: 1, + admission: crate::cell::actor::admission::new_cell_admission( + publisher.control().value().owner_fence(), + ), + incarnation, + code: Digest::from_bytes([5; 32]), + schema: 1, + role: CatalogRole::Application, + catalog, + interrupt: Arc::new(connection.get_interrupt_handle()), + durability_submitter: publisher.durability_submitter(), + publisher: Some(publisher), + publications: VecDeque::new(), + publication_bytes: 0, + unpublished_node_logs: 0, + queue: VecDeque::new(), + coordination: CoordinationState::serving_with_residency(true, Residency::Resident), + persisted_work: PersistedWorkInventory::unknown(), + demand: CellDemandState::default(), + resource_limits: cellule_ltx::Limits::default(), + inventory_refreshing: false, + inventory_revision: 2, + drain: None, + transfer: None, + resident_since_ms: 1, + last_used_ms: 1, + last_work_at: now, + compaction_retry_at: now, + hydration_retry_at: now, + next_due_ms: None, + published_sequence: 0, + }; + let pool = SqlWorkerPool::new(1, 1).unwrap(); + let cell = target.cell_id(); + let mut cells = HashMap::from([(cell, active)]); + let mut transitioning = HashSet::new(); + let mut tasks = JoinSet::new(); + let mut shutdown = ShutdownState::default(); + let node_lease = RuntimeNodeLease::ObjectOnly; + let unpublished = AtomicU64::new(0); + let (publications, _) = tokio::sync::broadcast::channel(1); + let mut movement = MovementBudget::with_requested_limit(2, 32, 1_000).unwrap(); + let mut permits = HashMap::new(); + // Each case completes the real actor task adapter, rather than testing only + // a sample cache. The old revision must not clear newer unknown-work state. + for (revision, result, known) in [ + ( + 1, + Ok(WorkerCellInventory { + commit_sequence: 0, + ..sample(1) + }), + false, + ), + (2, Ok(sample(101)), false), // An unpublished sequence is not valid inventory. + ( + 2, + Ok(WorkerCellInventory { + commit_sequence: 0, + ..sample(201) + }), + true, + ), + (1, Err(Error::Fenced), true), // Older failed probes cannot destroy newer rows. + (2, Err(Error::Fenced), false), + ] { + let active = cells.get_mut(&cell).unwrap(); + assert_eq!( + active.coordination.step(CoordinationInput::BeginInventory { + queue_empty: true, + publication_idle: true, + inventory_unknown: true, + refreshing: false, + lease_live: true, + }), + CoordinationDecision::Started + ); + let effect_id = active.begin_task(CoordinationEffect::Inventory); + active.inventory_refreshing = true; + crate::cell::actor::tasks::handle_task( + TaskResult::InventoryRefreshed { + cell, + generation: 1, + effect_id, + inventory_revision: revision, + result, + }, + &pool, + &mut cells, + &mut transitioning, + &mut tasks, + &mut shutdown, + &node_lease, + &unpublished, + &publications, + &mut movement, + &mut permits, + ); + let active = &cells[&cell]; + assert!(!active.inventory_refreshing); + assert_eq!(!active.persisted_work.is_unknown(), known); + assert_eq!(active.demand.fresh_sample(202, 0).is_some(), known); + assert_eq!(active.demand.stable_observations(), u8::from(known)); + } + assert!(tasks.is_empty()); + pool.shutdown().await.unwrap(); +} + +#[test] +fn demand_stability_counts_worker_measurements_not_page_reads_or_republished_time() { + let mut state = CellDemandState::default(); + state + .record(sample(1), cellule_ltx::Limits::default(), 1) + .unwrap(); + assert_eq!(state.stable_observations(), 1); + for _ in 0..10 { + assert_eq!(state.fresh_sample(100, 1), Some(sample(1))); + assert_eq!(state.stable_observations(), 1); + } + state + .record(sample(1), cellule_ltx::Limits::default(), 1) + .unwrap(); + assert_eq!(state.stable_observations(), 1); + assert!(!state.should_refresh(100, false)); + assert!(state.should_refresh(101, false)); + state + .record(sample(101), cellule_ltx::Limits::default(), 1) + .unwrap(); + assert_eq!(state.stable_observations(), 2); + assert!(!state.should_refresh(15_100, false)); + assert!(state.should_refresh(15_101, false)); + assert!(state.fresh_sample(100, 1).is_none()); + assert!(state.fresh_sample(30_101, 1).is_none()); + assert!(state.fresh_sample(102, 2).is_none()); +} + +#[test] +fn changed_mutated_unknown_or_regressing_demand_cannot_reuse_a_stable_sample() { + let mut state = CellDemandState::default(); + state + .record(sample(1), cellule_ltx::Limits::default(), 1) + .unwrap(); + state + .record(sample(101), cellule_ltx::Limits::default(), 1) + .unwrap(); + let changed = WorkerCellInventory { + database_bytes: 8192, + ..sample(201) + }; + state + .record(changed, cellule_ltx::Limits::default(), 1) + .unwrap(); + assert_eq!(state.stable_observations(), 1); + state.clear(); + assert!(state.fresh_sample(202, 1).is_none()); + assert_eq!(state.stable_observations(), 0); + assert!( + state + .record(sample(100), cellule_ltx::Limits::default(), 1) + .is_err() + ); + assert!(state.fresh_sample(203, 1).is_none()); + assert!(matches!( + state.record(sample(301), cellule_ltx::Limits::default(), 2), + Err(Error::PendingPublication) + )); + let unknown = WorkerCellInventory { + persisted_work: PersistedWorkInventory::unknown(), + ..sample(401) + }; + assert!( + state + .record(unknown, cellule_ltx::Limits::default(), 1) + .is_err() + ); + assert_eq!(state.stable_observations(), 0); + assert!(!state.should_refresh(402, false)); + assert!(state.should_refresh(1401, false)); +} + +#[test] +fn conservative_cost_covers_database_growth_and_rejects_unknown_or_overflowing_bounds() { + let limits = cellule_ltx::Limits::default(); + let cost = demand::transfer_cost(limits, 4096).unwrap(); + assert_eq!( + cost, + demand::transfer_cost(limits, limits.max_database_bytes).unwrap() + ); + assert_eq!( + cost.disk_bytes, + 2 * limits.max_database_bytes + limits.max_plan_bytes + (64 << 20) + ); + assert!(cost.memory_bytes >= super::super::ACTIVE_CELL_NATIVE_BYTES); + assert!(cost.file_descriptors >= crate::cell::worker::ACTIVE_CELL_FILE_DESCRIPTORS as u32); + assert_eq!(cost.job_credits, 1); + assert!(demand::transfer_cost(limits, 0).is_err()); + assert!(demand::transfer_cost(limits, limits.max_database_bytes + 1).is_err()); + assert!( + demand::transfer_cost( + cellule_ltx::Limits { + max_database_bytes: u64::MAX, + ..limits + }, + 4096 + ) + .is_err() + ); + assert!( + demand::transfer_cost( + cellule_ltx::Limits { + max_plan_bytes: 0, + ..limits + }, + 4096 + ) + .is_err() + ); +} + +#[test] +fn pages_include_every_transition_once_in_sorted_order_and_own_memory_admission() { + let cells = HashMap::new(); + let transitioning = (1..=255_u8) + .rev() + .map(|id| CellId::from_bytes([id; 32])) + .collect(); + let ledger = ledger(); + let first = collect_page( + &cells, + &transitioning, + 255, + session(), + None, + 128, + retained(&ledger), + ) + .unwrap(); + assert_eq!(first.owned_cells(), 0); + assert_eq!(first.transitioning_cells(), 255); + assert_eq!(first.entries().len(), 128); + assert_eq!(first.entries()[0].cell(), CellId::from_bytes([1; 32])); + let cursor = first.next().unwrap(); + assert_eq!( + CellInventoryCursor::from_bytes(&cursor.to_bytes()).unwrap(), + cursor + ); + let second = collect_page( + &cells, + &transitioning, + 255, + session(), + Some(cursor), + 128, + retained(&ledger), + ) + .unwrap(); + assert_eq!(second.entries().len(), 127); + assert_eq!(second.entries()[0].cell(), CellId::from_bytes([129; 32])); + assert_eq!(second.entries()[126].cell(), CellId::from_bytes([255; 32])); + assert_eq!(second.topology(), first.topology()); + assert!(second.next().is_none()); + assert_eq!( + ledger.snapshot().unwrap().used.retained_bytes(), + 2 * MAX_PAGE_BYTES as usize + ); + drop(first); + assert_eq!( + ledger.snapshot().unwrap().used.retained_bytes(), + MAX_PAGE_BYTES as usize + ); + drop(second); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); +} + +#[test] +fn cursor_refuses_membership_generation_and_runtime_replacement_without_leaking_bytes() { + let cells = HashMap::new(); + let mut transitioning = + HashSet::from([CellId::from_bytes([1; 32]), CellId::from_bytes([2; 32])]); + let ledger = ledger(); + let first = collect_page( + &cells, + &transitioning, + 2, + session(), + None, + 1, + retained(&ledger), + ) + .unwrap(); + let cursor = first.next().unwrap(); + drop(first); + for (generation, session) in [(3, session()), (2, SessionId::from_bytes([8; 16]))] { + assert!( + collect_page( + &cells, + &transitioning, + generation, + session, + Some(cursor), + 1, + retained(&ledger) + ) + .is_err() + ); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); + } + transitioning.remove(&CellId::from_bytes([2; 32])); + assert!( + collect_page( + &cells, + &transitioning, + 2, + session(), + Some(cursor), + 1, + retained(&ledger) + ) + .is_err() + ); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); +} + +#[test] +fn invalid_limits_and_cursor_keys_fail_closed_empty_inventory_finishes() { + let ledger = ledger(); + let cells = HashMap::new(); + let transitioning = HashSet::new(); + for limit in [0, MAX_PAGE_ENTRIES + 1, usize::MAX] { + assert!( + collect_page( + &cells, + &transitioning, + 0, + session(), + None, + limit, + retained(&ledger) + ) + .is_err() + ); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); + } + let page = collect_page( + &cells, + &transitioning, + 0, + session(), + None, + 1, + retained(&ledger), + ) + .unwrap(); + assert!(page.entries().is_empty()); + assert!(page.next().is_none()); + let cursor = CellInventoryCursor { + topology: page.topology(), + after: CellId::from_bytes([1; 32]), + }; + drop(page); + assert!( + collect_page( + &cells, + &transitioning, + 0, + session(), + Some(cursor), + 1, + retained(&ledger) + ) + .is_err() + ); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); + for width in [0, 63, 65, 1024] { + assert!(CellInventoryCursor::from_bytes(&vec![0; width]).is_err()); + } +} diff --git a/crates/cellule-runtime/src/cell/actor/lifecycle/activation.rs b/crates/cellule-runtime/src/cell/actor/lifecycle/activation.rs index a7a8943b..ad1b788d 100644 --- a/crates/cellule-runtime/src/cell/actor/lifecycle/activation.rs +++ b/crates/cellule-runtime/src/cell/actor/lifecycle/activation.rs @@ -15,6 +15,7 @@ pub(in crate::cell::actor) async fn activate_restored_and_publish( schema, root, reservation, + job, } = activation; pool.activate_restored( cell, @@ -24,6 +25,7 @@ pub(in crate::cell::actor) async fn activate_restored_and_publish( schema, root, reservation, + job, ) .await?; if let Err(error) = publisher.activate().await { diff --git a/crates/cellule-runtime/src/cell/actor/lifecycle/background.rs b/crates/cellule-runtime/src/cell/actor/lifecycle/background.rs index 1d4703c2..4e9e29de 100644 --- a/crates/cellule-runtime/src/cell/actor/lifecycle/background.rs +++ b/crates/cellule-runtime/src/cell/actor/lifecycle/background.rs @@ -74,42 +74,83 @@ pub(in crate::cell::actor) fn start_background_inventory( tasks: &mut JoinSet, node_lease: &RuntimeNodeLease, ) { - let candidates = cells - .iter_mut() - .filter_map(|(cell, active)| { - let decision = active.coordination.step(CoordinationInput::BeginInventory { - queue_empty: active.queue.is_empty(), - publication_idle: active.coordination.publication_count() == 0, - inventory_unknown: active.persisted_work.is_unknown(), - refreshing: active.inventory_refreshing, - lease_live: node_lease.check().is_ok(), - }); - if matches!(decision, CoordinationDecision::Fence) { - fence_active(active); - return None; - } - if !matches!(decision, CoordinationDecision::Started) { - return None; - } - let effect_id = active.begin_task(CoordinationEffect::Inventory); - active.inventory_refreshing = true; - Some((*cell, active.generation, active.role, effect_id)) + if node_lease.check().is_err() { + for active in cells.values_mut() { + fence_active(active); + } + return; + } + const MAX_INVENTORY_IN_FLIGHT: usize = 32; + let in_flight = cells + .values() + .filter(|active| active.inventory_refreshing) + .count(); + let slots = MAX_INVENTORY_IN_FLIGHT.saturating_sub(in_flight); + if slots == 0 { + return; + } + let now_ms = unix_millis(); + let mut candidates = cells + .iter() + .filter(|(_, active)| { + !active.busy() + && !active.draining() + && !active.inventory_refreshing + && active.queue.is_empty() + && active.coordination.publication_count() == 0 + && active + .demand + .should_refresh(now_ms, active.persisted_work.is_unknown()) }) + .map(|(cell, active)| (active.demand.refresh_priority(), *cell.as_bytes())) .collect::>(); - - for (cell, generation, role, effect_id) in candidates { + // Oldest attempted samples get the next slots; a hot or failing Cell cannot + // monopolize refresh. Shared SQL-job admission still bounds actual work. + candidates.sort_unstable(); + candidates.truncate(slots); + for (_, cell_bytes) in candidates { + let cell = CellId::from_bytes(cell_bytes); + let Some(active) = cells.get_mut(&cell) else { + continue; + }; + let decision = active.coordination.step(CoordinationInput::BeginInventory { + queue_empty: active.queue.is_empty(), + publication_idle: active.coordination.publication_count() == 0, + inventory_unknown: true, + refreshing: active.inventory_refreshing, + lease_live: node_lease.check().is_ok(), + }); + if matches!(decision, CoordinationDecision::Fence) { + fence_active(active); + continue; + } + if !matches!(decision, CoordinationDecision::Started) { + continue; + } + let effect_id = active.begin_task(CoordinationEffect::Inventory); + active.inventory_refreshing = true; + let generation = active.generation; + let role = active.role; + let inventory_revision = active.inventory_revision; let pool = pool.clone(); tasks.spawn(async move { let deadline = std::time::Instant::now() + SQL_WALL_DEADLINE; - let result = - tokio::time::timeout_at(deadline.into(), pool.persisted_work_inventory(cell, role)) - .await - .map_err(|_| Error::Deadline) - .and_then(|result| result); + let sql_deadline = SqlDeadline::new(deadline); + let result = tokio::time::timeout_at( + deadline.into(), + pool.fleet_inventory(cell, role, now_ms, sql_deadline.clone()), + ) + .await + .map_err(|_| { + sql_deadline.cancel_queued(); + Error::Deadline + }) + .and_then(|result| result); TaskResult::InventoryRefreshed { cell, generation, effect_id, + inventory_revision, result, } }); diff --git a/crates/cellule-runtime/src/cell/actor/lifecycle/eviction.rs b/crates/cellule-runtime/src/cell/actor/lifecycle/eviction.rs index 003ccb53..41aeb16a 100644 --- a/crates/cellule-runtime/src/cell/actor/lifecycle/eviction.rs +++ b/crates/cellule-runtime/src/cell/actor/lifecycle/eviction.rs @@ -120,7 +120,7 @@ pub(in crate::cell::actor) fn begin_idle_cell_eviction( active.admission.draining.store(true, Ordering::Release); active.admission.requests.close(); active.admission.bytes.close(); - active.drain = reply; + active.drain = reply.map(DrainReply::Unit); if matches!( schedule(active, true), CoordinationDecision::ReadyToDeactivate diff --git a/crates/cellule-runtime/src/cell/actor/lifecycle/scheduling.rs b/crates/cellule-runtime/src/cell/actor/lifecycle/scheduling.rs index f268f550..299f0857 100644 --- a/crates/cellule-runtime/src/cell/actor/lifecycle/scheduling.rs +++ b/crates/cellule-runtime/src/cell/actor/lifecycle/scheduling.rs @@ -40,13 +40,20 @@ pub(in crate::cell::actor) fn start_next( // Durable command outcomes, effects, Queue rows, and Workflow runs // remain release obligations until a fresh inventory proves otherwise. active.persisted_work = crate::primitives::maintenance::PersistedWorkInventory::unknown(); + active.demand.clear(); + let Some(revision) = active.inventory_revision.checked_add(1) else { + active.queue.push_front(work); + fence_active(active); + return; + }; + active.inventory_revision = revision; } let generation = active.generation; active.last_used_ms = unix_millis(); active.last_work_at = std::time::Instant::now(); let kind = match &work { - QueuedWork::Command(_) => AdmissionKind::Command, - QueuedWork::Query(_) => AdmissionKind::Query, + QueuedWork::Command(command) => command._work.kind, + QueuedWork::Query(query) => query._work.kind, QueuedWork::Resolve(_) => AdmissionKind::Resolve, QueuedWork::Migration(_) => AdmissionKind::Migration, }; @@ -114,6 +121,14 @@ pub(in crate::cell::actor) fn start_transfer_inspection( let Some(active) = cells.get_mut(&cell) else { return; }; + if active + .transfer + .as_ref() + .is_some_and(|transfer| transfer.maintenance.is_some()) + { + maintenance::inspect(cell, pool, cells, tasks, node_lease); + return; + } if active.transfer.is_none() || active.inventory_refreshing || !active.queue.is_empty() @@ -270,6 +285,7 @@ pub(in crate::cell::actor) fn start_deactivate( let generation = active.generation; let pool = pool.clone(); tasks.spawn(async move { + let mut released = None; let result = async { let mut publisher = active.publisher.ok_or(Error::Fenced)?; // A release that already published its root may leave a resume @@ -281,6 +297,17 @@ pub(in crate::cell::actor) fn start_deactivate( None => pool.deactivate(cell).await?, } publisher.release().await?; + let final_control = publisher.control().value(); + if final_control.state != crate::control::ControlState::Idle + || final_control.owner.is_some() + { + return Err(Error::Fenced); + } + released = Some(crate::fleet::operations::PublishedPosition { + incarnation: final_control.incarnation, + epoch: final_control.epoch, + root: final_control.root.clone().ok_or(Error::Fenced)?, + }); // A released Cell that still has a deadline publishes one bounded // hint key, so the scheduler finds it without scanning every shard. // The hint is an accelerator: a failed write costs a later Tick @@ -305,6 +332,7 @@ pub(in crate::cell::actor) fn start_deactivate( reply: active.drain, shutdown_drain: active.coordination.is_shutdown(), result, + released, } }); } @@ -341,6 +369,7 @@ pub(in crate::cell::actor) fn start_fenced_deactivate( reply: active.drain, shutdown_drain: active.coordination.is_shutdown(), result, + released: None, } }); } @@ -371,6 +400,7 @@ pub(in crate::cell::actor) fn start_orphan_deactivate( reply: None, shutdown_drain, result, + released: None, } }); transitioning.insert(cell); diff --git a/crates/cellule-runtime/src/cell/actor/maintenance/inspection.rs b/crates/cellule-runtime/src/cell/actor/maintenance/inspection.rs new file mode 100644 index 00000000..963b8134 --- /dev/null +++ b/crates/cellule-runtime/src/cell/actor/maintenance/inspection.rs @@ -0,0 +1,74 @@ +use super::*; + +pub(in crate::cell::actor) fn inspect( + cell: CellId, + pool: &SqlWorkerPool, + cells: &mut HashMap, + tasks: &mut JoinSet, + node_lease: &RuntimeNodeLease, +) { + let Some(active) = cells.get_mut(&cell) else { + return; + }; + let Some(state) = active + .transfer + .as_ref() + .and_then(|transfer| transfer.maintenance.as_ref()) + else { + return; + }; + let now = std::time::Instant::now(); + if now < state.next_check || now >= state.deadline { + return; + } + let deadline = state.deadline.min(now + SQL_WALL_DEADLINE); + let decision = active + .coordination + .step(CoordinationInput::BeginMaintenanceInventory { + refreshing: active.inventory_refreshing, + finalizing: state.closing, + queue_empty: active.queue.is_empty(), + publisher_ready: active.publisher.is_some(), + lease_live: node_lease.check().is_ok(), + }); + if decision == CoordinationDecision::Fence { + super::super::admission::fence_active(active); + return; + } + if decision != CoordinationDecision::Started { + return; + } + let effect_id = active.begin_task(CoordinationEffect::Inventory); + active.inventory_refreshing = true; + if let Some(state) = active + .transfer + .as_mut() + .and_then(|transfer| transfer.maintenance.as_mut()) + { + state.inventory_effect = Some(effect_id); + } + let generation = active.generation; + let role = active.role; + let pool = pool.clone(); + tasks.spawn(async move { + let sql_deadline = SqlDeadline::new(deadline); + let operation = pool.fleet_inventory(cell, role, unix_millis(), sql_deadline.clone()); + tokio::pin!(operation); + let result = match tokio::time::timeout_at(deadline.into(), &mut operation).await { + Ok(result) => result.map(|inventory| inventory.maintenance_work), + Err(_) => { + // A started read remains owned and joined. Queued reads may be + // cancelled through the ordinary worker deadline capability. + sql_deadline.cancel_queued(); + let _ = operation.await; + Err(Error::Deadline) + } + }; + TaskResult::MaintenancePreflight { + cell, + generation, + effect_id, + result, + } + }); +} diff --git a/crates/cellule-runtime/src/cell/actor/maintenance/mod.rs b/crates/cellule-runtime/src/cell/actor/maintenance/mod.rs new file mode 100644 index 00000000..108abfc4 --- /dev/null +++ b/crates/cellule-runtime/src/cell/actor/maintenance/mod.rs @@ -0,0 +1,64 @@ +//! Busy maintenance uses the existing worker, coordination and release effects. + +use super::*; +use crate::fleet::operations::{DrainBlocker, PublishedPosition}; + +mod inspection; +mod release; +pub(super) use inspection::inspect; +pub(super) use release::{begin, drive}; + +/// Result of a source-owned planned maintenance release. +#[derive(Debug)] +pub enum MaintenanceCellRelease { + /// Canonical close and authority release produced this exact final root. + Released(PublishedPosition), + /// This request started no canonical release. Foreground closure, if + /// installed, stays sticky; native completion remains available. + Refused { + /// The original preflight condition; it grants no release authority. + blocker: DrainBlocker, + /// Original inventory/admission failure, when one occurred. + error: Option, + }, +} + +pub(super) struct ReleaseRequest { + pub(super) cell: CellId, + pub(super) generation: u64, + pub(super) incarnation: crate::identity::IncarnationId, + pub(super) epoch: u64, + pub(super) deadline: std::time::Instant, + pub(super) reply: oneshot::Sender>, +} + +pub(super) struct ReleaseState { + pub(super) deadline: std::time::Instant, + pub(super) next_check: std::time::Instant, + pub(super) closing: bool, + pub(super) inventory_effect: Option, +} + +pub(super) fn refuse(reply: DrainReply, blocker: DrainBlocker, error: Option) { + match reply { + DrainReply::Maintenance(reply) => { + let _ = reply.send(Ok(MaintenanceCellRelease::Refused { blocker, error })); + } + reply => { + let _ = reply.send(Err(match error { + Some(error) => error, + None => Error::CellDraining, + })); + } + } +} + +pub(super) fn reopen_completion(active: &mut ActiveCell) { + active.coordination.step(CoordinationInput::AbortTransfer); + if !active.coordination.is_fenced() && !active.draining() { + // Semaphores remain open until ConfirmTransfer. No authority mutation + // has started here; this restores only exact native completion because + // the sticky foreground flag and kernel state remain installed. + active.admission.draining.store(false, Ordering::Release); + } +} diff --git a/crates/cellule-runtime/src/cell/actor/maintenance/release.rs b/crates/cellule-runtime/src/cell/actor/maintenance/release.rs new file mode 100644 index 00000000..e0760bdc --- /dev/null +++ b/crates/cellule-runtime/src/cell/actor/maintenance/release.rs @@ -0,0 +1,132 @@ +use super::*; + +pub(in crate::cell::actor) fn begin( + request: ReleaseRequest, + cells: &mut HashMap, + movement: &mut MovementBudget, + permits: &mut HashMap, +) -> Option { + let ReleaseRequest { + cell, + generation, + incarnation, + epoch, + deadline, + reply, + } = request; + let reply = DrainReply::Maintenance(reply); + let Some(active) = cells.get_mut(&cell) else { + let _ = reply.send(Err(Error::CellNotActive)); + return None; + }; + if active.generation != generation || active.incarnation != incarnation { + let _ = reply.send(Err(Error::Fenced)); + return None; + } + let Some(publisher) = active.publisher.as_ref() else { + refuse(reply, DrainBlocker::BusyExecution, None); + return None; + }; + if publisher.control().value().epoch != epoch { + let _ = reply.send(Err(Error::Fenced)); + return None; + } + if active.draining() || active.transfer.is_some() || active.drain.is_some() { + refuse(reply, DrainBlocker::BusyExecution, None); + return None; + } + if active.role == CatalogRole::Blob { + // Foreground closure cannot inventory external stream/upload/pin owners. + // Leave their completion path intact until that barrier is implemented. + refuse(reply, DrainBlocker::UnknownInventory, None); + return None; + } + let now = std::time::Instant::now(); + if now >= deadline { + refuse(reply, DrainBlocker::Deadline, None); + return None; + } + let mut permit = match movement.try_start_requested(unix_millis()) { + Ok(permit) => permit, + Err(error) => { + refuse(reply, DrainBlocker::MovementBudget, Some(error)); + return None; + } + }; + match active + .coordination + .step(CoordinationInput::BeginMaintenanceQuiescence) + { + CoordinationDecision::Started => {} + CoordinationDecision::Reject(reason) => { + movement.complete(&mut permit); + let _ = reply.send(Err(super::super::admission::rejection_error(reason))); + return None; + } + _ => { + movement.complete(&mut permit); + refuse(reply, DrainBlocker::BusyExecution, None); + return None; + } + } + active + .admission + .maintenance_quiescing + .store(true, Ordering::Release); + active.transfer = Some(TransferPreflight { + reply, + maintenance: Some(ReleaseState { + deadline, + next_check: now, + closing: false, + inventory_effect: None, + }), + }); + permits.insert(cell, permit); + Some(cell) +} + +pub(in crate::cell::actor) fn drive( + pool: &SqlWorkerPool, + cells: &mut HashMap, + transitioning: &mut HashSet, + tasks: &mut JoinSet, + node_lease: &RuntimeNodeLease, + movement: &mut MovementBudget, + permits: &mut HashMap, +) { + // Requested movement permits bound this scan. Do not scan the whole resident + // fleet on every ingress message or add a second timer/rate limiter. + let pending = permits + .keys() + .copied() + .filter(|cell| { + cells + .get(cell) + .and_then(|active| active.transfer.as_ref()) + .is_some_and(|transfer| transfer.maintenance.is_some()) + }) + .collect::>(); + for cell in pending { + let Some(active) = cells.get_mut(&cell) else { + continue; + }; + let expired = active + .transfer + .as_ref() + .and_then(|transfer| transfer.maintenance.as_ref()) + .is_some_and(|state| std::time::Instant::now() >= state.deadline); + if expired { + if let Some(transfer) = active.transfer.take() { + reopen_completion(active); + refuse(transfer.reply, DrainBlocker::Deadline, None); + } + if let Some(mut permit) = permits.remove(&cell) { + movement.complete(&mut permit); + } + continue_cell(cell, pool, cells, transitioning, tasks, node_lease); + } else { + inspect(cell, pool, cells, tasks, node_lease); + } + } +} diff --git a/crates/cellule-runtime/src/cell/actor/mod.rs b/crates/cellule-runtime/src/cell/actor/mod.rs index 5d31bc27..ce5e409c 100644 --- a/crates/cellule-runtime/src/cell/actor/mod.rs +++ b/crates/cellule-runtime/src/cell/actor/mod.rs @@ -14,14 +14,25 @@ use tokio::{ }; mod handle; +mod inventory; +pub use inventory::{ + CellInventoryCursor, CellInventoryEntry, CellInventoryPage, OwnedCellObservation, +}; use state::*; use task::*; mod acquire; +mod acquisition_observer; +mod prefix; +pub use acquisition_observer::{AcquisitionObservation, AcquisitionObserver}; mod admission; mod lifecycle; +mod maintenance; +pub use maintenance::MaintenanceCellRelease; +mod receiver; mod requests; pub(crate) mod routes; mod runtime; +pub use receiver::{PreparedCellReceiver, ReceiverState}; mod state; mod task; mod tasks; @@ -29,6 +40,7 @@ mod tasks; use lifecycle::*; use requests::*; +pub(crate) use handle::CommandWork; use handle::{CellAdmission, WorkAdmission}; pub use handle::{CellHandle, DueResident}; @@ -46,7 +58,9 @@ use crate::coordination::{ AdmissionKind, CoordinationDecision, CoordinationEffect, CoordinationInput, CoordinationState, RejectReason, Residency, }; +use crate::fleet::admission::NodeAdmission; use crate::fleet::eviction::{EvictionObservation, EvictionState, select_victims}; +use crate::fleet::operations::DrainBlocker; use crate::fleet::pressure::{ MovementBudget, MovementPermit, PressureClassifier, PressureSample, PressureState, }; diff --git a/crates/cellule-runtime/src/cell/actor/prefix.rs b/crates/cellule-runtime/src/cell/actor/prefix.rs new file mode 100644 index 00000000..04a90b04 --- /dev/null +++ b/crates/cellule-runtime/src/cell/actor/prefix.rs @@ -0,0 +1,82 @@ +//! Prefix verification through the runtime's configured origin I/O facilities. +use super::*; +use crate::control::authority::MAX_LINEAGE_ROOTS; + +// Conservative transient metadata envelope, independent of database body size: +// 64 descriptor pages (at most 96 decoded descriptors each), bounded concurrent +// 64 KiB root fetch/decode buffers, extents and body/index maps fit the fixed +// 16 MiB portion. Each of at most 10,000 inventory objects and `limit + 1` lineage +// roots receives 1 KiB for maps, vector growth and allocator overhead. Root bodies +// stream through the existing shared I/O host; application Store adapters own +// their stream chunk bounds. The permit spans all awaited verification work. +const ORIGIN_METADATA_BYTES: usize = 16 << 20; +const PREFIX_ENTRY_BYTES: usize = 1024; + +impl CellRuntime { + /// Verifies one exact root prefix through this runtime's shared LTX host. + /// + /// Transient lineage/origin metadata is reserved before I/O through the + /// shared retained-byte ledger, and released on completion or cancellation. + /// The caller owns the bounded future and authenticates canonical backend + /// mappings. This grants no authority, serving, root pin or fleet settlement. + /// Shutdown/admission and exact selected authority still require fresh checks. + pub async fn verify_root_prefix( + &self, + catalog: &CatalogProof, + authority: &CellAuthority, + replica: cellule_ltx::CellReplica, + prefix: cellule_ltx::RootRef, + root: cellule_ltx::RootRef, + limit: usize, + ) -> crate::Result { + let (_metadata, replica) = self.prefix_replica(catalog, replica, root, limit)?; + let proof = authority + .verify_root_prefix(prefix, root, &replica, limit) + .await?; + self.ensure_running()?; + Ok(proof) + } + + /// Verifies the exact original sealed recovery row through canonical retained + /// acquisition input, native materialization lineage and complete origin bytes. + /// + /// Uses the same shared admission and caller-owned deadline as root-prefix + /// verification. The caller authenticates original manifest/log/boot scope; + /// this grants no current serving, retention pin or aggregate settlement. + pub async fn verify_recovered_prefix( + &self, + catalog: &CatalogProof, + authority: &CellAuthority, + replica: cellule_ltx::CellReplica, + required: &crate::recovery::manifest::PinnedRecoveryCell, + root: cellule_ltx::RootRef, + limit: usize, + ) -> crate::Result { + let (_metadata, replica) = self.prefix_replica(catalog, replica, root, limit)?; + let proof = authority + .verify_recovered_prefix(required, root, &replica, limit) + .await?; + self.ensure_running()?; + Ok(proof) + } + + fn prefix_replica( + &self, + catalog: &CatalogProof, + replica: cellule_ltx::CellReplica, + root: cellule_ltx::RootRef, + limit: usize, + ) -> crate::Result<(NodeByteReservation, cellule_ltx::CellReplica)> { + self.ensure_running()?; + self.check_application_limits(catalog, replica.limits())?; + if catalog.entry().cell().as_bytes() != &root.cell { + return Err(Error::Control("Cell prefix catalog and root differ")); + } + if limit == 0 || limit > MAX_LINEAGE_ROOTS { + return Err(Error::Capacity("invalid Cell root lineage traversal bound")); + } + let bytes = ORIGIN_METADATA_BYTES + (MAX_LINEAGE_ROOTS + limit + 1) * PREFIX_ENTRY_BYTES; + let metadata = self.try_reserve_node_bytes(bytes)?; + Ok((metadata, self.replica_for_read(replica))) + } +} diff --git a/crates/cellule-runtime/src/cell/actor/receiver.rs b/crates/cellule-runtime/src/cell/actor/receiver.rs new file mode 100644 index 00000000..6c85de81 --- /dev/null +++ b/crates/cellule-runtime/src/cell/actor/receiver.rs @@ -0,0 +1,464 @@ +//! Runtime-owned receiver credit and cancellation-independent acquisition. + +use std::sync::{Mutex, Weak}; + +use super::*; +use crate::cell::worker::WorkerJobReservation; +use crate::fleet::operations::{ + AttemptId, MAX_ACTIVE_ATTEMPTS, MAX_RECORD_BYTES, MAX_RESTORE_BYTES, MoveAttemptSpec, + OperationError, ReceiverReservation, +}; + +/// Local receiver lifecycle hint. Authority and actor observations still prove serving. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum ReceiverState { + /// The runtime holds unused resource credit for the exact attempt. + Prepared, + /// Acquisition was accepted and owns its work independently of its waiter. + Activating, + /// Acquisition returned a live handle; recheck authority and actor readiness. + Activated, + /// Accepted work finished unsuccessfully; reconcile authority before retrying. + Failed, + /// Unused credit was synchronously returned without an authority transition. + Cancelled, +} + +/// Opaque reference to runtime-owned preparation. Dropping it does not cancel work. +/// +/// The runtime retains the reservation after a lost preparation reply. Shutdown +/// cancels unused credit and joins accepted activation even when callers retain +/// this reference. This local API supplies no operator or peer authorization. +#[derive(Clone)] +pub struct PreparedCellReceiver { + credit: Weak, +} + +impl PreparedCellReceiver { + /// Returns the immutable movement identity while the runtime retains it. + pub fn spec(&self) -> crate::Result { + Ok(self.credit()?.spec.clone()) + } + + /// Returns the exact session and expiry of the retained credit. + pub fn reservation(&self) -> crate::Result { + Ok(self.credit()?.reservation) + } + + /// Inspects a local lifecycle hint, never a durable movement result. + pub fn state(&self) -> crate::Result { + Ok(self + .credit()? + .state + .lock() + .map_err(|_| Error::RuntimeClosed)? + .phase) + } + + fn credit(&self) -> crate::Result> { + self.credit.upgrade().ok_or(Error::RuntimeClosed) + } +} + +#[derive(Default)] +pub(super) struct ReceiverRegistry { + credits: HashMap>, + tasks: Vec<(AttemptId, tokio::task::JoinHandle<()>)>, +} + +struct ReceiverCredit { + spec: MoveAttemptSpec, + reservation: ReceiverReservation, + catalog: CatalogProof, + replica: cellule_ltx::CellReplica, + destination: PathBuf, + state: Mutex, + // Bounded receipt state remains charged until retirement or shutdown. + _retained: ResourceReservation, +} + +struct CreditState { + phase: ReceiverState, + resources: Option, +} + +struct ReceiverResources { + cell: CellReservation, + job: WorkerJobReservation, + disk: cellule_ltx::DiskBudget, +} + +fn operation(error: OperationError) -> Error { + Error::FleetOperation(Box::new(error)) +} + +impl CellRuntime { + /// Reserves real Cell, memory, descriptor, disk, and affine SQL-job capacity. + /// + /// Call before releasing the source. Duplicate exact preparation returns the + /// same credit without another charge. At most two local receipts are kept; + /// the application retires joined terminal receipts after durable journaling. + /// Expiry permits cancellation; it never silently returns accepted credit. + pub fn prepare_receiver( + &self, + spec: MoveAttemptSpec, + catalog: CatalogProof, + replica: cellule_ltx::CellReplica, + destination: PathBuf, + expires_at_ms: i64, + now_ms: i64, + ) -> crate::Result { + spec.validate().map_err(operation)?; + if spec.destination != self.inner.session + || catalog.target()? != spec.target + || replica.scope() + != ( + *spec.target.cell_id().as_bytes(), + *spec.incarnation.as_bytes(), + ) + { + return Err(Error::Fenced); + } + self.check_application_limits(&catalog, replica.limits())?; + if destination.parent().is_none() { + return Err(Error::Control("Cell activation destination has no parent")); + } + let minimum = inventory::transfer_cost(replica.limits(), 1)?; + if spec.cost.memory_bytes < minimum.memory_bytes + || spec.cost.disk_bytes < minimum.disk_bytes + || spec.cost.file_descriptors < minimum.file_descriptors + || spec.cost.job_credits != 1 + || spec.cost.disk_bytes > MAX_RESTORE_BYTES + { + return Err(Error::Capacity( + "receiver demand does not cover validated Cell bounds", + )); + } + let mut registry = self + .inner + .receivers + .lock() + .map_err(|_| Error::RuntimeClosed)?; + self.ensure_acquiring()?; + if let Some(old) = registry.credits.get(&spec.id) { + if old.spec != spec + || old.reservation.expires_at_ms != expires_at_ms + || old.catalog.entry() != catalog.entry() + || old.catalog.revision() != catalog.revision() + || limits_identity(old.replica.limits()) != limits_identity(replica.limits()) + || old.destination != destination + { + return Err(operation(OperationError::Conflict)); + } + return Ok(PreparedCellReceiver { + credit: Arc::downgrade(old), + }); + } + if now_ms < 0 || now_ms >= spec.deadline_ms || expires_at_ms <= now_ms { + return Err(operation(OperationError::Deadline)); + } + if registry.credits.len() >= MAX_ACTIVE_ATTEMPTS { + return Err(operation(OperationError::Budget)); + } + let memory = usize::try_from(spec.cost.memory_bytes) + .map_err(|_| Error::Capacity("receiver memory cost overflow"))?; + let descriptors = usize::try_from(spec.cost.file_descriptors) + .map_err(|_| Error::Capacity("receiver descriptor cost overflow"))?; + let retained = self + .inner + .resources + .try_reserve(ResourceCost::zero().with_retained_bytes(MAX_RECORD_BYTES as usize))?; + let cell = self.inner.pool.reserve_activation_cost( + ResourceCost::active_cell() + .with_resident_bytes(memory) + .with_file_descriptors(descriptors), + )?; + let job = self.inner.pool.try_reserve_job(spec.target.cell_id())?; + let disk = self + .local_disk_budget() + .try_reserve(spec.cost.disk_bytes)? + .into_budget(); + let credit = Arc::new(ReceiverCredit { + reservation: ReceiverReservation { + session: spec.destination, + expires_at_ms, + }, + spec, + catalog, + replica, + destination, + state: Mutex::new(CreditState { + phase: ReceiverState::Prepared, + resources: Some(ReceiverResources { cell, job, disk }), + }), + _retained: retained, + }); + let handle = PreparedCellReceiver { + credit: Arc::downgrade(&credit), + }; + registry.credits.insert(credit.spec.id, credit); + Ok(handle) + } + + /// Recovers a lost preparation reply for this runtime's exact attempt. + pub fn prepared_receiver(&self, id: AttemptId) -> crate::Result> { + let registry = self + .inner + .receivers + .lock() + .map_err(|_| Error::RuntimeClosed)?; + self.ensure_running()?; + Ok(registry + .credits + .get(&id) + .map(|credit| PreparedCellReceiver { + credit: Arc::downgrade(credit), + })) + } + + /// Cancels only unused preparation. Accepted activation must be reconciled. + pub fn cancel_prepared_receiver(&self, prepared: &PreparedCellReceiver) -> crate::Result<()> { + let registry = self + .inner + .receivers + .lock() + .map_err(|_| Error::RuntimeClosed)?; + let credit = checked_credit(®istry, prepared)?; + let mut state = credit.state.lock().map_err(|_| Error::RuntimeClosed)?; + match state.phase { + ReceiverState::Prepared | ReceiverState::Cancelled => { + state.resources.take(); + state.phase = ReceiverState::Cancelled; + Ok(()) + } + _ => Err(operation(OperationError::Busy)), + } + } + + /// Transfers preparation into canonical Idle takeover and exact-root activation. + /// + /// The current Idle control must name the attempt's incarnation and an epoch + /// at least as new as its source. Intervening canonical acquisition/release + /// may advance that epoch. Admission is rechecked before CAS and after restore. A dropped + /// waiter cannot cancel accepted acquisition. Inspect the local lifecycle, + /// authority and actor after a lost reply; repeated dispatch never reacquires. + pub async fn activate_prepared_receiver( + &self, + prepared: &PreparedCellReceiver, + authority: CellAuthority, + observed: VersionedControl, + owner: Owner, + now_ms: i64, + ) -> crate::Result { + let response = { + let mut registry = self + .inner + .receivers + .lock() + .map_err(|_| Error::RuntimeClosed)?; + self.ensure_acquiring()?; + let credit = checked_credit(®istry, prepared)?; + let control = observed.value(); + if control.cell != credit.spec.target.cell_id() + || control.incarnation != credit.spec.incarnation + || control.epoch < credit.spec.source_epoch + || control.state != crate::control::ControlState::Idle + || control.owner.is_some() + || control.root.is_none() + || owner.session != credit.spec.destination + { + return Err(Error::Fenced); + } + if now_ms < 0 || now_ms >= credit.reservation.expires_at_ms { + return Err(operation(OperationError::Deadline)); + } + let mut state = credit.state.lock().map_err(|_| Error::RuntimeClosed)?; + if state.phase != ReceiverState::Prepared { + return Err(operation(OperationError::Busy)); + } + let resources = state.resources.take().ok_or(Error::RuntimeClosed)?; + state.phase = ReceiverState::Activating; + drop(state); + let runtime = self.clone(); + let (reply, response) = oneshot::channel(); + let id = credit.spec.id; + // The registry owns completion, and the task owns the runtime until + // it joins all accepted preparation. Neither depends on the RPC. + let task = tokio::spawn(async move { + let completion = ActivationCompletion(Arc::clone(&credit)); + let result = runtime + .activate_receiver_credit(&credit, resources, authority, observed, owner) + .await; + if let Ok(mut state) = credit.state.lock() { + state.phase = if result.is_ok() { + ReceiverState::Activated + } else { + ReceiverState::Failed + }; + } + let _ = reply.send(result); + drop(completion); + }); + registry.tasks.push((id, task)); + response + }; + response.await.map_err(|_| Error::RuntimeClosed)? + } + + async fn activate_receiver_credit( + &self, + credit: &ReceiverCredit, + resources: ReceiverResources, + authority: CellAuthority, + observed: VersionedControl, + owner: Owner, + ) -> crate::Result { + let ReceiverResources { cell, job, disk } = resources; + let host = self + .inner + .replica_host + .clone() + .with_local_disk_budget(disk.clone()); + let result = async { + let replica = Self::replica_with_host_cache( + credit.replica.clone(), + host.clone(), + &credit.destination, + ) + .await?; + self.acquire_idle_reserved( + credit.catalog.clone(), + replica, + authority, + observed, + credit.destination.clone(), + owner, + cell, + Some(job), + None, + ) + .await + } + .await; + host.drain_cache_fills().await; + let finished = disk.finish_preparation(); + match result { + Ok(handle) => { + finished?; + Ok(handle) + } + Err(error) => { + let _ = finished; + Err(error) + } + } + } + + /// Retires a joined terminal local receipt after its result is durably recorded. + /// + /// The caller must reconcile failed/unknown authority outcomes first. This + /// never closes a serving actor, and never establishes journal permit cleanup. + /// Attempt identities must never be reused after retirement. + pub async fn retire_prepared_receiver( + &self, + prepared: &PreparedCellReceiver, + ) -> crate::Result<()> { + let task = { + let mut registry = self + .inner + .receivers + .lock() + .map_err(|_| Error::RuntimeClosed)?; + let credit = checked_credit(®istry, prepared)?; + let state = credit.state.lock().map_err(|_| Error::RuntimeClosed)?; + if matches!( + state.phase, + ReceiverState::Prepared | ReceiverState::Activating + ) { + return Err(operation(OperationError::Busy)); + } + let index = registry + .tasks + .iter() + .position(|(id, _)| *id == credit.spec.id); + if index.is_some_and(|index| !registry.tasks[index].1.is_finished()) { + return Err(operation(OperationError::Busy)); + } + let task = index.map(|index| registry.tasks.swap_remove(index).1); + registry.credits.remove(&credit.spec.id); + task + }; + if let Some(task) = task { + task.await.map_err(Error::WorkerJoin)?; + } + Ok(()) + } + + pub(super) async fn drain_prepared_receivers(&self) -> crate::Result<()> { + let tasks = { + let mut registry = self + .inner + .receivers + .lock() + .map_err(|_| Error::RuntimeClosed)?; + for credit in registry.credits.values() { + let mut state = credit.state.lock().map_err(|_| Error::RuntimeClosed)?; + if state.phase == ReceiverState::Prepared { + state.resources.take(); + state.phase = ReceiverState::Cancelled; + } + } + std::mem::take(&mut registry.tasks) + }; + let mut failure = None; + for (_, task) in tasks { + if let Err(error) = task.await { + failure.get_or_insert(Error::WorkerJoin(error)); + } + } + self.inner + .receivers + .lock() + .map_err(|_| Error::RuntimeClosed)? + .credits + .clear(); + failure.map_or(Ok(()), Err) + } +} + +fn checked_credit( + registry: &ReceiverRegistry, + prepared: &PreparedCellReceiver, +) -> crate::Result> { + let credit = prepared.credit()?; + if registry + .credits + .get(&credit.spec.id) + .is_none_or(|current| !Arc::ptr_eq(current, &credit)) + { + return Err(Error::Fenced); + } + Ok(credit) +} + +// A panic leaves an inspectable unknown result, never a reusable preparation. +struct ActivationCompletion(Arc); +impl Drop for ActivationCompletion { + fn drop(&mut self) { + if let Ok(mut state) = self.0.state.lock() + && state.phase == ReceiverState::Activating + { + state.phase = ReceiverState::Failed; + } + } +} + +fn limits_identity(limits: cellule_ltx::Limits) -> (u64, u64, u64, u64, usize) { + ( + limits.max_database_bytes, + limits.max_capture_bytes, + limits.max_file_bytes, + limits.max_plan_bytes, + limits.max_segments, + ) +} diff --git a/crates/cellule-runtime/src/cell/actor/runtime.rs b/crates/cellule-runtime/src/cell/actor/runtime.rs index c281113e..3e5ae0b4 100644 --- a/crates/cellule-runtime/src/cell/actor/runtime.rs +++ b/crates/cellule-runtime/src/cell/actor/runtime.rs @@ -154,6 +154,7 @@ impl CellRuntime { let publications = broadcast::Sender::new(PUBLICATION_NOTIFICATIONS); let node_lease = Arc::new(node_lease); let unpublished_node_log_bytes = Arc::new(AtomicU64::new(0)); + let node_admission = NodeAdmission::default(); runtime.spawn(run( receiver, pool.clone(), @@ -161,6 +162,7 @@ impl CellRuntime { Arc::clone(&unpublished_node_log_bytes), telemetry.clone(), publications.clone(), + node_admission.clone(), )); Ok(Self { inner: Arc::new(RuntimeInner { @@ -169,10 +171,11 @@ impl CellRuntime { resources, primitive_jobs, shutting_down: AtomicBool::new(false), - accepting_cells: AtomicBool::new(true), + node_admission, session, pool, replica_host, + receivers: std::sync::Mutex::new(receiver::ReceiverRegistry::default()), application_limits: OnceLock::new(), node_lease, node_durability: Arc::new(std::sync::RwLock::new(None)), @@ -295,6 +298,9 @@ impl CellRuntime { durability: Arc, ) -> crate::Result<()> { self.ensure_running()?; + if durability.identity()?.0 != self.inner.session { + return Err(Error::Control("Cell runtime node durability boot differs")); + } let mut slot = self .inner .node_durability @@ -312,17 +318,28 @@ impl CellRuntime { /// Returns the currently installed node-log durability binding. #[must_use] pub fn node_durability(&self) -> Option<(ApplicationId, Arc)> { + self.try_node_durability().ok().flatten() + } + + /// Reads the original local binding, preserving a failed lock as an error. + /// Absence supplies no directory authority or supervisor-completion proof. + pub fn try_node_durability( + &self, + ) -> crate::Result)>> { self.inner .node_durability .read() - .ok() - .and_then(|slot| slot.clone()) + .map(|slot| slot.clone()) + .map_err(|_| Error::Control("Cell runtime node durability lock poisoned")) } - /// Replaces the active node-log durability binding after an epoch close. + /// Replaces the expected node-log binding after an epoch close. The identity + /// check and replacement share one write lock; stale supervisors cannot + /// overwrite a different binding installed while they awaited provider I/O. pub fn replace_node_durability( &self, application: ApplicationId, + expected: &Arc, durability: Arc, ) -> crate::Result> { self.ensure_running()?; @@ -331,7 +348,7 @@ impl CellRuntime { .node_durability .write() .map_err(|_| Error::Control("Cell runtime node durability lock poisoned"))?; - let Some((installed_application, _)) = slot.as_ref() else { + let Some((installed_application, installed)) = slot.as_ref() else { return Err(Error::Control( "Cell runtime node durability is not installed", )); @@ -341,6 +358,23 @@ impl CellRuntime { "Cell runtime node durability application changed", )); } + if !Arc::ptr_eq(installed, expected) { + return Err(Error::Control( + "Cell runtime node durability binding changed", + )); + } + let previous_identity = installed.identity()?; + let identity = durability.identity()?; + if identity.0 != previous_identity.0 || identity.1 != previous_identity.1 { + return Err(Error::Control( + "Cell runtime node durability identity changed", + )); + } + if identity.2 <= previous_identity.2 { + return Err(Error::Control( + "Cell runtime node durability epoch did not advance", + )); + } let (_, previous) = slot .replace((application, durability)) .ok_or(Error::Control("Cell runtime node durability disappeared"))?; @@ -352,6 +386,7 @@ impl CellRuntime { /// Returns the first unobserved background release failure, including one /// completed before shutdown was requested, after closing the remaining work. pub async fn shutdown(&self) -> crate::Result<()> { + self.inner.node_admission.begin_drain()?; if self .inner .shutting_down @@ -361,6 +396,9 @@ impl CellRuntime { return Err(Error::RuntimeClosed); } self.inner.primitive_jobs.close(); + // Cancel unused receiver credit and join accepted takeover before the + // actor/worker close barrier. Retained caller handles are weak references. + let receivers = self.drain_prepared_receivers().await; let (reply, response) = oneshot::channel(); self.inner .sender @@ -376,21 +414,32 @@ impl CellRuntime { // Admission and replica work are stopped. Optional fills outlive their // readers, so keep artifacts/executors until accepted fills complete. self.inner.replica_host.drain_cache_fills().await; - drain.and(workers).and(durability) + receivers.and(drain).and(workers).and(durability) } /// Stops new Cell acquisition while existing owners continue serving. pub fn stop_acquiring(&self) -> crate::Result<()> { self.ensure_running()?; - self.inner.accepting_cells.store(false, Ordering::Release); - Ok(()) + self.inner.node_admission.cordon() } /// Reports whether a new owner may be acquired on this node. #[must_use] pub fn is_acquiring(&self) -> bool { !self.inner.shutting_down.load(Ordering::Acquire) - && self.inner.accepting_cells.load(Ordering::Acquire) + && self.inner.node_admission.check_new_role().is_ok() + } + + /// Shares the node's lifecycle and pressure gate with local role facilities. + /// The host installs this gate on its follower store before readiness. + #[must_use] + pub fn node_admission(&self) -> NodeAdmission { + self.inner.node_admission.clone() + } + + /// Returns the stable local classifier sample without refreshing its time. + pub fn operational_sample(&self) -> crate::Result> { + self.inner.node_admission.sample() } /// Starts bounded, actor-owned eviction of safe idle Cells. @@ -441,6 +490,39 @@ impl CellRuntime { response.await.map_err(|_| Error::RuntimeClosed)? } + /// Observes a bounded page of all local owners and lifecycle transitions. + /// + /// The cursor pins ownership topology, not SQL state or authority. Changed + /// topology rejects the cursor; restart the scan. Busy/draining Cells are + /// retained, and each release still needs fresh generation/authority checks. + /// The page owns retained-byte admission until dropped, including when its + /// caller cancels before receiving the actor's response. + pub async fn fleet_cells_page( + &self, + cursor: Option, + limit: usize, + ) -> crate::Result { + self.ensure_running()?; + inventory::validate_limit(limit)?; + let retained = self.inner.resources.try_reserve( + ResourceCost::zero() + .with_retained_bytes(crate::fleet::operations::MAX_PAGE_BYTES as usize), + )?; + let (reply, response) = oneshot::channel(); + self.inner + .sender + .send(Message::FleetCellsPage { + session: self.inner.session, + cursor, + limit, + retained, + reply, + }) + .await + .map_err(|_| Error::RuntimeClosed)?; + response.await.map_err(|_| Error::RuntimeClosed)? + } + /// Lists tenant-scoped targets of active, non-draining owners for maintenance. /// /// Targets come from verified activation proofs, including owners whose @@ -475,17 +557,41 @@ impl CellRuntime { cell: CellId, source: SessionId, generation: u64, + ) -> crate::Result<()> { + let (reply, response) = oneshot::channel(); + self.request_idle_release(cell, source, generation, None, DrainReply::Unit(reply)) + .await?; + response.await.map_err(|_| Error::RuntimeClosed)? + } + + /// Stops new foreground work for one exact maintenance source. + /// + /// Already actor-admitted work and native exact-lease completion/validation + /// continue through the normal transaction and durability gates. This + /// transition is sticky for this activation and returns once installed; + /// it does not prove primitive settlement, release, or fleet relocation. + /// Applications authorize maintenance and retain its durable node intent + /// before invoking this local boundary for a selected Cell. + pub async fn quiesce_cell_at( + &self, + cell: CellId, + source: SessionId, + generation: u64, + incarnation: crate::identity::IncarnationId, + epoch: u64, ) -> crate::Result<()> { self.ensure_running()?; - if source != self.inner.session || generation == 0 { + if source != self.inner.session || generation == 0 || epoch == 0 { return Err(Error::Fenced); } let (reply, response) = oneshot::channel(); self.inner .sender - .send(Message::ReleaseIdleCell { + .send(Message::QuiesceCell { cell, generation, + incarnation, + epoch, reply, }) .await @@ -493,6 +599,104 @@ impl CellRuntime { response.await.map_err(|_| Error::RuntimeClosed)? } + /// Quiesces and releases a busy maintenance source through canonical publication. + /// + /// The deadline bounds preflight, not an already confirmed release. A refused + /// preflight proves this request started no canonical release. Foreground + /// closure remains sticky; native completion remains available after refusal. + /// The runtime owns accepted work independently of the caller's waiter. + pub async fn release_maintenance_cell_at( + &self, + cell: CellId, + source: SessionId, + generation: u64, + incarnation: crate::identity::IncarnationId, + epoch: u64, + deadline: tokio::time::Instant, + ) -> crate::Result { + self.ensure_running()?; + if source != self.inner.session || generation == 0 || epoch == 0 { + return Err(Error::Fenced); + } + let (reply, response) = oneshot::channel(); + self.inner + .sender + .send(Message::ReleaseMaintenanceCell( + maintenance::ReleaseRequest { + cell, + generation, + incarnation, + epoch, + deadline: deadline.into_std(), + reply, + }, + )) + .await + .map_err(|_| Error::RuntimeClosed)?; + response.await.map_err(|_| Error::RuntimeClosed)? + } + + /// Releases an exact source identity and returns the canonical final position. + /// + /// The actor checks incarnation and ownership epoch before closing admission, + /// then its existing worker-close/publication path captures the exact released + /// root. A later authority read cannot replace this proof with a newer root. + /// [`Error::CellReleaseRefused`] identifies a definite refusal before this + /// request began canonical close; all other errors retain uncertainty. + pub async fn release_idle_cell_at( + &self, + cell: CellId, + source: SessionId, + generation: u64, + incarnation: crate::identity::IncarnationId, + epoch: u64, + ) -> crate::Result { + if epoch == 0 { + return Err(Error::CellReleaseRefused { + blocker: DrainBlocker::IncompleteObservation, + source: Box::new(Error::Fenced), + }); + } + let (reply, response) = oneshot::channel(); + self.request_idle_release( + cell, + source, + generation, + Some((incarnation, epoch)), + DrainReply::Position(reply), + ) + .await + .map_err(|source| Error::CellReleaseRefused { + blocker: DrainBlocker::IncompleteObservation, + source: Box::new(source), + })?; + response.await.map_err(|_| Error::RuntimeClosed)? + } + + async fn request_idle_release( + &self, + cell: CellId, + source: SessionId, + generation: u64, + expected: Option<(crate::identity::IncarnationId, u64)>, + reply: DrainReply, + ) -> crate::Result<()> { + self.ensure_running()?; + if source != self.inner.session || generation == 0 { + return Err(Error::Fenced); + } + self.inner + .sender + .send(Message::ReleaseIdleCell { + cell, + generation, + expected, + reply, + }) + .await + .map_err(|_| Error::RuntimeClosed) + } + /// Feeds one measured node sample into the actor-owned hysteretic pressure /// controller. Sustained shedding starts the same bounded idle-eviction /// path exposed by [`Self::evict_idle`]. @@ -620,6 +824,23 @@ impl CellRuntime { /// and returns `RuntimeClosed` once terminal drain begins. pub fn try_reserve_node_bytes(&self, bytes: usize) -> crate::Result { self.ensure_running()?; + self.try_reserve_node_metadata_bytes(bytes) + } + + /// Reserves bounded node lifecycle metadata in the same retained-byte ledger. + /// + /// Metadata owners may be installed before a live node lease exists, or + /// observed after fencing. This token grants no native work, role admission + /// or authority. Native work still uses [`Self::try_reserve_node_bytes`]. + /// Zero/full reservations fail with capacity; terminal drain closes new + /// metadata admission with `RuntimeClosed`. Dropping the token releases it. + pub fn try_reserve_node_metadata_bytes( + &self, + bytes: usize, + ) -> crate::Result { + if self.inner.shutting_down.load(Ordering::Acquire) { + return Err(Error::RuntimeClosed); + } if bytes == 0 { return Err(Error::Capacity("node retained bytes")); } diff --git a/crates/cellule-runtime/src/cell/actor/state.rs b/crates/cellule-runtime/src/cell/actor/state.rs index 886251bf..0fb77544 100644 --- a/crates/cellule-runtime/src/cell/actor/state.rs +++ b/crates/cellule-runtime/src/cell/actor/state.rs @@ -8,10 +8,11 @@ pub(super) struct RuntimeInner { pub(super) resources: ResourceLedger, pub(super) primitive_jobs: Arc, pub(super) shutting_down: AtomicBool, - pub(super) accepting_cells: AtomicBool, + pub(super) node_admission: NodeAdmission, pub(super) session: SessionId, pub(super) pool: SqlWorkerPool, pub(super) replica_host: cellule_ltx::Host, + pub(super) receivers: std::sync::Mutex, pub(super) application_limits: OnceLock>, pub(super) node_lease: Arc, pub(super) node_durability: NodeDurabilitySlot, @@ -63,6 +64,7 @@ pub(super) struct RestoredActivation { pub(super) schema: u32, pub(super) root: cellule_ltx::RootRef, pub(super) reservation: CellReservation, + pub(super) job: Option, } pub(super) struct BootstrapActivation { @@ -115,6 +117,13 @@ pub(super) enum Message { IdleTransferCandidates { reply: oneshot::Sender>, }, + FleetCellsPage { + session: SessionId, + cursor: Option, + limit: usize, + retained: ResourceReservation, + reply: oneshot::Sender>, + }, ActiveCatalogEntries { reply: oneshot::Sender>>, }, @@ -124,11 +133,20 @@ pub(super) enum Message { UnreleasedCellCount { reply: oneshot::Sender>, }, - ReleaseIdleCell { + QuiesceCell { cell: CellId, generation: u64, + incarnation: crate::identity::IncarnationId, + epoch: u64, reply: oneshot::Sender>, }, + ReleaseMaintenanceCell(maintenance::ReleaseRequest), + ReleaseIdleCell { + cell: CellId, + generation: u64, + expected: Option<(crate::identity::IncarnationId, u64)>, + reply: DrainReply, + }, ObservePressure { sample: PressureSample, reply: oneshot::Sender>, @@ -253,11 +271,15 @@ pub(super) struct ActiveCell { pub(super) queue: VecDeque, pub(super) coordination: CoordinationState, pub(super) persisted_work: crate::primitives::maintenance::PersistedWorkInventory, + pub(super) demand: inventory::CellDemandState, + pub(super) resource_limits: cellule_ltx::Limits, pub(super) inventory_refreshing: bool, - pub(super) drain: Option>>, + pub(super) inventory_revision: u64, + pub(super) drain: Option, // Transfer closes the old capability and installs a fresh one; failed fresh // inventory must leave the current owner serving through that capability. pub(super) transfer: Option, + pub(super) resident_since_ms: i64, pub(super) last_used_ms: i64, pub(super) last_work_at: std::time::Instant, pub(super) compaction_retry_at: std::time::Instant, @@ -271,7 +293,54 @@ pub(super) struct ActiveCell { } pub(super) struct TransferPreflight { - pub(super) reply: oneshot::Sender>, + pub(super) reply: DrainReply, + pub(super) maintenance: Option, +} + +pub(super) enum DrainReply { + Maintenance(oneshot::Sender>), + Unit(oneshot::Sender>), + Position(oneshot::Sender>), +} + +impl DrainReply { + /// Only call before this request transfers its reply to canonical close. + /// A later close/publication error must remain an uncertain release. + pub(super) fn refuse(self, blocker: DrainBlocker, error: Error) { + let error = if matches!(&self, Self::Position(_)) { + Error::CellReleaseRefused { + blocker, + source: Box::new(error), + } + } else { + error + }; + let _ = self.send(Err(error)); + } + + pub(super) fn send(self, result: crate::Result<()>) -> Result<(), crate::Result<()>> { + self.send_released(result, None) + } + + pub(super) fn send_released( + self, + result: crate::Result<()>, + released: Option, + ) -> Result<(), crate::Result<()>> { + match self { + Self::Maintenance(reply) => reply + .send( + result + .and_then(|()| released.ok_or(Error::Fenced)) + .map(MaintenanceCellRelease::Released), + ) + .map_err(|result| result.map(|_| ())), + Self::Unit(reply) => reply.send(result), + Self::Position(reply) => reply + .send(result.and_then(|()| released.ok_or(Error::Fenced))) + .map_err(|result| result.map(|_| ())), + } + } } pub(super) struct QueuedPublication { @@ -352,7 +421,7 @@ pub(super) enum TaskResult { Arc, Option, )>, - persisted_work: crate::Result, + inventory: crate::Result, }, Hydrated { cell: CellId, @@ -364,7 +433,8 @@ pub(super) enum TaskResult { cell: CellId, generation: u64, effect_id: u64, - result: crate::Result, + inventory_revision: u64, + result: crate::Result, }, TransferPreflight { cell: CellId, @@ -372,6 +442,12 @@ pub(super) enum TaskResult { effect_id: u64, result: crate::Result, }, + MaintenancePreflight { + cell: CellId, + generation: u64, + effect_id: u64, + result: crate::Result, + }, Executed { cell: CellId, generation: u64, @@ -450,9 +526,10 @@ pub(super) enum TaskResult { Deactivated { cell: CellId, generation: u64, - reply: Option>>, + reply: Option, shutdown_drain: bool, result: crate::Result<()>, + released: Option, }, } diff --git a/crates/cellule-runtime/src/cell/actor/task.rs b/crates/cellule-runtime/src/cell/actor/task.rs index 38dc6276..f31f43a9 100644 --- a/crates/cellule-runtime/src/cell/actor/task.rs +++ b/crates/cellule-runtime/src/cell/actor/task.rs @@ -13,6 +13,7 @@ pub(super) async fn run( unpublished_node_log_bytes: Arc, telemetry: crate::fleet::telemetry::CellTelemetryHandle, publications: broadcast::Sender, + admission: NodeAdmission, ) { let mut cells = HashMap::::new(); let mut transitioning = HashSet::::new(); @@ -41,6 +42,17 @@ pub(super) async fn run( pressure_tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); pressure_tick.tick().await; loop { + if !shutdown.draining { + maintenance::drive( + &pool, + &mut cells, + &mut transitioning, + &mut tasks, + &node_lease, + &mut movement, + &mut movement_permits, + ); + } if shutdown.draining { if tasks.is_empty() { if !cells.is_empty() || !transitioning.is_empty() { @@ -93,7 +105,7 @@ pub(super) async fn run( } break; }; - handle_message(message, &mut receiver, &pool, &mut cells, &mut transitioning, &mut tasks, &mut shutdown, &node_lease, &telemetry, &mut pressure, &mut movement, &mut movement_permits, &mut next_generation); + handle_message(message, &mut receiver, &pool, &mut cells, &mut transitioning, &mut tasks, &mut shutdown, &node_lease, &telemetry, &mut pressure, &admission, &mut movement, &mut movement_permits, &mut next_generation); } _ = renewal_tick.tick() => { renewals.scan_due(&cells); @@ -114,6 +126,7 @@ pub(super) async fn run( &pool, &telemetry, &mut pressure, + &admission, &mut cells, &mut transitioning, &mut tasks, @@ -148,7 +161,7 @@ pub(super) async fn run( } break; }; - handle_message(message, &mut receiver, &pool, &mut cells, &mut transitioning, &mut tasks, &mut shutdown, &node_lease, &telemetry, &mut pressure, &mut movement, &mut movement_permits, &mut next_generation); + handle_message(message, &mut receiver, &pool, &mut cells, &mut transitioning, &mut tasks, &mut shutdown, &node_lease, &telemetry, &mut pressure, &admission, &mut movement, &mut movement_permits, &mut next_generation); } result = tasks.join_next() => { let Some(Ok(result)) = result else { @@ -182,6 +195,7 @@ pub(super) async fn run( &pool, &telemetry, &mut pressure, + &admission, &mut cells, &mut transitioning, &mut tasks, @@ -197,10 +211,15 @@ pub(super) async fn run( /// /// A ledger that cannot be read, or a node without configured limits, yields no /// sample: pressure policy must never stop the actor loop. +#[expect( + clippy::too_many_arguments, + reason = "the sample shares the actor ledger, classifier, admission gate, and movement state" +)] fn sample_node_pressure( pool: &SqlWorkerPool, telemetry: &crate::fleet::telemetry::CellTelemetryHandle, pressure: &mut PressureClassifier, + admission: &NodeAdmission, cells: &mut HashMap, transitioning: &mut HashSet, tasks: &mut JoinSet, @@ -217,6 +236,7 @@ fn sample_node_pressure( sample, telemetry, pressure, + admission, pool, cells, transitioning, @@ -238,6 +258,7 @@ fn classify_pressure_sample( sample: PressureSample, telemetry: &crate::fleet::telemetry::CellTelemetryHandle, pressure: &mut PressureClassifier, + admission: &NodeAdmission, pool: &SqlWorkerPool, cells: &mut HashMap, transitioning: &mut HashSet, @@ -246,6 +267,7 @@ fn classify_pressure_sample( movement_permits: &mut HashMap, ) -> crate::Result { let state = pressure.observe(sample)?; + admission.observe(state, sample.at_ms)?; telemetry.pressure_state(state); if matches!(state, PressureState::Shedding | PressureState::Critical) && movement.in_flight() < 2 @@ -280,7 +302,9 @@ pub(super) fn start_shutdown_drain( for (cell, active) in cells.iter_mut() { if let Some(transfer) = active.transfer.take() { active.coordination.step(CoordinationInput::AbortTransfer); - let _ = transfer.reply.send(Err(Error::CellDraining)); + transfer + .reply + .refuse(DrainBlocker::BusyExecution, Error::CellDraining); } active.inventory_refreshing = false; active.coordination.step(CoordinationInput::BeginShutdown); @@ -325,6 +349,7 @@ pub(super) fn handle_message( node_lease: &RuntimeNodeLease, telemetry: &crate::fleet::telemetry::CellTelemetryHandle, pressure: &mut PressureClassifier, + admission: &NodeAdmission, movement: &mut MovementBudget, movement_permits: &mut HashMap, next_generation: &mut u64, @@ -365,26 +390,33 @@ pub(super) fn handle_message( bootstrap_and_publish(cell, &pool, &mut publisher, *activation).await } }; - let (result, persisted_work) = match result { - Ok(hydration) => { - match pool.interrupt_handle(cell).await { - Ok(interrupt) => { - let persisted_work = - pool.persisted_work_inventory(cell, role).await; - (Ok((Arc::new(interrupt), hydration)), persisted_work) - } - Err(error) => { - let result = - match cleanup_failed_activation(cell, &pool, &mut publisher) - .await - { - Ok(()) => error, - Err(cleanup) => cleanup, - }; - (Err(result), Err(Error::CellNotActive)) - } + let (result, inventory) = match result { + Ok(hydration) => match pool.interrupt_handle(cell).await { + Ok(interrupt) => { + let inventory = pool + .fleet_inventory( + cell, + role, + unix_millis(), + SqlDeadline::new(std::time::Instant::now() + SQL_WALL_DEADLINE), + ) + .await; + (Ok((Arc::new(interrupt), hydration)), inventory) } - } + Err(error) => { + let result = match cleanup_failed_activation( + cell, + &pool, + &mut publisher, + ) + .await + { + Ok(()) => error, + Err(cleanup) => cleanup, + }; + (Err(result), Err(Error::CellNotActive)) + } + }, Err(error) => { let result = match cleanup_failed_activation(cell, &pool, &mut publisher).await { @@ -403,7 +435,7 @@ pub(super) fn handle_message( admission, reply, result, - persisted_work, + inventory, } }); } @@ -412,12 +444,16 @@ pub(super) fn handle_message( send_command_reply(&mut command, Err(Error::CellNotActive)); return; }; - if active.transfer.is_some() { + if active + .transfer + .as_ref() + .is_some_and(|transfer| transfer.maintenance.is_none()) + { send_command_reply(&mut command, Err(Error::CellDraining)); return; } match active.coordination.step(CoordinationInput::Admit { - kind: AdmissionKind::Command, + kind: command._work.kind, admission_matches: Arc::ptr_eq(&active.admission, &command.admission), }) { CoordinationDecision::Admit => { @@ -435,12 +471,16 @@ pub(super) fn handle_message( send_query_reply(&mut query, Err(Error::CellNotActive)); return; }; - if active.transfer.is_some() { + if active + .transfer + .as_ref() + .is_some_and(|transfer| transfer.maintenance.is_none()) + { send_query_reply(&mut query, Err(Error::CellDraining)); return; } match active.coordination.step(CoordinationInput::Admit { - kind: AdmissionKind::Query, + kind: query._work.kind, admission_matches: Arc::ptr_eq(&active.admission, &query.admission), }) { CoordinationDecision::Admit => { @@ -458,7 +498,11 @@ pub(super) fn handle_message( send_resolve_reply(&mut resolve, Err(Error::CellNotActive)); return; }; - if active.transfer.is_some() { + if active + .transfer + .as_ref() + .is_some_and(|transfer| transfer.maintenance.is_none()) + { send_resolve_reply(&mut resolve, Ok(Resolution::Unknown)); return; } @@ -602,14 +646,16 @@ pub(super) fn handle_message( if active.transfer.is_some() { if let Some(transfer) = active.transfer.take() { active.coordination.step(CoordinationInput::AbortTransfer); - let _ = transfer.reply.send(Err(Error::CellDraining)); + transfer + .reply + .refuse(DrainBlocker::BusyExecution, Error::CellDraining); } let decision = active.coordination.step(CoordinationInput::BeginDrain); if let CoordinationDecision::Reject(reason) = decision { let _ = reply.send(Err(rejection_error(reason))); return; } - active.drain = Some(reply); + active.drain = Some(DrainReply::Unit(reply)); if matches!( schedule(active, node_lease.check().is_ok()), CoordinationDecision::ReadyToDeactivate @@ -623,7 +669,7 @@ pub(super) fn handle_message( let _ = reply.send(Err(rejection_error(reason))); return; } - active.drain = Some(reply); + active.drain = Some(DrainReply::Unit(reply)); match schedule(active, node_lease.check().is_ok()) { CoordinationDecision::ReadyToDeactivate => { start_deactivate(cell, pool, cells, transitioning, tasks); @@ -662,6 +708,24 @@ pub(super) fn handle_message( .collect(); let _ = reply.send(Ok(candidates)); } + Message::FleetCellsPage { + session, + cursor, + limit, + retained, + reply, + } => { + let result = inventory::collect_page( + cells, + transitioning, + *next_generation, + session, + cursor, + limit, + retained, + ); + let _ = reply.send(result); + } Message::ActiveCatalogEntries { reply } => { if node_lease.check().is_err() { let _ = reply.send(Err(Error::Fenced)); @@ -687,35 +751,97 @@ pub(super) fn handle_message( Message::UnreleasedCellCount { reply } => { let _ = reply.send(Ok(cells.len().saturating_add(transitioning.len()))); } + Message::QuiesceCell { + cell, + generation, + incarnation, + epoch, + reply, + } => { + let result = (|| { + let active = cells.get_mut(&cell).ok_or(Error::CellNotActive)?; + if active.generation != generation || active.incarnation != incarnation { + return Err(Error::Fenced); + } + let publisher = active.publisher.as_ref().ok_or(Error::CellDraining)?; + if publisher.control().value().epoch != epoch { + return Err(Error::Fenced); + } + match active + .coordination + .step(CoordinationInput::BeginMaintenanceQuiescence) + { + CoordinationDecision::Started => { + // Keep the same bounded request/byte permits and exact + // completion capability. Ordinary work checks both this + // shared flag and the actor's ordered kernel admission. + active + .admission + .maintenance_quiescing + .store(true, Ordering::Release); + Ok(()) + } + CoordinationDecision::Reject(reason) => Err(rejection_error(reason)), + _ => Err(Error::CellDraining), + } + })(); + let _ = reply.send(result); + } + Message::ReleaseMaintenanceCell(request) => { + if let Some(cell) = maintenance::begin(request, cells, movement, movement_permits) { + continue_cell(cell, pool, cells, transitioning, tasks, node_lease); + } + } Message::ReleaseIdleCell { cell, generation, + expected, reply, } => { if node_lease.check().is_err() { - let _ = reply.send(Err(Error::Fenced)); + reply.refuse(DrainBlocker::IncompleteObservation, Error::Fenced); return; } let Some(active) = cells.get(&cell) else { - let _ = reply.send(Err(Error::CellNotActive)); + reply.refuse(DrainBlocker::IncompleteObservation, Error::CellNotActive); return; }; + if let Some((incarnation, epoch)) = expected { + if active.incarnation != incarnation || active.generation != generation { + reply.refuse(DrainBlocker::IncompleteObservation, Error::Fenced); + return; + } + match active.publisher.as_ref() { + Some(publisher) if publisher.control().value().epoch != epoch => { + reply.refuse(DrainBlocker::IncompleteObservation, Error::Fenced); + return; + } + None => { + reply.refuse(DrainBlocker::BusyExecution, Error::CellDraining); + return; + } + Some(_) => {} + } + } if active.generation != generation || active.draining() || active.transfer.is_some() || active.inventory_refreshing { - let _ = reply.send(Err(Error::CellDraining)); + reply.refuse(DrainBlocker::BusyExecution, Error::CellDraining); return; } let Ok(mut permit) = movement.try_start_requested(unix_millis()) else { - let _ = reply.send(Err(Error::Capacity("movement budget"))); + reply.refuse( + DrainBlocker::MovementBudget, + Error::Capacity("movement budget"), + ); return; }; { let Some(active) = cells.get_mut(&cell) else { movement.complete(&mut permit); - let _ = reply.send(Err(Error::CellNotActive)); + reply.refuse(DrainBlocker::IncompleteObservation, Error::CellNotActive); return; }; if active.transfer.is_some() @@ -723,7 +849,7 @@ pub(super) fn handle_message( || active.inventory_refreshing { movement.complete(&mut permit); - let _ = reply.send(Err(Error::CellDraining)); + reply.refuse(DrainBlocker::BusyExecution, Error::CellDraining); return; } let decision = @@ -739,22 +865,22 @@ pub(super) fn handle_message( CoordinationDecision::Fence => { fence_active(active); movement.complete(&mut permit); - let _ = reply.send(Err(Error::Fenced)); + reply.refuse(DrainBlocker::IncompleteObservation, Error::Fenced); return; } CoordinationDecision::Reject(reason) => { movement.complete(&mut permit); - let _ = reply.send(Err(rejection_error(reason))); + reply.refuse(DrainBlocker::BusyExecution, rejection_error(reason)); return; } CoordinationDecision::Ignored => { movement.complete(&mut permit); - let _ = reply.send(Err(Error::CellDraining)); + reply.refuse(DrainBlocker::BusyExecution, Error::CellDraining); return; } _ => { movement.complete(&mut permit); - let _ = reply.send(Err(Error::CellDraining)); + reply.refuse(DrainBlocker::BusyExecution, Error::CellDraining); return; } } @@ -762,7 +888,10 @@ pub(super) fn handle_message( active.admission.requests.close(); active.admission.bytes.close(); active.admission = new_cell_admission(active.admission.owner_fence); - active.transfer = Some(TransferPreflight { reply }); + active.transfer = Some(TransferPreflight { + reply, + maintenance: None, + }); }; movement_permits.insert(cell, permit); continue_cell(cell, pool, cells, transitioning, tasks, node_lease); @@ -772,6 +901,7 @@ pub(super) fn handle_message( sample, telemetry, pressure, + admission, pool, cells, transitioning, @@ -826,6 +956,9 @@ pub(super) fn reject_fenced_message(message: Message) { Message::IdleTransferCandidates { reply } => { let _ = reply.send(Err(Error::Fenced)); } + Message::FleetCellsPage { reply, .. } => { + let _ = reply.send(Err(Error::Fenced)); + } Message::ActiveCatalogEntries { reply } => { let _ = reply.send(Err(Error::Fenced)); } @@ -835,9 +968,15 @@ pub(super) fn reject_fenced_message(message: Message) { Message::UnreleasedCellCount { reply } => { let _ = reply.send(Err(Error::Fenced)); } - Message::ReleaseIdleCell { reply, .. } => { + Message::QuiesceCell { reply, .. } => { let _ = reply.send(Err(Error::Fenced)); } + Message::ReleaseMaintenanceCell(request) => { + let _ = request.reply.send(Err(Error::Fenced)); + } + Message::ReleaseIdleCell { reply, .. } => { + reply.refuse(DrainBlocker::IncompleteObservation, Error::Fenced); + } Message::ObservePressure { reply, .. } => { let _ = reply.send(Err(Error::Fenced)); } diff --git a/crates/cellule-runtime/src/cell/actor/tasks/activation.rs b/crates/cellule-runtime/src/cell/actor/tasks/activation.rs index 52b7ded1..97907e5c 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/activation.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/activation.rs @@ -20,7 +20,7 @@ pub(super) fn handle_activated( Arc, Option, )>, - persisted_work: crate::Result, + inventory: crate::Result, ) { let TaskContext { pool, @@ -93,9 +93,19 @@ pub(super) fn handle_activated( Residency::Sparse } }); - let persisted_work = match persisted_work { - Ok(inventory) => inventory, - Err(_) => { + let resource_limits = publisher.resource_limits(); + let mut demand = inventory::CellDemandState::default(); + let persisted_work = match inventory { + Ok(sample) => { + if let Err(error) = demand.record(sample, resource_limits, published_sequence) { + tracing::debug!(cell = ?cell, error = ?error, "Cell demand remains unknown after activation"); + crate::primitives::maintenance::PersistedWorkInventory::unknown() + } else { + sample.persisted_work + } + } + Err(error) => { + tracing::debug!(cell = ?cell, error = ?error, "Cell inventory remains unknown after activation"); // An inventory read is a safety precondition for // eviction. Unknown accounting must remain ineligible. crate::primitives::maintenance::PersistedWorkInventory::unknown() @@ -104,6 +114,7 @@ pub(super) fn handle_activated( // A restored or bootstrapped owner can already have a reader policy. // Its hint is advisory; receivers still reload the new authority. let _ = publications.send(catalog.entry().clone()); + let resident_since_ms = unix_millis(); cells.insert( cell, ActiveCell { @@ -123,10 +134,14 @@ pub(super) fn handle_activated( queue: VecDeque::new(), coordination: CoordinationState::serving_with_residency(true, residency), persisted_work, + demand, + resource_limits, inventory_refreshing: false, + inventory_revision: 1, drain: None, transfer: None, - last_used_ms: unix_millis(), + resident_since_ms, + last_used_ms: resident_since_ms, last_work_at: std::time::Instant::now(), compaction_retry_at: std::time::Instant::now(), hydration_retry_at: std::time::Instant::now(), diff --git a/crates/cellule-runtime/src/cell/actor/tasks/maintenance_transfer.rs b/crates/cellule-runtime/src/cell/actor/tasks/maintenance_transfer.rs new file mode 100644 index 00000000..586756a0 --- /dev/null +++ b/crates/cellule-runtime/src/cell/actor/tasks/maintenance_transfer.rs @@ -0,0 +1,128 @@ +//! Fresh maintenance readiness and the canonical final release barrier. + +use super::super::maintenance::{refuse, reopen_completion}; +use super::*; +use crate::fleet::operations::DrainBlocker; +use crate::primitives::maintenance_readiness::MaintenanceWorkInventory; + +pub(super) fn handle( + context: TaskContext<'_>, + cell: CellId, + generation: u64, + effect_id: u64, + result: crate::Result, +) { + let TaskContext { + pool, + cells, + transitioning, + tasks, + node_lease, + movement, + movement_permits, + .. + } = context; + let Some(active) = cells.get_mut(&cell) else { + return; + }; + if active.generation != generation + || !active + .coordination + .effect_matches(effect_id, CoordinationEffect::Inventory) + { + return; + } + active.finish_task(effect_id, CoordinationEffect::Inventory); + active.inventory_refreshing = false; + let Some(mut transfer) = active.transfer.take() else { + continue_cell(cell, pool, cells, transitioning, tasks, node_lease); + return; + }; + let Some(state) = transfer.maintenance.as_mut() else { + // A stale completion cannot classify a different transfer as maintenance. + active.transfer = Some(transfer); + continue_cell(cell, pool, cells, transitioning, tasks, node_lease); + return; + }; + if state.inventory_effect != Some(effect_id) { + // A previous request may expire while its read is still owned. Complete + // that effect, but never feed its snapshot/error to a newer request. + active.transfer = Some(transfer); + continue_cell(cell, pool, cells, transitioning, tasks, node_lease); + return; + } + state.inventory_effect = None; + let now = std::time::Instant::now(); + let failure = if node_lease.check().is_err() || active.coordination.is_fenced() { + super::super::admission::fence_active(active); + let _ = transfer.reply.send(Err(Error::Fenced)); + true + } else if now >= state.deadline { + reopen_completion(active); + refuse(transfer.reply, DrainBlocker::Deadline, None); + true + } else { + match result { + Err(Error::Fenced) => { + super::super::admission::fence_active(active); + let _ = transfer.reply.send(Err(Error::Fenced)); + true + } + Err(error) => { + reopen_completion(active); + refuse(transfer.reply, DrainBlocker::UnknownInventory, Some(error)); + true + } + Ok(inventory) if !inventory.is_transferable() => { + reopen_completion(active); + state.closing = false; + state.next_check = now + HYDRATION_TICK; + active.transfer = Some(transfer); + false + } + Ok(_) if !state.closing => { + let decision = + active + .coordination + .step(CoordinationInput::BeginTransferPreflight { + queue_empty: active.queue.is_empty(), + publication_idle: active.coordination.publication_count() == 0, + lease_live: node_lease.check().is_ok(), + }); + if decision == CoordinationDecision::Started { + // Close every new capability before joining accepted native + // work and refreshing readiness again. Keep semaphore owners + // intact until confirm, so a definite refusal can restore + // only native completion on this same exact activation. + active.admission.draining.store(true, Ordering::Release); + state.closing = true; + } + state.next_check = now + HYDRATION_TICK; + active.transfer = Some(transfer); + false + } + Ok(_) => { + if active.queue.is_empty() + && active.coordination.can_deactivate() + && active.publisher.is_some() + && active.unpublished_node_logs == 0 + && active.coordination.step(CoordinationInput::ConfirmTransfer) + == CoordinationDecision::ReadyToDeactivate + { + active.admission.requests.close(); + active.admission.bytes.close(); + active.drain = Some(transfer.reply); + start_deactivate(cell, pool, cells, transitioning, tasks); + return; + } + state.next_check = now + HYDRATION_TICK; + active.transfer = Some(transfer); + false + } + } + }; + if failure && let Some(mut permit) = movement_permits.remove(&cell) { + movement.complete(&mut permit); + } + continue_cell(cell, pool, cells, transitioning, tasks, node_lease); +} diff --git a/crates/cellule-runtime/src/cell/actor/tasks/mod.rs b/crates/cellule-runtime/src/cell/actor/tasks/mod.rs index 6c31cb6f..e0216007 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/mod.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/mod.rs @@ -13,6 +13,7 @@ use super::admission::{ use super::*; mod activation; +mod maintenance_transfer; mod movement; mod publication; mod residency; @@ -71,18 +72,10 @@ pub(super) fn handle_task( admission, reply, result, - persisted_work, + inventory, } => activation::handle_activated( - context, - cell, - generation, - role, - catalog, - publisher, - admission, - reply, - result, - persisted_work, + context, cell, generation, role, catalog, publisher, admission, reply, result, + inventory, ), TaskResult::Hydrated { cell, @@ -169,6 +162,12 @@ pub(super) fn handle_task( effect_id, result, } => movement::handle_transfer_preflight(context, cell, generation, effect_id, result), + TaskResult::MaintenancePreflight { + cell, + generation, + effect_id, + result, + } => maintenance_transfer::handle(context, cell, generation, effect_id, result), TaskResult::Migrated { cell, generation, @@ -195,8 +194,16 @@ pub(super) fn handle_task( cell, generation, effect_id, + inventory_revision, result, - } => residency::handle_inventory_refreshed(context, cell, generation, effect_id, result), + } => residency::handle_inventory_refreshed( + context, + cell, + generation, + effect_id, + inventory_revision, + result, + ), TaskResult::Renewed { cell, generation, @@ -210,8 +217,15 @@ pub(super) fn handle_task( reply, shutdown_drain, result, - } => { - residency::handle_deactivated(context, cell, generation, reply, shutdown_drain, result) - } + released, + } => residency::handle_deactivated( + context, + cell, + generation, + reply, + shutdown_drain, + result, + released, + ), } } diff --git a/crates/cellule-runtime/src/cell/actor/tasks/movement.rs b/crates/cellule-runtime/src/cell/actor/tasks/movement.rs index 9f02d24e..524ad813 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/movement.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/movement.rs @@ -100,7 +100,9 @@ pub(super) fn handle_transfer_preflight( } } } else { - let _ = transfer.reply.send(transfer_result); + if let Err(error) = transfer_result { + transfer.reply.refuse(DrainBlocker::UnknownInventory, error); + } if let Some(mut permit) = movement_permits.remove(&cell) { movement.complete(&mut permit); } diff --git a/crates/cellule-runtime/src/cell/actor/tasks/residency.rs b/crates/cellule-runtime/src/cell/actor/tasks/residency.rs index d8cc9726..9e22dd1e 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/residency.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/residency.rs @@ -8,7 +8,8 @@ pub(super) fn handle_inventory_refreshed( cell: CellId, generation: u64, effect_id: u64, - result: crate::Result, + inventory_revision: u64, + result: crate::Result, ) { let TaskContext { pool, @@ -30,8 +31,33 @@ pub(super) fn handle_inventory_refreshed( } active.finish_task(effect_id, CoordinationEffect::Inventory); active.inventory_refreshing = false; - if let Ok(inventory) = result { - active.persisted_work = inventory; + // An inventory effect can finish after another foreground mutation starts. + // Finish its effect, but do not let its older rows clear that mutation's + // unknown-work marker or recreate obsolete demand. + if active.inventory_revision != inventory_revision { + continue_cell(cell, pool, cells, transitioning, tasks, node_lease); + return; + } + match result { + Ok(sample) => { + if let Err(error) = + active + .demand + .record(sample, active.resource_limits, active.published_sequence) + { + active.persisted_work = + crate::primitives::maintenance::PersistedWorkInventory::unknown(); + tracing::debug!(cell = ?cell, error = ?error, "Cell demand remains unknown after inventory"); + } else { + active.persisted_work = sample.persisted_work; + } + } + Err(error) => { + active.persisted_work = + crate::primitives::maintenance::PersistedWorkInventory::unknown(); + active.demand.failed(unix_millis()); + tracing::debug!(cell = ?cell, error = ?error, "Cell demand inspection failed"); + } } continue_cell(cell, pool, cells, transitioning, tasks, node_lease); } @@ -85,9 +111,10 @@ pub(super) fn handle_deactivated( context: TaskContext<'_>, cell: CellId, generation: u64, - reply: Option>>, + reply: Option, shutdown_drain: bool, result: crate::Result<()>, + released: Option, ) { let TaskContext { cells, @@ -111,7 +138,7 @@ pub(super) fn handle_deactivated( match reply { Some(reply) => { let failed = result.is_err(); - match reply.send(result) { + match reply.send_released(result, released) { Err(Err(error)) => fail_shutdown(shutdown, error), _ if runtime_waiting && failed => fail_shutdown( shutdown, diff --git a/crates/cellule-runtime/src/cell/actor/tasks/work.rs b/crates/cellule-runtime/src/cell/actor/tasks/work.rs index 372bab42..b4d9353d 100644 --- a/crates/cellule-runtime/src/cell/actor/tasks/work.rs +++ b/crates/cellule-runtime/src/cell/actor/tasks/work.rs @@ -30,12 +30,12 @@ pub(super) fn handle_executed( if active.generation != generation || !active .coordination - .effect_matches(effect_id, CoordinationEffect::Work(AdmissionKind::Command)) + .effect_matches(effect_id, CoordinationEffect::Work(command._work.kind)) { send_command_task_reply(&mut command, result); return; } - active.finish_task(effect_id, CoordinationEffect::Work(AdmissionKind::Command)); + active.finish_task(effect_id, CoordinationEffect::Work(command._work.kind)); if node_lease.check().is_err() { result = Err(command.operation.unknown(Error::Fenced)); fenced = true; @@ -150,12 +150,12 @@ pub(super) fn handle_queried( if active.generation != generation || !active .coordination - .effect_matches(effect_id, CoordinationEffect::Work(AdmissionKind::Query)) + .effect_matches(effect_id, CoordinationEffect::Work(query._work.kind)) { send_query_reply(&mut query, result); return; } - active.finish_task(effect_id, CoordinationEffect::Work(AdmissionKind::Query)); + active.finish_task(effect_id, CoordinationEffect::Work(query._work.kind)); if node_lease.check().is_err() { result = Err(Error::Fenced); fenced = true; diff --git a/crates/cellule-runtime/src/cell/actor/tests.rs b/crates/cellule-runtime/src/cell/actor/tests.rs index ab03ed0c..25a63885 100644 --- a/crates/cellule-runtime/src/cell/actor/tests.rs +++ b/crates/cellule-runtime/src/cell/actor/tests.rs @@ -157,9 +157,10 @@ async fn shutdown_after_release( TaskResult::Deactivated { cell, generation: 1, - reply, + reply: reply.map(DrainReply::Unit), shutdown_drain: false, result, + released: None, }, &pool, &mut cells, @@ -188,6 +189,7 @@ async fn shutdown_after_release( &node_lease, &crate::fleet::telemetry::CellTelemetryHandle::default(), &mut pressure, + &NodeAdmission::default(), &mut movement, &mut permits, &mut generation, diff --git a/crates/cellule-runtime/src/cell/catalog/mod.rs b/crates/cellule-runtime/src/cell/catalog/mod.rs index 5bc3b4a8..7fc72ad1 100644 --- a/crates/cellule-runtime/src/cell/catalog/mod.rs +++ b/crates/cellule-runtime/src/cell/catalog/mod.rs @@ -157,6 +157,7 @@ pub struct CatalogShardScan { pages: Vec, next_page: usize, previous: Option, + token: Option, } impl CatalogShardScan { @@ -283,6 +284,12 @@ impl CellCatalog { self.application } + /// Returns the tenant every entry is scoped to. + #[must_use] + pub const fn tenant(&self) -> TenantId { + self.tenant + } + pub(crate) fn matches_identity( &self, identity: crate::cell::application::ApplicationIdentity, @@ -384,8 +391,14 @@ impl CellCatalog { /// Pins one shard head for bounded immutable-page iteration. pub async fn scan_shard(&self, shard: u8) -> Result { let observed = self.load_head(shard).await?; - let (revision, pages) = observed - .map(|observed| (observed.head.revision, observed.head.pages)) + let (revision, pages, token) = observed + .map(|observed| { + ( + observed.head.revision, + observed.head.pages, + Some(observed.token), + ) + }) .unwrap_or_default(); Ok(CatalogShardScan { catalog: self.clone(), @@ -394,6 +407,7 @@ impl CellCatalog { pages, next_page: 0, previous: None, + token, }) } @@ -734,5 +748,8 @@ impl CellCatalog { } mod codec; +mod scan; + +pub use scan::{CatalogScan, CatalogScanReceipt}; use codec::*; diff --git a/crates/cellule-runtime/src/cell/catalog/scan.rs b/crates/cellule-runtime/src/cell/catalog/scan.rs new file mode 100644 index 00000000..755feff4 --- /dev/null +++ b/crates/cellule-runtime/src/cell/catalog/scan.rs @@ -0,0 +1,158 @@ +//! Complete, bounded traversal through the ordinary verified shard reader. + +use super::*; + +/// Streaming traversal of all 256 heads captured before the first page read. +/// +/// Dropping or failing a traversal supplies no completion receipt. Each page +/// uses the same identity, digest and locator checks as `scan_shard`. +pub struct CatalogScan { + catalog: CellCatalog, + shards: Vec, + next_shard: usize, + entry_limit: usize, + entries: usize, + failed: bool, +} + +/// Complete verified traversal and its original tenant/application head set. +/// +/// Heads are observed and rechecked sequentially, not in a global transaction. +/// This is neither a durable pin nor proof of a complete fleet application set, +/// process joining, Cell authority, availability or successor serving. Callers +/// supply those barriers and retain the delivered entries before using this +/// receipt in an operation. Later provisioning can invalidate the observation. +pub struct CatalogScanReceipt { + catalog: CellCatalog, + shards: Vec, + entries: usize, +} + +impl CellCatalog { + /// Captures every shard head for a bounded complete streaming traversal. + /// + /// The nonzero row limit is checked before I/O. Heads retain at most 256 + /// locators each; only one verified page of at most 256 entries is returned + /// per call. No task, authority mutation or storage listing is started. + pub async fn scan_all(&self, entry_limit: usize) -> Result { + if entry_limit == 0 || entry_limit > MAX_ENTRIES * 256 { + return Err(Error::Capacity("invalid complete catalog scan row limit")); + } + let mut shards = Vec::with_capacity(256); + for shard in 0..=u8::MAX { + shards.push(self.scan_shard(shard).await?); + } + Ok(CatalogScan { + catalog: self.clone(), + shards, + next_shard: 0, + entry_limit, + entries: 0, + failed: false, + }) + } +} + +impl CatalogScan { + /// Reads the next immutable page in global Cell order, skipping empty heads. + /// + /// Any read, verification or limit failure permanently prevents completion. + /// Already delivered pages are partial observations, not a complete set. + pub async fn next_page(&mut self) -> Result> { + if self.failed { + return Err(Error::Catalog("complete catalog scan previously failed")); + } + while let Some(shard) = self.shards.get_mut(self.next_shard) { + match shard.next_page().await { + Ok(Some(page)) => { + if page.entries.len() > self.entry_limit - self.entries { + self.failed = true; + return Err(Error::Capacity("complete catalog scan exceeds row limit")); + } + self.entries += page.entries.len(); + return Ok(Some(page)); + } + Ok(None) => self.next_shard += 1, + Err(error) => { + self.failed = true; + return Err(error); + } + } + } + Ok(None) + } + + /// Consumes an exhausted scan and rechecks every original head, including + /// absence and ETag. Partial and failed scans cannot yield a receipt. + pub async fn finish(self) -> Result { + if self.failed || self.next_shard != self.shards.len() { + return Err(Error::Catalog("complete catalog scan is not exhausted")); + } + let receipt = CatalogScanReceipt { + catalog: self.catalog, + shards: self.shards, + entries: self.entries, + }; + receipt.revalidate().await?; + Ok(receipt) + } +} + +impl CatalogScanReceipt { + /// Returns the original tenant scope. + #[must_use] + pub const fn tenant(&self) -> TenantId { + self.catalog.tenant + } + + /// Returns the original application scope. + #[must_use] + pub const fn application(&self) -> ApplicationId { + self.catalog.application + } + + /// Returns the number of entries delivered by the complete traversal. + #[must_use] + pub const fn entry_count(&self) -> usize { + self.entries + } + + /// Returns the captured revision of one shard; zero records an absent head. + #[must_use] + pub fn revision(&self, shard: u8) -> u64 { + self.shards[usize::from(shard)].revision() + } + + /// Returns the original ordered immutable page digests of one shard. + #[must_use] + pub fn page_digests(&self, shard: u8) -> Vec { + self.shards[usize::from(shard)].page_digests() + } + + /// Rechecks the captured heads through the same original catalog adapter. + /// + /// No retry or refresh substitutes a newer set for the original capture. + /// Success supplies sequential observations, not an atomic fleet barrier. + pub async fn revalidate(&self) -> Result<()> { + for original in &self.shards { + let current = self.catalog.load_head(original.shard).await?; + let matches = match current { + None => original.token.is_none(), + Some(current) => { + original.token.as_ref() == Some(¤t.token) + && original.revision == current.head.revision + && original.pages.len() == current.head.pages.len() + && original + .pages + .iter() + .zip(¤t.head.pages) + .all(|(a, b)| a.digest == b.digest && a.first == b.first) + } + }; + if !matches { + return Err(Error::Catalog("complete catalog scan head changed")); + } + } + Ok(()) + } +} diff --git a/crates/cellule-runtime/src/cell/executor/mod.rs b/crates/cellule-runtime/src/cell/executor/mod.rs index e8f50e5f..4ec8b55d 100644 --- a/crates/cellule-runtime/src/cell/executor/mod.rs +++ b/crates/cellule-runtime/src/cell/executor/mod.rs @@ -6,7 +6,6 @@ use cellule_ltx::{CaptureBatch, Db, TransactionError, rusqlite::OptionalExtensio use crate::cell::catalog::CatalogRole; use crate::identity::{CellId, Digest}; use crate::identity::{IncarnationId, RequestId}; -use crate::primitives::maintenance::PersistedWorkInventory; use crate::primitives::maintenance::TransferWorkInventory; use crate::{Error, Result}; @@ -699,16 +698,16 @@ impl CellExecutor { self.db.hydration().map_err(Into::into) } - /// Reads the durable work classes that can block safe owner release. - pub(crate) fn persisted_work_inventory( + pub(crate) fn transfer_work_inventory( &mut self, role: CatalogRole, - ) -> Result { + now_ms: i64, + ) -> Result { if self.fenced { return Err(Error::Fenced); } let result = self.db.query_with(|connection| { - crate::primitives::maintenance::inspect_persisted_work(connection, role) + crate::primitives::maintenance::inspect_transfer_work(connection, role, now_ms) }); if let Some(error) = self.db.take_io_error() { self.fenced = true; @@ -728,16 +727,41 @@ impl CellExecutor { } } - pub(crate) fn transfer_work_inventory( + /// Captures bounded worker diagnostics without scanning application rows. + pub(crate) fn fleet_inventory( &mut self, role: CatalogRole, now_ms: i64, - ) -> Result { + ) -> Result { if self.fenced { return Err(Error::Fenced); } let result = self.db.query_with(|connection| { - crate::primitives::maintenance::inspect_transfer_work(connection, role, now_ms) + let persisted_work = + crate::primitives::maintenance::inspect_persisted_work(connection, role)?; + let transfer_work = + crate::primitives::maintenance::inspect_transfer_work(connection, role, now_ms)?; + let maintenance_work = + crate::primitives::maintenance_readiness::inspect(connection, role, now_ms)?; + let pages: u64 = connection.query_row("PRAGMA page_count", [], |row| row.get(0))?; + let page_size: u64 = connection.query_row("PRAGMA page_size", [], |row| row.get(0))?; + let commit_sequence: u64 = connection.query_row( + "SELECT commit_sequence FROM sys_meta WHERE singleton = 1", + [], + |row| row.get(0), + )?; + let database_bytes = pages + .checked_mul(page_size) + .filter(|bytes| *bytes > 0) + .ok_or(Error::Capacity("invalid measured Cell database size"))?; + Ok(crate::cell::worker::WorkerCellInventory { + persisted_work, + transfer_work, + maintenance_work, + database_bytes, + commit_sequence, + observed_at_ms: now_ms, + }) }); if let Some(error) = self.db.take_io_error() { self.fenced = true; diff --git a/crates/cellule-runtime/src/cell/worker/inventory.rs b/crates/cellule-runtime/src/cell/worker/inventory.rs new file mode 100644 index 00000000..32bddfcd --- /dev/null +++ b/crates/cellule-runtime/src/cell/worker/inventory.rs @@ -0,0 +1,13 @@ +//! One worker-serialized measurement of durable work and logical database size. + +use crate::primitives::maintenance::{PersistedWorkInventory, TransferWorkInventory}; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(crate) struct WorkerCellInventory { + pub(crate) persisted_work: PersistedWorkInventory, + pub(crate) transfer_work: TransferWorkInventory, + pub(crate) maintenance_work: crate::primitives::maintenance_readiness::MaintenanceWorkInventory, + pub(crate) database_bytes: u64, + pub(crate) commit_sequence: u64, + pub(crate) observed_at_ms: i64, +} diff --git a/crates/cellule-runtime/src/cell/worker/mod.rs b/crates/cellule-runtime/src/cell/worker/mod.rs index dbec5460..9211d2f4 100644 --- a/crates/cellule-runtime/src/cell/worker/mod.rs +++ b/crates/cellule-runtime/src/cell/worker/mod.rs @@ -25,15 +25,16 @@ use crate::fleet::resource::{ }; use crate::identity::{CellId, Digest}; use crate::primitives::effects::InboxDelivery; -use crate::primitives::maintenance::PersistedWorkInventory; use crate::primitives::maintenance::TransferWorkInventory; use crate::registry::MigrationPlan; use crate::{Error, Result}; +mod inventory; mod run; +pub(crate) use inventory::WorkerCellInventory; const MAX_WORKERS: usize = 16; -const MAX_ACTIVE_CELLS: usize = 10_000; +pub(crate) const MAX_ACTIVE_CELLS: usize = 10_000; const WORKER_QUEUE: usize = 256; const DEFAULT_PAGE_IO_DEADLINE: Duration = Duration::from_secs(30); @@ -287,6 +288,10 @@ impl SqlWorkerPool { } /// Opens and verifies one exact immutable root on its assigned SQL worker. + #[expect( + clippy::too_many_arguments, + reason = "restore transfers both lasting Cell and optional prepaid job ownership" + )] pub(crate) async fn activate_restored( &self, cell: CellId, @@ -296,22 +301,28 @@ impl SqlWorkerPool { schema: u32, root: cellule_ltx::RootRef, reservation: CellReservation, + job: Option, ) -> Result<()> { let (reply, response) = oneshot::channel(); - self.send( + let command = WorkerCommand::ActivateRestored { cell, - WorkerCommand::ActivateRestored { - cell, - database: Box::new(database), - destination, - incarnation, - schema, - root, - reservation, - reply, - }, - ) - .await?; + database: Box::new(database), + destination, + incarnation, + schema, + root, + reservation, + reply, + }; + let reservation = match job { + Some(reservation) => reservation, + None => self.reserve_job(cell).await?, + }; + let command = WorkerCommand::Reserved { + command: Box::new(command), + reservation, + }; + self.send(cell, command).await?; receive(response).await } @@ -548,30 +559,41 @@ impl SqlWorkerPool { receive(response).await } - pub(crate) async fn persisted_work_inventory( + pub(crate) async fn transfer_work_inventory( &self, cell: CellId, role: CatalogRole, - ) -> Result { + now_ms: i64, + ) -> Result { let (reply, response) = oneshot::channel(); - self.send_worker_job(cell, WorkerCommand::PersistedWork { cell, role, reply }) - .await?; + self.send_worker_job( + cell, + WorkerCommand::TransferWork { + cell, + role, + now_ms, + reply, + }, + ) + .await?; receive(response).await } - pub(crate) async fn transfer_work_inventory( + pub(crate) async fn fleet_inventory( &self, cell: CellId, role: CatalogRole, now_ms: i64, - ) -> Result { + deadline: SqlDeadline, + ) -> Result { let (reply, response) = oneshot::channel(); self.send_worker_job( cell, - WorkerCommand::TransferWork { + WorkerCommand::FleetInventory { cell, role, now_ms, + deadline, reply, }, ) @@ -912,6 +934,26 @@ impl SqlWorkerPool { self.job_reservation(permit.map_err(|_| Error::RuntimeClosed)?) } + /// Holds the incoming Cell's actual affine worker without waiting. + pub(crate) fn try_reserve_job(&self, cell: CellId) -> Result { + let permits = { + let lifecycle = self + .inner + .lifecycle + .lock() + .map_err(|_| Error::RuntimeClosed)?; + if lifecycle.closing { + return Err(Error::RuntimeClosed); + } + Arc::clone(&self.inner.worker_permits[worker_index(cell, self.inner.worker_count)]) + }; + let permit = permits.try_acquire_owned().map_err(|error| match error { + tokio::sync::TryAcquireError::Closed => Error::RuntimeClosed, + tokio::sync::TryAcquireError::NoPermits => Error::Capacity("incoming Cell worker"), + })?; + self.job_reservation(permit) + } + async fn reserve_job(&self, cell: CellId) -> Result { let worker_permits = { let lifecycle = self @@ -955,6 +997,13 @@ impl SqlWorkerPool { } pub(crate) fn reserve_activation(&self) -> Result { + self.reserve_activation_cost(ResourceCost::active_cell()) + } + + pub(crate) fn reserve_activation_cost(&self, cost: ResourceCost) -> Result { + if cost.active_cells() != 1 { + return Err(Error::Capacity("activation must reserve exactly one Cell")); + } let lifecycle = self .inner .lifecycle @@ -966,7 +1015,7 @@ impl SqlWorkerPool { let reservation = self .inner .resources - .try_reserve(ResourceCost::active_cell()) + .try_reserve(cost) .map_err(|error| match error { Error::Capacity(_) => Error::Capacity("active Cells per node"), error => error, @@ -1112,16 +1161,18 @@ enum WorkerCommand { cell: CellId, reply: oneshot::Sender>>, }, - PersistedWork { + TransferWork { cell: CellId, role: CatalogRole, - reply: oneshot::Sender>, + now_ms: i64, + reply: oneshot::Sender>, }, - TransferWork { + FleetInventory { cell: CellId, role: CatalogRole, now_ms: i64, - reply: oneshot::Sender>, + deadline: SqlDeadline, + reply: oneshot::Sender>, }, Resolve { cell: CellId, diff --git a/crates/cellule-runtime/src/cell/worker/run.rs b/crates/cellule-runtime/src/cell/worker/run.rs index 7afd098a..06779e33 100644 --- a/crates/cellule-runtime/src/cell/worker/run.rs +++ b/crates/cellule-runtime/src/cell/worker/run.rs @@ -58,7 +58,7 @@ fn run_worker_command( incarnation, schema, root, - reservation, + reservation: cell_reservation, reply, } => { let result = match cells.entry(cell) { @@ -72,10 +72,11 @@ fn run_worker_command( .map(|executor| { entry.insert(ActiveCell { executor, - _reservation: reservation, + _reservation: cell_reservation, }); }), }; + drop(reservation.take()); let _ = reply.send(result); } WorkerCommand::Bootstrap(bootstrap) => { @@ -262,24 +263,29 @@ fn run_worker_command( .and_then(|active| active.executor.hydration()); let _ = reply.send(result); } - WorkerCommand::PersistedWork { cell, role, reply } => { + WorkerCommand::TransferWork { + cell, + role, + now_ms, + reply, + } => { let result = cells .get_mut(&cell) .ok_or(Error::CellNotActive) - .and_then(|active| active.executor.persisted_work_inventory(role)); + .and_then(|active| active.executor.transfer_work_inventory(role, now_ms)); drop(reservation.take()); let _ = reply.send(result); } - WorkerCommand::TransferWork { + WorkerCommand::FleetInventory { cell, role, now_ms, + deadline, reply, } => { - let result = cells - .get_mut(&cell) - .ok_or(Error::CellNotActive) - .and_then(|active| active.executor.transfer_work_inventory(role, now_ms)); + let result = run_native_callback(cells, cell, deadline, |active| { + active.executor.fleet_inventory(role, now_ms) + }); drop(reservation.take()); let _ = reply.send(result); } diff --git a/crates/cellule-runtime/src/cell/worker/tests/hydration.rs b/crates/cellule-runtime/src/cell/worker/tests/hydration.rs index d57fd61d..798cf8e7 100644 --- a/crates/cellule-runtime/src/cell/worker/tests/hydration.rs +++ b/crates/cellule-runtime/src/cell/worker/tests/hydration.rs @@ -26,6 +26,7 @@ async fn cancelled_hydration_releases_fetch_bytes_without_installing_pages() { 1, cold.root, pool.reserve_activation().unwrap(), + None, ) .await .unwrap(); @@ -112,6 +113,7 @@ async fn expired_worker_deadline_preserves_sparse_cell_for_retry() { 1, fixture.root, pool.reserve_activation().unwrap(), + None, ) .await .unwrap(); diff --git a/crates/cellule-runtime/src/cell/worker/tests/inventory.rs b/crates/cellule-runtime/src/cell/worker/tests/inventory.rs new file mode 100644 index 00000000..61593be5 --- /dev/null +++ b/crates/cellule-runtime/src/cell/worker/tests/inventory.rs @@ -0,0 +1,88 @@ +use super::*; + +#[tokio::test(flavor = "multi_thread")] +async fn fleet_probe_measures_logical_sparse_size_and_preserves_errors_and_job_accounting() { + let fixture = sparse_activation(21, Store::new(Arc::new(InMemory::new())), 262_144).await; + let expected = + u64::from(fixture.database.page_count()) * u64::from(fixture.database.page_size()); + let pool = SqlWorkerPool::new(1, 1).unwrap(); + pool.configure_retained_capacity(16 << 20).unwrap(); + let cell = fixture.cell; + pool.activate_restored( + cell, + RestoredDatabase::Paged(Box::new(fixture.database)), + fixture.destination, + fixture.incarnation, + 1, + fixture.root, + pool.reserve_activation().unwrap(), + None, + ) + .await + .unwrap(); + let sample = pool + .fleet_inventory( + cell, + CatalogRole::Application, + 10, + SqlDeadline::new(Instant::now() + Duration::from_secs(5)), + ) + .await + .unwrap(); + assert_eq!(sample.database_bytes, expected); + assert!(sample.database_bytes >= 262_144); + assert_eq!(sample.commit_sequence, 0); + assert_eq!(sample.observed_at_ms, 10); + assert!(sample.transfer_work.is_settled()); + assert!(sample.persisted_work.is_empty()); + assert_eq!( + pool.resource_ledger() + .snapshot() + .unwrap() + .used + .worker_jobs(), + 0 + ); + assert!(matches!( + pool.fleet_inventory( + cell, + CatalogRole::Application, + 11, + SqlDeadline::new(Instant::now() - Duration::from_secs(1)) + ) + .await, + Err(Error::Deadline) + )); + let schema_error = pool + .fleet_inventory( + cell, + CatalogRole::Queue, + 12, + SqlDeadline::new(Instant::now() + Duration::from_secs(5)), + ) + .await + .unwrap_err(); + assert!(matches!(schema_error, Error::Sqlite(_))); + assert!(std::error::Error::source(&schema_error).is_some()); + assert_eq!( + pool.resource_ledger() + .snapshot() + .unwrap() + .used + .worker_jobs(), + 0 + ); + pool.fence(cell).await.unwrap(); + assert!(matches!( + pool.fleet_inventory( + cell, + CatalogRole::Application, + 13, + SqlDeadline::new(Instant::now() + Duration::from_secs(5)) + ) + .await, + Err(Error::Fenced) + )); + pool.discard(cell).await.unwrap(); + pool.shutdown().await.unwrap(); +} diff --git a/crates/cellule-runtime/src/cell/worker/tests/mod.rs b/crates/cellule-runtime/src/cell/worker/tests/mod.rs index 0366a9ad..67d2f4cb 100644 --- a/crates/cellule-runtime/src/cell/worker/tests/mod.rs +++ b/crates/cellule-runtime/src/cell/worker/tests/mod.rs @@ -9,6 +9,8 @@ use object_store::{ }; use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; +mod inventory; + struct SparseActivation { _source: tempfile::TempDir, _destination: tempfile::TempDir, @@ -132,6 +134,7 @@ async fn sparse_fault_pool_progresses_under_saturated_sql_workers() { 1, first.root, first_reservation, + None, ), pool.activate_restored( second.cell, @@ -141,6 +144,7 @@ async fn sparse_fault_pool_progresses_under_saturated_sql_workers() { 1, second.root, second_reservation, + None, ), ) }) @@ -223,6 +227,7 @@ async fn background_hydration_leaves_both_workers_query_admission_available() { 1, activation.root, pool.reserve_activation().unwrap(), + None, ) .await .unwrap(); @@ -362,6 +367,7 @@ async fn rustfs_worker_interference(worker_count: usize) { 1, activation.root, pool.reserve_activation().unwrap(), + None, ) .await .unwrap(); @@ -473,6 +479,7 @@ async fn rustfs_worker_interference(worker_count: usize) { 1, cold.root, pool.reserve_activation().unwrap(), + None, ) .await .unwrap(); diff --git a/crates/cellule-runtime/src/client/local.rs b/crates/cellule-runtime/src/client/local.rs index 140c16db..9446bfe5 100644 --- a/crates/cellule-runtime/src/client/local.rs +++ b/crates/cellule-runtime/src/client/local.rs @@ -53,13 +53,21 @@ impl CellTransport for LocalCellTransport { let owner_fence = handle.owner_fence(); let input_bytes = command.input.len(); let output_limit = command.output_limit as usize; + let lease_completion = registry.command_is_lease_completion( + command.module, + command.operation_id, + command.codec_version, + )?; handle - .execute( - command.identity, - command.operation_digest, - command.now_ms, - input_bytes, - output_limit, + .execute_registered( + lease_completion, + crate::cell::actor::CommandWork { + identity: command.identity, + operation_digest: command.operation_digest, + now_ms: command.now_ms, + operation_bytes: input_bytes, + max_result_bytes: output_limit, + }, move |transaction| { let (sequence, now_ms) = next_metadata(transaction, command.now_ms)?; let started = Instant::now(); @@ -112,35 +120,45 @@ impl CellTransport for LocalCellTransport { let schema = handle.schema(); let input_bytes = query.input.len(); let output_limit = query.output_limit as usize; + let lease_validation = registry.query_is_lease_validation( + query.module, + query.operation_id, + query.codec_version, + )?; let output = handle - .query(input_bytes, output_limit, move |connection| { - let (commit_sequence, now_ms) = current_metadata(connection, query.now_ms)?; - observed_sequence.store(commit_sequence, Ordering::Release); - let started = Instant::now(); - let result = registry.execute_query( - connection, - QueryInvocation { - module: query.module, - operation_id: query.operation_id, - codec_version: query.codec_version, - schema, - cell, - commit_sequence, - now_ms, - input: &query.input, - }, - ); - telemetry.primitive_operation( - query.module, - PrimitiveOperationKind::Query, - match &result { - Ok(_) => PrimitiveOperationOutcome::Success, - Err(_) => PrimitiveOperationOutcome::Failed, - }, - started.elapsed(), - ); - result - }) + .query_registered( + lease_validation, + input_bytes, + output_limit, + move |connection| { + let (commit_sequence, now_ms) = current_metadata(connection, query.now_ms)?; + observed_sequence.store(commit_sequence, Ordering::Release); + let started = Instant::now(); + let result = registry.execute_query( + connection, + QueryInvocation { + module: query.module, + operation_id: query.operation_id, + codec_version: query.codec_version, + schema, + cell, + commit_sequence, + now_ms, + input: &query.input, + }, + ); + telemetry.primitive_operation( + query.module, + PrimitiveOperationKind::Query, + match &result { + Ok(_) => PrimitiveOperationOutcome::Success, + Err(_) => PrimitiveOperationOutcome::Failed, + }, + started.elapsed(), + ); + result + }, + ) .await?; let commit_sequence = sequence.load(Ordering::Acquire); if query diff --git a/crates/cellule-runtime/src/client/mod.rs b/crates/cellule-runtime/src/client/mod.rs index 4b3fbad6..b6a30219 100644 --- a/crates/cellule-runtime/src/client/mod.rs +++ b/crates/cellule-runtime/src/client/mod.rs @@ -54,7 +54,7 @@ mod replica; mod routing; mod runtime; -pub use replica::CellReadReplica; +pub use replica::{CellReadReplica, ReadReplicaLifecycleObservation, ReadReplicaSource}; pub use routing::ReplicaReadRouter; pub use local::command_operation_digest; diff --git a/crates/cellule-runtime/src/client/replica.rs b/crates/cellule-runtime/src/client/replica/mod.rs similarity index 53% rename from crates/cellule-runtime/src/client/replica.rs rename to crates/cellule-runtime/src/client/replica/mod.rs index f97cd83f..85d51b0b 100644 --- a/crates/cellule-runtime/src/client/replica.rs +++ b/crates/cellule-runtime/src/client/replica/mod.rs @@ -2,23 +2,29 @@ use std::{ path::Path, - sync::Arc, + sync::{ + Arc, + atomic::{AtomicUsize, Ordering}, + }, time::{Duration, Instant}, }; use cellule_ltx::{CellReplica, ReadOnlyRoot}; -use tokio::sync::{Mutex, RwLock, Semaphore}; +use tokio::sync::{Mutex, Notify, RwLock, Semaphore}; use super::*; use crate::cell::actor::CellRuntime; use crate::control::authority::CellAuthority; -use crate::control::{Control, ControlState, Owner}; +use crate::control::{Control, ControlState, Owner, RootRef}; use crate::fleet::resource::ResourceReservation; use crate::identity::SessionId; use crate::node::NodeDirectory; const QUERY_DEADLINE: Duration = Duration::from_secs(5); +mod observation; +pub use observation::ReadReplicaLifecycleObservation; + /// One immutable replica snapshot that serves explicit, position-tagged reads. /// /// The caller owns routing and authorization. Every successful @@ -33,17 +39,150 @@ pub struct CellReadReplica { replica: CellReplica, target: CellTarget, expected: CellDescription, - snapshot: Arc>, + snapshot: Arc>, + lifetime: Arc, refresh_gate: Arc>, query_gate: Arc, } +struct ReplicaState { + receipt: Receipt, + snapshot: Option>, +} + +/// Opaque exact reader source observed through canonical Cell authority. +/// +/// Preparing this value creates no local reader or resource reservation. An +/// application can journal its exact responsibility before opening the view. +/// It does not grant admission or prove current serving; opening checks the +/// owner again and verifies every dependency of this pinned root. #[derive(Clone)] +pub struct ReadReplicaSource { + target: CellTarget, + description: CellDescription, + owner: Owner, + epoch: u64, + root: RootRef, + node: crate::identity::NodeId, + fleet: Digest, +} + +impl ReadReplicaSource { + /// Returns the original Cell target. + #[must_use] + pub fn target(&self) -> &CellTarget { + &self.target + } + /// Returns the catalog scope, code and schema of the observed source. + #[must_use] + pub const fn description(&self) -> CellDescription { + self.description + } + /// Returns the original owner boot and endpoint. + #[must_use] + pub fn owner(&self) -> &Owner { + &self.owner + } + /// Returns the original authority epoch. + #[must_use] + pub const fn epoch(&self) -> u64 { + self.epoch + } + /// Returns the exact authority-pinned immutable root to open. + #[must_use] + pub fn root(&self) -> &RootRef { + &self.root + } + /// Returns the physical source node from its verified boot advertisement. + #[must_use] + pub const fn node(&self) -> crate::identity::NodeId { + self.node + } + /// Returns the source boot's signed fleet scope. + #[must_use] + pub const fn fleet(&self) -> Digest { + self.fleet + } +} + struct ReplicaSnapshot { owner: Owner, epoch: u64, + node: crate::identity::NodeId, + fleet: Digest, view: Arc, _admission: Arc, + // Fields drop in declaration order: the last root's resource charges must + // be released before its lifetime wakes a close waiter. + _lifetime: LifetimeGuard, +} + +struct SnapshotInputs { + owner: Owner, + epoch: u64, + node: crate::identity::NodeId, + fleet: Digest, + admission: Arc, + operation: Arc, +} + +struct ReaderOpening { + source: ReadReplicaSource, + admission: Arc, +} + +const CLOSED: usize = 1 << (usize::BITS - 1); + +#[derive(Default)] +struct ReplicaLifetime { + // Closure and admission share one CAS word: a close waiter cannot observe + // zero and then miss a newly accepted operation. Snapshots are retained by + // an already accepted open/refresh; their jobs own the same operation guard. + state: AtomicUsize, + changed: Notify, +} + +struct LifetimeGuard(Arc); + +impl ReplicaLifetime { + fn acquire(self: &Arc, new_operation: bool) -> Result { + self.state + .try_update(Ordering::AcqRel, Ordering::Acquire, |state| { + let count = state & !CLOSED; + if (state & CLOSED != 0 && (new_operation || count == 0)) || count == CLOSED - 1 { + None + } else { + Some(state + 1) + } + }) + .map_err(|state| { + if state & CLOSED != 0 && (new_operation || state & !CLOSED == 0) { + Error::Fenced + } else { + Error::Capacity("read replica lifetime count") + } + })?; + Ok(LifetimeGuard(Arc::clone(self))) + } + + async fn join(&self) { + loop { + let changed = self.changed.notified(); + tokio::pin!(changed); + changed.as_mut().enable(); + if self.state.load(Ordering::Acquire) & !CLOSED == 0 { + return; + } + changed.await; + } + } +} + +impl Drop for LifetimeGuard { + fn drop(&mut self) { + self.0.state.fetch_sub(1, Ordering::AcqRel); + self.0.changed.notify_waiters(); + } } impl CellReadReplica { @@ -69,8 +208,32 @@ impl CellReadReplica { target: CellTarget, destination: &Path, ) -> Result { + runtime.node_admission().check_new_role()?; + // Ordinary opening retains its existing admission-before-provider-I/O + // order. Explicit source preparation alone creates no local obligation. let admission = Arc::new(runtime.reserve_read_view()?); - let replica = runtime.replica_for_read(replica); + let source = Self::prepare_source(®istry, &authority, &directory, target).await?; + Self::open_admitted( + runtime, + registry, + authority, + directory, + replica, + ReaderOpening { source, admission }, + destination, + ) + .await + } + + /// Observes an exact source without opening a view or reserving resources. + /// Journal enrollment before calling `open_source`; a later publication + /// cannot silently replace the root named by this value. + pub async fn prepare_source( + registry: &Registry, + authority: &CellAuthority, + directory: &NodeDirectory, + target: CellTarget, + ) -> Result { let cell = target.cell_id(); let observed = authority.load(cell).await?.ok_or(Error::CellNotActive)?; let control = observed.value(); @@ -84,27 +247,104 @@ impl CellReadReplica { if !registry.supports_module_code(module, control.code, control.schema) { return Err(Error::Registry("replica module or code is unsupported")); } - if !directory.is_live(owner.session, unix_time_ms()?).await? { - return Err(Error::Fenced); + let boot = directory + .load_if_live(owner.session, unix_time_ms()?) + .await? + .ok_or(Error::Fenced)?; + Ok(ReadReplicaSource { + target, + description: CellDescription { + cell, + incarnation: control.incarnation, + code: control.code, + schema: control.schema, + }, + owner, + epoch: control.epoch, + root: control.root.clone().ok_or(Error::Fenced)?, + node: boot.advertisement().node(), + fleet: boot.advertisement().fleet(), + }) + } + + /// Opens a previously checked exact source through the ordinary read path. + /// + /// The same owner/epoch must still be serving at installation. Newer roots + /// under that owner are allowed, but this view opens the original pinned + /// root. Cordon and closed runtime admission still reject a new view. + pub async fn open_source( + runtime: CellRuntime, + registry: Arc, + authority: CellAuthority, + directory: NodeDirectory, + replica: CellReplica, + source: ReadReplicaSource, + destination: &Path, + ) -> Result { + runtime.node_admission().check_new_role()?; + let admission = Arc::new(runtime.reserve_read_view()?); + Self::open_admitted( + runtime, + registry, + authority, + directory, + replica, + ReaderOpening { source, admission }, + destination, + ) + .await + } + + async fn open_admitted( + runtime: CellRuntime, + registry: Arc, + authority: CellAuthority, + directory: NodeDirectory, + replica: CellReplica, + opening: ReaderOpening, + destination: &Path, + ) -> Result { + runtime.node_admission().check_new_role()?; + let ReaderOpening { source, admission } = opening; + let ReadReplicaSource { + target, + description: expected, + owner, + epoch, + root, + node, + fleet, + } = source; + let (module, _) = registry + .namespace_contract(target.namespace()) + .ok_or(Error::Registry("replica namespace is not registered"))?; + if !registry.supports_module_code(module, expected.code, expected.schema) { + return Err(Error::Registry("replica module or code is unsupported")); } - let root = control.ltx_root().ok_or(Error::Fenced)?; - let verified = replica.open_root(&root).await?; - if verified.schema() != control.schema { + let lifetime = Arc::new(ReplicaLifetime::default()); + let operation = Arc::new(lifetime.acquire(true)?); + let replica = runtime.replica_for_read(replica); + let verified = replica + .open_root(&root.to_ltx(expected.cell, expected.incarnation)) + .await?; + if verified.schema() != expected.schema { return Err(Error::Fenced); } - let expected = CellDescription { - cell, - incarnation: control.incarnation, - code: control.code, - schema: control.schema, - }; - let view = open_view(&runtime, verified, destination, Arc::clone(&admission)).await?; - let snapshot = ReplicaSnapshot { - owner, - epoch: control.epoch, - view, - _admission: admission, - }; + let snapshot = open_view( + &runtime, + verified, + destination, + SnapshotInputs { + owner, + epoch, + node, + fleet, + admission, + operation, + }, + ) + .await?; + let position = receipt(expected, snapshot.view.root().commit_sequence); let opened = Self { runtime, registry, @@ -113,7 +353,11 @@ impl CellReadReplica { replica, target, expected, - snapshot: Arc::new(RwLock::new(snapshot.clone())), + snapshot: Arc::new(RwLock::new(ReplicaState { + receipt: position, + snapshot: Some(snapshot.clone()), + })), + lifetime, refresh_gate: Arc::new(Mutex::new(())), query_gate: Arc::new(Semaphore::new(1)), }; @@ -121,11 +365,10 @@ impl CellReadReplica { Ok(opened) } - /// Returns the exact snapshot position this reader serves. + /// Returns the last installed exact snapshot position, including after close. #[must_use] pub async fn receipt(&self) -> Receipt { - let snapshot = self.snapshot.read().await; - self.snapshot_receipt(&snapshot) + self.snapshot.read().await.receipt } /// Returns the verified position and whether the original owner is still live. @@ -134,28 +377,67 @@ impl CellReadReplica { /// query or takeover. Changed authority or closed admission rejects it. pub async fn readiness(&self) -> Result<(Receipt, bool)> { self.runtime.ensure_running()?; - let snapshot = self.snapshot.read().await.clone(); + let _operation = self.lifetime.acquire(true)?; + let snapshot = self.current_snapshot().await?; self.confirm_snapshot(&snapshot).await?; - let live = self + let boot = self .directory - .is_live(snapshot.owner.session, unix_time_ms()?) + .load_if_live(snapshot.owner.session, unix_time_ms()?) .await?; + if boot.as_ref().is_some_and(|boot| { + boot.advertisement().node() != snapshot.node + || boot.advertisement().fleet() != snapshot.fleet + }) { + return Err(Error::Fenced); + } + let live = boot.is_some(); + if self.query_gate.is_closed() { + return Err(Error::Fenced); + } Ok((self.snapshot_receipt(&snapshot), live)) } /// Closes reader admission across every clone before eviction or writable activation. pub fn close(&self) { + self.lifetime.state.fetch_or(CLOSED, Ordering::AcqRel); self.query_gate.close(); } + /// Closes admission, detaches snapshots from every retained peer clone, + /// and joins accepted queries, native SQL and refresh work. The returned + /// receipt is the last installed position, not current authority/readiness. + /// A cancelled waiter leaves closure installed; another waiter can join the + /// same retained work. Provider failures still belong to the original work. + pub async fn close_and_join(&self) -> Receipt { + self.close(); + let receipt = { + let _refresh = self.refresh_gate.lock().await; + let mut state = self.snapshot.write().await; + state.snapshot.take(); + state.receipt + }; + self.lifetime.join().await; + receipt + } + + async fn current_snapshot(&self) -> Result> { + self.snapshot + .read() + .await + .snapshot + .clone() + .ok_or(Error::Fenced) + } + /// Installs a newer exact root without disrupting queries using the old view. /// /// The destination must be fresh and private. Concurrent refreshes are /// serialized; a failed or stale refresh leaves the serving view intact. pub async fn refresh(&self, destination: &Path) -> Result { self.runtime.ensure_running()?; + let operation = Arc::new(self.lifetime.acquire(true)?); let _refresh = self.refresh_gate.lock().await; - let current = self.snapshot.read().await.clone(); + let current = self.current_snapshot().await?; self.confirm_authority(¤t).await?; let observed = self .authority @@ -178,15 +460,30 @@ impl CellReadReplica { if verified.schema() != self.expected.schema { return Err(Error::Fenced); } - let replacement = ReplicaSnapshot { - owner: current.owner.clone(), - epoch: current.epoch, - view: open_view(&self.runtime, verified, destination, Arc::clone(&admission)).await?, - _admission: admission, - }; + let replacement = open_view( + &self.runtime, + verified, + destination, + SnapshotInputs { + owner: current.owner.clone(), + epoch: current.epoch, + node: current.node, + fleet: current.fleet, + admission, + operation: Arc::clone(&operation), + }, + ) + .await?; self.confirm_authority(&replacement).await?; let receipt = self.snapshot_receipt(&replacement); - *self.snapshot.write().await = replacement; + let mut state = self.snapshot.write().await; + // Close may have raced the final provider read. A closed reader cannot + // install a replacement behind the detachment barrier. + if self.query_gate.is_closed() { + return Err(Error::Fenced); + } + state.receipt = receipt; + state.snapshot = Some(replacement); Ok(receipt) } @@ -224,6 +521,7 @@ impl CellReadReplica { pub(crate) async fn query_encoded(&self, query: EncodedQuery) -> Result { self.runtime.ensure_running()?; + let accepted_work = Arc::new(self.lifetime.acquire(true)?); if self.query_gate.is_closed() { return Err(Error::Fenced); } @@ -243,7 +541,7 @@ impl CellReadReplica { return Err(Error::Registry("replica query contract changed")); } validate_description(&self.registry, module, self.expected, operation)?; - let snapshot = self.snapshot.read().await.clone(); + let snapshot = self.current_snapshot().await?; let observed = self.snapshot_receipt(&snapshot); if let Some(minimum) = query .minimum @@ -273,7 +571,11 @@ impl CellReadReplica { let schema = self.expected.schema; let sequence = observed.commit_sequence; let now_ms = unix_time_ms()?; + let native_operation = Arc::clone(&accepted_work); let mut task = tokio::task::spawn_blocking(move || { + // Declare this first so cancellation cannot wake close before SQL + // view, job and semaphore charges have all been released. + let _operation = native_operation; let _permit = permit; let _job = job; // Caller cancellation can drop the reader while SQL is running. @@ -365,7 +667,11 @@ impl CellReadReplica { // Either provider read may stall beyond the observed lease. Admission // and expiry must still hold when the result is actually released. let release_ms = unix_time_ms()?; - let live = live.is_some_and(|node| node.advertisement().expires_at_ms() > release_ms); + let live = live.is_some_and(|node| { + node.advertisement().expires_at_ms() > release_ms + && node.advertisement().node() == snapshot.node + && node.advertisement().fleet() == snapshot.fleet + }); if self.query_gate.is_closed() || !self.same_owner_and_code(current.value(), snapshot) || !live @@ -394,20 +700,32 @@ async fn open_view( runtime: &CellRuntime, verified: cellule_ltx::VerifiedRoot, destination: &Path, - admission: Arc, -) -> Result> { + inputs: SnapshotInputs, +) -> Result> { let job = runtime.reserve_sql_job().await?; let destination = destination.to_owned(); // VFS faults need LTX's blocking pool for directory-cache I/O. SQLite must // use separate SQL admission, retaining both charges if its waiter cancels. - tokio::task::spawn_blocking(move || { + tokio::task::spawn_blocking(move || -> Result> { + let operation = inputs.operation; let _job = job; - let _admission = admission; + let admission = inputs.admission; + let owner = inputs.owner; + let checked_root = verified; + let destination_path = destination; cellule_ltx::with_paged_io_deadline(Instant::now() + QUERY_DEADLINE, || { - verified.open_read_only(&destination).map(Arc::new) + let view = Arc::new(checked_root.open_read_only(&destination_path)?); + Ok(Arc::new(ReplicaSnapshot { + owner, + epoch: inputs.epoch, + node: inputs.node, + fleet: inputs.fleet, + view, + _admission: admission, + _lifetime: operation.0.acquire(false)?, + })) }) }) .await .map_err(Error::WorkerJoin)? - .map_err(Error::from) } diff --git a/crates/cellule-runtime/src/client/replica/observation.rs b/crates/cellule-runtime/src/client/replica/observation.rs new file mode 100644 index 00000000..e17ff01d --- /dev/null +++ b/crates/cellule-runtime/src/client/replica/observation.rs @@ -0,0 +1,99 @@ +//! Read-only lifetime diagnostics from the canonical reader admission word. + +use super::*; + +/// Local snapshot and accepted-work closure shared by every reader clone. +/// +/// This is interval evidence, not current Cell authority, durable enrollment +/// retirement, replacement-policy satisfaction or permission to stop a node. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ReadReplicaLifecycleObservation { + receipt: Receipt, + admission_closed: bool, + snapshot_attached: bool, + retained_lifetimes: usize, +} + +impl ReadReplicaLifecycleObservation { + /// Returns the last installed position, including after snapshot detachment. + #[must_use] + pub const fn receipt(&self) -> Receipt { + self.receipt + } + + /// Reports the canonical irreversible closure of new reader operations. + #[must_use] + pub const fn admission_closed(&self) -> bool { + self.admission_closed + } + + /// Reports a view still installed in the state shared by all reader clones. + #[must_use] + pub const fn snapshot_attached(&self) -> bool { + self.snapshot_attached + } + + /// Counts original lifetime guards for snapshots and accepted operations. + /// + /// Accepted native work retains its guard after caller cancellation. This + /// count is neither a query count nor a count of retained handle copies. + #[must_use] + pub const fn retained_lifetimes(&self) -> usize { + self.retained_lifetimes + } + + /// Reports detached shared state and completion of all original lifetimes. + /// + /// Closed plus zero is stable: the same admission CAS rejects new operations + /// and snapshots cannot acquire a guard after the final lifetime is gone. + /// Remote authority, producer retirement and replacement policy still need + /// independent evidence, even when peer handles remain retained locally. + #[must_use] + pub const fn locally_joined(&self) -> bool { + self.admission_closed && !self.snapshot_attached && self.retained_lifetimes == 0 + } +} + +impl CellReadReplica { + /// Observes the original admission and join state without starting new work. + /// + /// The snapshot read lock prevents replacement/detachment during capture; + /// one load reads closure and lifetime count from their shared CAS word. + /// Open counts remain advisory because accepted work may begin afterwards. + /// The observation remains available after close and runtime shutdown. + pub async fn lifecycle_observation(&self) -> ReadReplicaLifecycleObservation { + let snapshot = self.snapshot.read().await; + let lifetime = self.lifetime.state.load(Ordering::Acquire); + ReadReplicaLifecycleObservation { + receipt: snapshot.receipt, + admission_closed: lifetime & CLOSED != 0, + snapshot_attached: snapshot.snapshot.is_some(), + retained_lifetimes: lifetime & !CLOSED, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn closed_zero_cannot_acquire_another_operation_or_snapshot_lifetime() { + let lifetime = Arc::new(ReplicaLifetime::default()); + let operation = lifetime.acquire(true).unwrap(); + let snapshot = lifetime.acquire(false).unwrap(); + assert_eq!(lifetime.state.load(Ordering::Acquire), 2); + lifetime.state.fetch_or(CLOSED, Ordering::AcqRel); + assert!(matches!(lifetime.acquire(true), Err(Error::Fenced))); + // An already accepted open may finish while its original lifetime lives. + let finishing_snapshot = lifetime.acquire(false).unwrap(); + drop(operation); + drop(snapshot); + assert_eq!(lifetime.state.load(Ordering::Acquire), CLOSED | 1); + drop(finishing_snapshot); + assert_eq!(lifetime.state.load(Ordering::Acquire), CLOSED); + assert!(matches!(lifetime.acquire(true), Err(Error::Fenced))); + assert!(matches!(lifetime.acquire(false), Err(Error::Fenced))); + assert_eq!(lifetime.state.load(Ordering::Acquire), CLOSED); + } +} diff --git a/crates/cellule-runtime/src/control/authority/acquisition/codec.rs b/crates/cellule-runtime/src/control/authority/acquisition/codec.rs new file mode 100644 index 00000000..3a06357e --- /dev/null +++ b/crates/cellule-runtime/src/control/authority/acquisition/codec.rs @@ -0,0 +1,51 @@ +use super::*; + +const DOMAIN: &[u8] = b"cellule.cell-acquisition.v1\0"; +impl CellAcquisitionRecord { + pub(super) fn encode(&self) -> Result> { + self.validate()?; + let mut body = DOMAIN.to_vec(); + for control in [&self.input, &self.materialized] { + let bytes = control.encode()?; + // Control's canonical envelope is at most 8 KiB. + body.extend_from_slice(&(bytes.len() as u32).to_be_bytes()); + body.extend_from_slice(&bytes); + } + if body.len() as u64 > MAX_ACQUISITION_BYTES { + return Err(Error::Control("acquisition record exceeds bound")); + } + Ok(body) + } + pub(super) fn decode(body: &[u8]) -> Result { + if body.len() as u64 > MAX_ACQUISITION_BYTES { + return Err(Error::Control("acquisition record exceeds bound")); + } + let mut remaining = body + .strip_prefix(DOMAIN) + .ok_or(Error::Control("unknown acquisition record version"))?; + let mut control = || -> Result { + let prefix = remaining + .get(..4) + .ok_or(Error::Control("truncated acquisition record"))?; + let length = u32::from_be_bytes([prefix[0], prefix[1], prefix[2], prefix[3]]) as usize; + if length > MAX_CONTROL_BYTES as usize { + return Err(Error::Control("acquisition Control exceeds bound")); + } + let bytes = remaining + .get(4..4 + length) + .ok_or(Error::Control("truncated acquisition Control"))?; + let value = Control::decode(bytes)?; + remaining = &remaining[4 + length..]; + Ok(value) + }; + let record = Self { + input: control()?, + materialized: control()?, + }; + if !remaining.is_empty() { + return Err(Error::Control("trailing acquisition record bytes")); + } + record.validate()?; + Ok(record) + } +} diff --git a/crates/cellule-runtime/src/control/authority/acquisition/mod.rs b/crates/cellule-runtime/src/control/authority/acquisition/mod.rs new file mode 100644 index 00000000..7806b7f5 --- /dev/null +++ b/crates/cellule-runtime/src/control/authority/acquisition/mod.rs @@ -0,0 +1,141 @@ +//! Immutable successful-claim input and materialized position before admission. +use super::*; +use crate::control::ControlState; + +mod codec; +mod prefix; +pub use prefix::VerifiedRecoveryPrefix; +#[cfg(test)] +mod tests; + +const MAX_ACQUISITION_BYTES: u64 = 20 * 1024; + +/// Historical input of a canonical acquisition and its materialized control. +/// +/// Only canonical runtime acquisition produces this record, after ownership CAS +/// and optional recovery publication, before actor admission. The record can +/// survive failed activation; it proves neither restore completion, current +/// serving, complete acknowledged-prefix coverage nor maintenance settlement. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct CellAcquisitionRecord { + input: Control, + materialized: Control, +} +impl CellAcquisitionRecord { + /// Exact successful ownership-CAS input, including any recovery overlay. + #[must_use] + pub const fn input(&self) -> &Control { + &self.input + } + /// Canonical claimed/materialized position before actor admission. + #[must_use] + pub const fn materialized(&self) -> &Control { + &self.materialized + } + + fn validate(&self) -> Result<()> { + self.input.encode()?; + self.materialized.encode()?; + let owner = self.materialized.owner.clone().ok_or(Error::Fenced)?; + if !matches!( + self.input.state, + ControlState::Idle | ControlState::Recovering | ControlState::Serving + ) || self + .input + .owner + .as_ref() + .is_some_and(|source| source.session == owner.session) + { + return Err(Error::Fenced); + } + let claimed = self.input.takeover(owner)?; + if claimed.recovery.is_some() { + claimed.validate_transition(&self.materialized, Transition::PublishRecovery)?; + } else if claimed != self.materialized { + return Err(Error::Control( + "acquisition materialization changed its input", + )); + } + Ok(()) + } +} +impl CellAuthority { + /// Reads immutable historical acquisition metadata, never an authority grant. + /// Missing legacy or canceled-before-publication metadata remains `None`. + pub async fn acquisition_record( + &self, + cell: CellId, + incarnation: IncarnationId, + epoch: u64, + ) -> Result> { + if epoch < 2 { + return Err(Error::Control("acquisition epoch precedes takeover")); + } + let path = + self.layout + .acquisition_record_path(cell.as_bytes(), incarnation.as_bytes(), epoch); + let (body, _) = match self + .layout + .store() + .get_with_etag_bounded(&path, MAX_ACQUISITION_BYTES) + .await + { + Ok(value) => value, + Err(StorageError::NotFound { .. }) => return Ok(None), + Err(source) => return Err(source.into()), + }; + let record = CellAcquisitionRecord::decode(&body)?; + if record.materialized.cell != cell + || record.materialized.incarnation != incarnation + || record.materialized.epoch != epoch + { + return Err(Error::Control( + "acquisition record path differs from its scope", + )); + } + Ok(Some(record)) + } + + pub(crate) async fn retain_acquisition( + &self, + input: &Control, + materialized: &Control, + ) -> Result<()> { + let record = CellAcquisitionRecord { + input: input.clone(), + materialized: materialized.clone(), + }; + let body = Bytes::from(record.encode()?); + let cell = materialized.cell; + let incarnation = materialized.incarnation; + let epoch = materialized.epoch; + if let Some(original) = self.acquisition_record(cell, incarnation, epoch).await? { + return if original == record { + Ok(()) + } else { + Err(Error::Control("acquisition history conflicts")) + }; + } + let path = + self.layout + .acquisition_record_path(cell.as_bytes(), incarnation.as_bytes(), epoch); + match self + .layout + .store() + .create_strict_with_etag(&path, body) + .await + { + Ok(_) => Ok(()), + Err(source) => { + // Adopt only the exact original immutable publication. Missing, + // conflicting or failed rereads preserve the original write error. + if let Ok(Some(original)) = self.acquisition_record(cell, incarnation, epoch).await + && original == record + { + return Ok(()); + } + Err(source.into()) + } + } + } +} diff --git a/crates/cellule-runtime/src/control/authority/acquisition/prefix.rs b/crates/cellule-runtime/src/control/authority/acquisition/prefix.rs new file mode 100644 index 00000000..46493af0 --- /dev/null +++ b/crates/cellule-runtime/src/control/authority/acquisition/prefix.rs @@ -0,0 +1,159 @@ +use super::*; +use crate::recovery::manifest::PinnedRecoveryCell; + +/// Exact canonical suffix materialization and verified successor origin graph. +/// +/// The caller obtains the required row from the original sealed-log manifest and +/// authenticates that complete scope and its backend mappings. This observation +/// grants no current ownership, native serving, storage pin or fleet settlement. +pub struct VerifiedRecoveryPrefix { + required: PinnedRecoveryCell, + acquisition_epoch: u64, + proof: VerifiedRootPrefix, +} +impl VerifiedRecoveryPrefix { + /// Exact original manifest row, without replacing any epoch or tail boundary. + #[must_use] + pub const fn required(&self) -> &PinnedRecoveryCell { + &self.required + } + /// Original acquisition that materialized this precise recovery input. + #[must_use] + pub const fn acquisition_epoch(&self) -> u64 { + self.acquisition_epoch + } + /// Materialized prefix and complete current successor origin observation. + #[must_use] + pub const fn proof(&self) -> &VerifiedRootPrefix { + &self.proof + } +} + +impl CellAuthority { + /// Proves materialization of one exact original sealed recovery row, followed + /// by verified derivation and complete current successor origin availability. + /// + /// Binds the original closed owner and exact suffix, then searches bounded + /// canonical acquisitions through the selected successor epoch. Interrupted + /// claims may materialize later; counters or a different overlay cannot + /// substitute. Rechecks selected authority after the origin walk. The caller + /// owns admission and a finite deadline; use the runtime wrapper for shared + /// memory/I/O admission. Missing legacy metadata refuses with a typed error. + pub async fn verify_recovered_prefix( + &self, + required: &PinnedRecoveryCell, + root: cellule_ltx::RootRef, + replica: &cellule_ltx::CellReplica, + limit: usize, + ) -> Result { + if limit == 0 || limit > MAX_LINEAGE_ROOTS { + return Err(Error::Capacity("invalid Cell root lineage traversal bound")); + } + if required.application.as_bytes() != self.layout.application_id() + || required.cell.as_bytes() != &root.cell + || required.incarnation.as_bytes() != &root.incarnation + || replica.scope() != (root.cell, root.incarnation) + || required.cell_epoch == 0 + { + return Err(Error::Control("recovered prefix scope differs")); + } + let first = required + .cell_epoch + .checked_add(1) + .ok_or(Error::Control("recovered acquisition epoch overflow"))?; + let selected = self.load(required.cell).await?.ok_or(Error::Fenced)?; + if selected.value().incarnation != required.incarnation + || selected.value().ltx_root() != Some(root) + || selected.value().state != ControlState::Serving + || selected.value().epoch < first + { + return Err(Error::Fenced); + } + let count = selected.value().epoch - required.cell_epoch; + if count > limit as u64 { + return Err(Error::Capacity( + "recovery acquisition search bound exceeded", + )); + } + let original = self + .owner_observation(required.cell, required.incarnation, required.cell_epoch) + .await? + .ok_or(Error::OwnerHistoryIncomplete { + cell: required.cell, + incarnation: required.incarnation, + epoch: required.cell_epoch, + })?; + if original.recovery.as_ref() != Some(&required.recovery) + || original + .owner + .as_ref() + .is_none_or(|owner| owner.session != required.recovery.leader_session) + { + return Err(Error::Control("original owner recovery scope differs")); + } + // An interrupted claim can retain the same original overlay at + // a later ownership epoch. Choose its last canonical materialization; + // owner/counter changes cannot substitute for the exact original suffix. + let mut materialized = None; + let mut missing = None; + for epoch in first..=selected.value().epoch { + let Some(record) = self + .acquisition_record(required.cell, required.incarnation, epoch) + .await? + else { + missing.get_or_insert(epoch); + continue; + }; + if record.input.recovery.as_ref() != Some(&required.recovery) { + continue; + } + let prefix = record + .materialized + .ltx_root() + .ok_or(Error::Control("recovery materialization lacks root"))?; + if prefix.position.txid != required.recovery.final_txid + || prefix.position.checksum != required.recovery.final_checksum + || prefix.commit_sequence != required.recovery.final_commit_sequence + { + return Err(Error::Control("recovery materialization endpoint differs")); + } + materialized = Some((epoch, prefix)); + } + let (epoch, prefix) = match materialized { + Some(value) => value, + None => { + return Err(match missing { + Some(epoch) => Error::AcquisitionHistoryIncomplete { + cell: required.cell, + incarnation: required.incarnation, + epoch, + }, + None => Error::Control("canonical acquisition recovery input differs"), + }); + } + }; + let proof = self + .verify_root_prefix(prefix, root, replica, limit) + .await?; + let confirmed = self.load(required.cell).await?.ok_or(Error::Fenced)?; + if confirmed.value().owner != selected.value().owner + || confirmed.value().epoch != selected.value().epoch + || confirmed.value().incarnation != required.incarnation + || confirmed.value().state != ControlState::Serving + || confirmed.value().ltx_root() != Some(root) + { + return Err(Error::Fenced); + } + Ok(VerifiedRecoveryPrefix { + required: PinnedRecoveryCell { + application: required.application, + cell: required.cell, + incarnation: required.incarnation, + cell_epoch: required.cell_epoch, + recovery: required.recovery.clone(), + }, + acquisition_epoch: epoch, + proof, + }) + } +} diff --git a/crates/cellule-runtime/src/control/authority/acquisition/tests.rs b/crates/cellule-runtime/src/control/authority/acquisition/tests.rs new file mode 100644 index 00000000..5fc7d533 --- /dev/null +++ b/crates/cellule-runtime/src/control/authority/acquisition/tests.rs @@ -0,0 +1,463 @@ +use super::super::tests::FaultStore; +use super::*; +use crate::{ + control::{Owner, RootRef}, + identity::{Digest, SessionId}, +}; +use cellule_store::Store; +use object_store::ObjectStoreExt; +use object_store::{memory::InMemory, path::Path}; +use std::sync::{Arc, atomic::Ordering}; + +fn input() -> Control { + Control::initial( + CellId::from_bytes([1; 32]), + IncarnationId::from_bytes([2; 16]), + Owner { + session: SessionId::from_bytes([3; 16]), + endpoint: "https://original.example".into(), + }, + Digest::from_bytes([4; 32]), + 1, + ) + .unwrap() +} +fn successor(input: &Control) -> Control { + input + .takeover(Owner { + session: SessionId::from_bytes([5; 16]), + endpoint: "https://successor.example".into(), + }) + .unwrap() +} +fn authority(store: Arc) -> CellAuthority { + CellAuthority::new(CellStorageLayout::new( + Store::new(store), + Path::from("acquisition-test"), + [7; 16], + )) +} +fn record(input: Control) -> CellAcquisitionRecord { + CellAcquisitionRecord { + materialized: successor(&input), + input, + } +} +#[test] +fn codec_preserves_exact_rootless_and_idle_inputs() { + let mut published = input(); + published.state = ControlState::Serving; + published.root = Some(RootRef { + digest: Digest::from_bytes([6; 32]), + txid: 1, + checksum: cellule_ltx::types::CHECKSUM_FLAG | 7, + commit_sequence: 8, + }); + let idle = published.release().unwrap(); + for input in [input(), published, idle] { + let original = record(input); + let body = original.encode().unwrap(); + assert_eq!(CellAcquisitionRecord::decode(&body).unwrap(), original); + for length in 0..body.len() { + assert!(CellAcquisitionRecord::decode(&body[..length]).is_err()); + } + let mut changed = body.clone(); + changed.push(0); + assert!(CellAcquisitionRecord::decode(&changed).is_err()); + let mut changed = body; + changed[0] ^= 1; + assert!(CellAcquisitionRecord::decode(&changed).is_err()); + } + assert!(CellAcquisitionRecord::decode(&vec![0; MAX_ACQUISITION_BYTES as usize + 1]).is_err()); +} +#[test] +fn codec_refuses_changed_root_scope_owner_and_claim_position() { + let original = record(input()); + for change in 0..5 { + let mut changed = original.clone(); + match change { + 0 => changed.materialized.cell = CellId::from_bytes([8; 32]), + 1 => changed.materialized.epoch += 1, + 2 => { + changed.materialized.owner.as_mut().unwrap().session = + original.input.owner.as_ref().unwrap().session + } + 3 => changed.materialized.code = Digest::from_bytes([9; 32]), + _ => { + changed.materialized.root = Some(RootRef { + digest: Digest::from_bytes([6; 32]), + txid: 1, + checksum: cellule_ltx::types::CHECKSUM_FLAG | 7, + commit_sequence: 8, + }) + } + } + assert!(changed.encode().is_err()); + } +} +#[tokio::test] +async fn immutable_metadata_reconstructs_and_never_invents_serving() { + let store = Arc::new(InMemory::new()); + let authority = authority(store.clone()); + let input = input(); + let materialized = successor(&input); + authority + .retain_acquisition(&input, &materialized) + .await + .unwrap(); + authority + .retain_acquisition(&input, &materialized) + .await + .unwrap(); + let independent = super::tests::authority(store); + let retained = independent + .acquisition_record(input.cell, input.incarnation, 2) + .await + .unwrap() + .unwrap(); + assert_eq!(retained.input(), &input); + assert_eq!(retained.materialized(), &materialized); + assert_eq!(retained.materialized().state, ControlState::Recovering); + assert!(independent.load(input.cell).await.unwrap().is_none()); + let mut different = input.clone(); + different.revision += 1; + different.progress += 1; + assert!( + authority + .retain_acquisition(&different, &successor(&different)) + .await + .is_err() + ); + assert_eq!( + independent + .acquisition_record(input.cell, input.incarnation, 2) + .await + .unwrap(), + Some(retained) + ); + assert!( + independent + .acquisition_record(input.cell, input.incarnation, 3) + .await + .unwrap() + .is_none() + ); + assert!( + independent + .acquisition_record(input.cell, input.incarnation, 1) + .await + .is_err() + ); +} +#[tokio::test] +async fn original_publication_failure_and_lost_reply_preserve_exact_history() { + for lost in [false, true] { + let store = Arc::new(FaultStore::default()); + let authority = authority(store.clone()); + let input = input(); + let materialized = successor(&input); + store + .fault + .store(if lost { 2 } else { 1 }, Ordering::SeqCst); + let result = authority.retain_acquisition(&input, &materialized).await; + if lost { + result.unwrap(); + } else { + assert!(matches!(result, Err(Error::Storage(_)))); + assert!( + authority + .acquisition_record(input.cell, input.incarnation, 2) + .await + .unwrap() + .is_none() + ); + authority + .retain_acquisition(&input, &materialized) + .await + .unwrap(); + } + assert_eq!( + authority + .acquisition_record(input.cell, input.incarnation, 2) + .await + .unwrap(), + Some(record(input)) + ); + } +} +#[tokio::test] +async fn failed_original_read_is_not_absence() { + let store = Arc::new(FaultStore::default()); + let authority = authority(store.clone()); + let input = input(); + let materialized = successor(&input); + authority + .retain_acquisition(&input, &materialized) + .await + .unwrap(); + store.fault.store(6, Ordering::SeqCst); + assert!(matches!( + authority + .acquisition_record(input.cell, input.incarnation, 2) + .await, + Err(Error::Storage(_)), + )); + assert_eq!( + authority + .acquisition_record(input.cell, input.incarnation, 2) + .await + .unwrap(), + Some(record(input)), + ); +} + +#[tokio::test] +async fn cancellation_before_publication_cannot_invent_a_record() { + let store = Arc::new(FaultStore::default()); + let authority = authority(store.clone()); + let input = input(); + let materialized = successor(&input); + store.fault.store(3, Ordering::SeqCst); + let original = input.clone(); + let next = materialized.clone(); + let publisher = authority.clone(); + let blocked = tokio::spawn(async move { publisher.retain_acquisition(&original, &next).await }); + store.entered.notified().await; + blocked.abort(); + assert!(blocked.await.unwrap_err().is_cancelled()); + assert!( + authority + .acquisition_record(input.cell, input.incarnation, 2) + .await + .unwrap() + .is_none() + ); + authority + .retain_acquisition(&input, &materialized) + .await + .unwrap(); + assert_eq!( + authority + .acquisition_record(input.cell, input.incarnation, 2) + .await + .unwrap(), + Some(record(input)), + ); +} + +#[tokio::test] +async fn competing_identical_publications_adopt_one_original_record() { + let store = Arc::new(FaultStore::default()); + let first = authority(store.clone()); + let second = authority(store.clone()); + let input = input(); + let materialized = successor(&input); + store.fault.store(3, Ordering::SeqCst); + let original = input.clone(); + let next = materialized.clone(); + let original_authority = first.clone(); + let blocked = tokio::spawn(async move { + original_authority + .retain_acquisition(&original, &next) + .await + }); + store.entered.notified().await; + second + .retain_acquisition(&input, &materialized) + .await + .unwrap(); + store.resume.notify_one(); + blocked.await.unwrap().unwrap(); + assert_eq!( + first + .acquisition_record(input.cell, input.incarnation, 2) + .await + .unwrap(), + Some(record(input)) + ); +} +#[tokio::test] +async fn missing_foreign_and_corrupt_history_is_not_successful_acquisition() { + let backend = Arc::new(InMemory::new()); + let authority = authority(backend.clone()); + let input = input(); + let original = record(input.clone()); + let path = authority.layout().acquisition_record_path( + input.cell.as_bytes(), + input.incarnation.as_bytes(), + 2, + ); + assert!( + authority + .acquisition_record(input.cell, input.incarnation, 2) + .await + .unwrap() + .is_none() + ); + let mut foreign = input.clone(); + foreign.cell = CellId::from_bytes([9; 32]); + backend + .put(&path, Bytes::from(record(foreign).encode().unwrap()).into()) + .await + .unwrap(); + assert!( + authority + .acquisition_record(input.cell, input.incarnation, 2) + .await + .is_err() + ); + let mut corrupt = original.encode().unwrap(); + corrupt.push(0); + backend + .put(&path, Bytes::from(corrupt).into()) + .await + .unwrap(); + assert!( + authority + .acquisition_record(input.cell, input.incarnation, 2) + .await + .is_err() + ); + backend + .put( + &path, + Bytes::from(vec![0; MAX_ACQUISITION_BYTES as usize + 1]).into(), + ) + .await + .unwrap(); + assert!( + authority + .acquisition_record(input.cell, input.incarnation, 2) + .await + .is_err() + ); +} + +fn suffix() -> crate::recovery::manifest::PinnedRecoveryCell { + crate::recovery::manifest::PinnedRecoveryCell { + application: crate::identity::ApplicationId::from_bytes([7; 16]), + cell: CellId::from_bytes([1; 32]), + incarnation: IncarnationId::from_bytes([2; 16]), + cell_epoch: 1, + recovery: crate::control::RecoveryOverlayRef { + leader_session: SessionId::from_bytes([3; 16]), + log_epoch: 1, + manifest_digest: Digest::from_bytes([9; 32]), + first_node_sequence: 1, + last_node_sequence: 2, + predecessor: RootRef { + digest: Digest::from_bytes([6; 32]), + txid: 1, + checksum: cellule_ltx::types::CHECKSUM_FLAG, + commit_sequence: 1, + }, + final_txid: 2, + final_checksum: cellule_ltx::types::CHECKSUM_FLAG | 1, + final_commit_sequence: 2, + }, + } +} +// These shape fixtures exercise refusal before any origin graph can succeed. +async fn original_suffix_scope( + authority: &CellAuthority, + required: &crate::recovery::manifest::PinnedRecoveryCell, + selected: &Control, +) { + let mut original = input(); + original.state = ControlState::Serving; + original.root = Some(required.recovery.predecessor.clone()); + let original = original.attach_recovery(required.recovery.clone()).unwrap(); + authority.retain_owner(&original).await.unwrap(); + authority + .layout + .store() + .create_strict( + &authority.layout.control_path(required.cell.as_bytes()), + Bytes::from(selected.encode().unwrap()), + ) + .await + .unwrap(); +} +#[tokio::test] +async fn recovered_prefix_requires_original_acquisition_and_preserves_read_errors() { + let store = Arc::new(FaultStore::default()); + let authority = authority(store.clone()); + let required = suffix(); + let root = cellule_ltx::RootRef { + cell: *required.cell.as_bytes(), + incarnation: *required.incarnation.as_bytes(), + digest: [8; 32], + position: cellule_ltx::Position { + txid: 2, + checksum: required.recovery.final_checksum, + }, + commit_sequence: 2, + }; + let replica = cellule_ltx::CellReplica::new( + authority.layout.clone(), + root.cell, + root.incarnation, + cellule_ltx::Limits::default(), + ) + .unwrap(); + let mut selected = successor(&input()); + selected.state = ControlState::Serving; + selected.root = Some(RootRef::from_ltx(required.cell, required.incarnation, root).unwrap()); + original_suffix_scope(&authority, &required, &selected).await; + store.fault.store(6, Ordering::SeqCst); + assert!(matches!( + authority + .verify_recovered_prefix(&required, root, &replica, 0) + .await, + Err(Error::Capacity(_)) + )); + assert_eq!(store.fault.load(Ordering::SeqCst), 6); + assert!(matches!( + authority + .verify_recovered_prefix(&required, root, &replica, 8) + .await, + Err(Error::Storage(_)) + )); + assert!( + matches!(authority.verify_recovered_prefix(&required, root, &replica, 8).await, + Err(Error::AcquisitionHistoryIncomplete { cell, incarnation, epoch }) + if cell==required.cell && incarnation==required.incarnation && epoch==2) + ); +} +#[tokio::test] +async fn matching_endpoint_cannot_replace_a_sealed_recovery_input() { + let authority = authority(Arc::new(InMemory::new())); + let required = suffix(); + let mut original = input(); + original.state = ControlState::Serving; + original.root = Some(RootRef { + digest: Digest::from_bytes([8; 32]), + txid: required.recovery.final_txid, + checksum: required.recovery.final_checksum, + commit_sequence: required.recovery.final_commit_sequence, + }); + let claimed = successor(&original); + authority + .retain_acquisition(&original, &claimed) + .await + .unwrap(); + let root = claimed.ltx_root().unwrap(); + let mut selected = claimed.clone(); + selected.state = ControlState::Serving; + original_suffix_scope(&authority, &required, &selected).await; + let replica = cellule_ltx::CellReplica::new( + authority.layout.clone(), + root.cell, + root.incarnation, + cellule_ltx::Limits::default(), + ) + .unwrap(); + assert!(matches!( + authority + .verify_recovered_prefix(&required, root, &replica, 8) + .await, + Err(Error::Control( + "canonical acquisition recovery input differs" + )) + )); +} diff --git a/crates/cellule-runtime/src/control/authority/history.rs b/crates/cellule-runtime/src/control/authority/history.rs new file mode 100644 index 00000000..bb39b398 --- /dev/null +++ b/crates/cellule-runtime/src/control/authority/history.rs @@ -0,0 +1,199 @@ +use super::*; +use crate::control::ControlState; + +/// Complete owner observations for one current Cell incarnation. +/// +/// Every closed ownership epoch must have its retained original control. The +/// current owner, if present, is the final row. Missing legacy/restore history +/// fails closed; absence cannot exclude a failed original session. This metadata +/// proves no process joining, complete catalog scope, data availability, current +/// successor serving or operation completion. +pub struct CellOwnerHistory { + current: Control, + owners: Vec, +} + +impl CellOwnerHistory { + /// Exact current control checked before and after history collection. + #[must_use] + pub const fn current(&self) -> &Control { + &self.current + } + + /// Full original owner observations in strictly increasing epoch order. + #[must_use] + pub fn owners(&self) -> &[Control] { + &self.owners + } +} + +impl CellAuthority { + /// Loads the full retained owner observation for one exact epoch. + /// + /// The version 1 path contains canonical control JSON, bounded by 8 KiB. + /// A retained observation may come from an unsuccessful departure proposal; + /// it supplies no proof that its release/takeover/tombstone CAS committed. + pub async fn owner_observation( + &self, + cell: CellId, + incarnation: IncarnationId, + epoch: u64, + ) -> Result> { + Ok(self + .load_owner_observation(cell, incarnation, epoch) + .await? + .map(|(control, _)| control)) + } + + /// Collects every original owner epoch without listing objects or changing + /// authority. The caller bounds retained rows and the enclosing deadline. + /// Concurrent authority changes refuse this observation rather than return + /// an incomplete or mixed history. This reads only the current incarnation. + pub async fn owner_history(&self, cell: CellId, limit: usize) -> Result { + let current = self.load(cell).await?.ok_or(Error::Fenced)?.value; + let count = match current.state { + ControlState::Tombstoned => current.epoch - 1, + _ => current.epoch, + }; + if limit == 0 || count > limit as u64 { + return Err(Error::Capacity("Cell owner history exceeds its row limit")); + } + let closed = if current.owner.is_some() { + count - 1 + } else { + count + }; + let mut owners = Vec::::new(); + for epoch in 1..=closed { + let owner = self + .owner_observation(cell, current.incarnation, epoch) + .await? + .ok_or(Error::OwnerHistoryIncomplete { + cell, + incarnation: current.incarnation, + epoch, + })?; + if owner.revision >= current.revision + || owner.progress >= current.progress + || owners.last().is_some_and(|previous| { + previous.revision >= owner.revision || previous.progress >= owner.progress + }) + { + return Err(Error::Control("Cell owner history is not ordered")); + } + owners.push(owner); + } + if current.owner.is_some() { + owners.push(current.clone()); + } + if self + .load(cell) + .await? + .is_none_or(|latest| latest.value != current) + { + return Err(Error::Fenced); + } + Ok(CellOwnerHistory { current, owners }) + } + + pub(super) async fn retain_owner(&self, observed: &Control) -> Result<()> { + let path = self.layout.owner_observation_path( + observed.cell.as_bytes(), + observed.incarnation.as_bytes(), + observed.epoch, + ); + let body = Bytes::from(observed.encode()?); + let previous = self + .load_owner_observation(observed.cell, observed.incarnation, observed.epoch) + .await?; + if let Some((previous, _)) = &previous + && retained(previous, observed)? + { + return Ok(()); + } + let written = match previous { + Some((_, token)) => self.layout.store().update(&path, body, token).await, + None => { + self.layout + .store() + .create_strict_with_etag(&path, body) + .await + } + }; + match written { + Ok(_) => Ok(()), + Err(source) => { + // Inspect ambiguous publication before allowing authority + // departure. Absence, older history or an unavailable re-read + // preserves the original write error. The existing coordinator + // owns retries and full predicate revalidation; no loop here + // can indefinitely retain a drained owner's metadata job. + if let Ok(Some((current, _))) = self + .load_owner_observation(observed.cell, observed.incarnation, observed.epoch) + .await + && retained(¤t, observed)? + { + return Ok(()); + } + Err(source.into()) + } + } + } + + async fn load_owner_observation( + &self, + cell: CellId, + incarnation: IncarnationId, + epoch: u64, + ) -> Result> { + if epoch == 0 { + return Err(Error::Control("Cell owner history epoch is zero")); + } + let path = + self.layout + .owner_observation_path(cell.as_bytes(), incarnation.as_bytes(), epoch); + let (body, token) = match self + .layout + .store() + .get_with_etag_bounded(&path, MAX_CONTROL_BYTES) + .await + { + Ok(value) => value, + Err(StorageError::NotFound { .. }) => return Ok(None), + Err(source) => return Err(source.into()), + }; + let control = Control::decode(&body)?; + if control.cell != cell + || control.incarnation != incarnation + || control.epoch != epoch + || control.owner.is_none() + { + return Err(Error::Control( + "Cell owner history path differs from its scope", + )); + } + Ok(Some((control, token))) + } +} + +// The same canonical epoch may be proposed for departure at several revisions. +// A delayed proposal cannot replace a later observation or conflate boot owners. +fn retained(previous: &Control, observed: &Control) -> Result { + let ordered_progress = if previous.revision >= observed.revision { + previous.progress.checked_sub(observed.progress) + == Some(previous.revision - observed.revision) + } else { + observed.progress.checked_sub(previous.progress) + == Some(observed.revision - previous.revision) + }; + if previous.owner != observed.owner + || previous.cell != observed.cell + || previous.incarnation != observed.incarnation + || previous.epoch != observed.epoch + || !ordered_progress + || (previous.revision == observed.revision && previous != observed) + { + return Err(Error::Control("Cell owner history observations conflict")); + } + Ok(previous.revision >= observed.revision) +} diff --git a/crates/cellule-runtime/src/control/authority/lineage/codec.rs b/crates/cellule-runtime/src/control/authority/lineage/codec.rs new file mode 100644 index 00000000..08e7e776 --- /dev/null +++ b/crates/cellule-runtime/src/control/authority/lineage/codec.rs @@ -0,0 +1,81 @@ +use super::*; + +const DOMAIN: &[u8] = b"cellule.root-lineage.v1\0"; +impl CellRootLineage { + pub(super) fn encode(&self) -> Result> { + self.validate()?; + let mut body = DOMAIN.to_vec(); + body.extend_from_slice(&self.root.cell); + body.extend_from_slice(&self.root.incarnation); + write_root(&mut body, &self.root); + body.extend_from_slice(&(self.predecessors.len() as u16).to_be_bytes()); + for parent in &self.predecessors { + write_root(&mut body, parent); + } + let checksum = *blake3::hash(&body).as_bytes(); + body.extend_from_slice(&checksum); + if body.len() as u64 > MAX_LINEAGE_BYTES { + return Err(Error::Control("Cell root lineage envelope exceeds bound")); + } + Ok(body) + } + pub(super) fn decode(body: &[u8]) -> Result { + if body.len() as u64 > MAX_LINEAGE_BYTES || body.len() < DOMAIN.len() + 32 { + return Err(Error::Control("invalid Cell root lineage envelope")); + } + let (payload, checksum) = body.split_at(body.len() - 32); + if blake3::hash(payload).as_bytes().as_slice() != checksum { + return Err(Error::Control("Cell root lineage checksum differs")); + } + let mut remaining = payload + .strip_prefix(DOMAIN) + .ok_or(Error::Control("unknown Cell root lineage version"))?; + let cell = take::<32>(&mut remaining)?; + let incarnation = take::<16>(&mut remaining)?; + let root = read_root(&mut remaining, cell, incarnation)?; + let count = u16::from_be_bytes(take::<2>(&mut remaining)?) as usize; + if count > MAX_PREDECESSORS { + return Err(Error::Capacity( + "Cell root lineage predecessor bound exceeded", + )); + } + let mut predecessors = Vec::with_capacity(count); + for _ in 0..count { + predecessors.push(read_root(&mut remaining, cell, incarnation)?); + } + if !remaining.is_empty() { + return Err(Error::Control("trailing Cell root lineage bytes")); + } + let record = Self { root, predecessors }; + record.validate()?; + Ok(record) + } +} +fn take(remaining: &mut &[u8]) -> Result<[u8; N]> { + let bytes = remaining + .get(..N) + .ok_or(Error::Control("truncated Cell root lineage"))?; + let value = bytes + .try_into() + .map_err(|_| Error::Control("truncated Cell root lineage"))?; + *remaining = &remaining[N..]; + Ok(value) +} +fn write_root(body: &mut Vec, root: &RootRef) { + body.extend_from_slice(&root.digest); + body.extend_from_slice(&root.position.txid.to_be_bytes()); + body.extend_from_slice(&root.position.checksum.to_be_bytes()); + body.extend_from_slice(&root.commit_sequence.to_be_bytes()); +} +fn read_root(remaining: &mut &[u8], cell: [u8; 32], incarnation: [u8; 16]) -> Result { + Ok(RootRef { + cell, + incarnation, + digest: take::<32>(remaining)?, + position: cellule_ltx::Position { + txid: u64::from_be_bytes(take::<8>(remaining)?), + checksum: u64::from_be_bytes(take::<8>(remaining)?), + }, + commit_sequence: u64::from_be_bytes(take::<8>(remaining)?), + }) +} diff --git a/crates/cellule-runtime/src/control/authority/lineage/mod.rs b/crates/cellule-runtime/src/control/authority/lineage/mod.rs new file mode 100644 index 00000000..c4bbce69 --- /dev/null +++ b/crates/cellule-runtime/src/control/authority/lineage/mod.rs @@ -0,0 +1,195 @@ +//! Canonical verified-preparation links, independent of ownership authority. +use super::*; +use cellule_ltx::{CellReplica, PreparedRoot, RootRef}; +use std::collections::BTreeMap; + +mod codec; +#[cfg(test)] +mod tests; + +#[cfg(test)] +mod preparation_tests; +mod verify; +pub use verify::VerifiedRootPrefix; + +const MAX_LINEAGE_BYTES: u64 = 8 * 1024; +const MAX_PREDECESSORS: usize = 64; +pub(crate) const MAX_LINEAGE_ROOTS: usize = 10_000; + +/// Retained verified preparations that produced one exact immutable root. +/// +/// Only opaque native `PreparedRoot` values add links. Multiple verified inputs +/// can produce identical root bytes, including representation-only compaction. +/// Links accumulate by ETag CAS; delayed writers cannot replace existing links. +/// A proposal need not have won authority. Starting from a separately selected +/// root proves verified derivation, never which proposal won ownership or serving. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct CellRootLineage { + root: RootRef, + predecessors: Vec, +} +impl CellRootLineage { + /// Exact root whose verified preparations are retained. + #[must_use] + pub const fn root(&self) -> RootRef { + self.root + } + /// Every retained distinct input, in strictly increasing digest order. + #[must_use] + pub fn predecessors(&self) -> &[RootRef] { + &self.predecessors + } + fn validate(&self) -> Result<()> { + validate_root(&self.root)?; + if self.predecessors.len() > MAX_PREDECESSORS { + return Err(Error::Capacity( + "Cell root lineage predecessor bound exceeded", + )); + } + for (index, parent) in self.predecessors.iter().enumerate() { + validate_root(parent)?; + if parent.cell != self.root.cell + || parent.incarnation != self.root.incarnation + || parent.digest == self.root.digest + || parent.position.txid > self.root.position.txid + || (parent.position.txid == self.root.position.txid + && parent.position.checksum != self.root.position.checksum) + || parent.commit_sequence > self.root.commit_sequence + || (parent.commit_sequence == self.root.commit_sequence + && parent.position != self.root.position) + || (index > 0 && self.predecessors[index - 1].digest >= parent.digest) + { + return Err(Error::Control("invalid Cell root lineage predecessor")); + } + } + Ok(()) + } +} +fn validate_root(root: &RootRef) -> Result<()> { + if root.position.txid == 0 + || root.position.checksum & cellule_ltx::types::CHECKSUM_FLAG == 0 + || root.commit_sequence > i64::MAX as u64 + { + return Err(Error::Control("invalid Cell root lineage position")); + } + Ok(()) +} + +impl CellAuthority { + /// Bounded origin read of retained verified preparations, without authority. + /// Missing legacy/manual-publication metadata remains `None`. + pub async fn root_lineage(&self, root: RootRef) -> Result> { + Ok(self + .load_root_lineage(root) + .await? + .map(|(record, _)| record)) + } + + async fn load_root_lineage(&self, root: RootRef) -> Result> { + validate_root(&root)?; + let path = self + .layout + .root_lineage_path(&root.cell, &root.incarnation, &root.digest); + let (body, token) = match self + .layout + .store() + .get_with_etag_bounded(&path, MAX_LINEAGE_BYTES) + .await + { + Ok(value) => value, + Err(StorageError::NotFound { .. }) => return Ok(None), + Err(source) => return Err(source.into()), + }; + let record = CellRootLineage::decode(&body)?; + if record.root != root { + return Err(Error::Control( + "Cell root lineage path differs from its root", + )); + } + Ok(Some((record, token))) + } + + pub(crate) async fn retain_root_lineage(&self, prepared: &PreparedRoot) -> Result<()> { + self.retain_verified_link(prepared.root(), prepared.predecessor()) + .await + } + + pub(crate) async fn retain_root_preparation( + &self, + preparation: cellule_ltx::RootPreparation, + ) -> Result<()> { + self.retain_verified_link(preparation.root(), preparation.predecessor()) + .await + } + + // Private: only opaque native preparation factories supply production links. + async fn retain_verified_link(&self, root: RootRef, parent: Option) -> Result<()> { + validate_root(&root)?; + if parent == Some(root) { + return Ok(()); + } + let proposed = CellRootLineage { + root, + predecessors: parent.into_iter().collect(), + }; + let body = Bytes::from(proposed.encode()?); + let path = self + .layout + .root_lineage_path(&root.cell, &root.incarnation, &root.digest); + // Native fresh roots normally have fresh immutable metadata identities. + // Strict creation itself proves absence and publication atomically; an + // absence GET adds a round trip to every command without adding safety. + let conflict = match self + .layout + .store() + .create_strict_with_etag(&path, body) + .await + { + Ok(_) => return Ok(()), + Err(source @ StorageError::StateConflict { .. }) => source, + Err(source) => { + // A lost create reply is adopted only after exact native input + // confirmation. Missing/corrupt/failed confirmation preserves + // the original publication error for its existing work owner. + if let Ok(Some(current)) = self.root_lineage(root).await + && parent.is_none_or(|parent| current.predecessors.contains(&parent)) + { + return Ok(()); + } + return Err(source.into()); + } + }; + let Some((mut record, token)) = self.load_root_lineage(root).await? else { + return Err(conflict.into()); + }; + if let Some(parent) = parent { + match record + .predecessors + .binary_search_by_key(&parent.digest, |root| root.digest) + { + Ok(index) if record.predecessors[index] == parent => return Ok(()), + Ok(_) => return Err(Error::Control("Cell root lineage digest changed position")), + Err(index) => record.predecessors.insert(index, parent), + } + } else { + return Ok(()); + } + // One bounded conflict merge belongs to this publication attempt. ETag + // CAS preserves every earlier verified input; no detached retry owner. + let body = Bytes::from(record.encode()?); + let written = self.layout.store().update(&path, body, token).await; + match written { + Ok(_) => Ok(()), + Err(source) => { + // Accept only a confirmed equal/superset link. No metadata loop + // can outlive the existing publisher/acquisition work owner. + if let Ok(Some(current)) = self.root_lineage(root).await + && parent.is_none_or(|parent| current.predecessors.contains(&parent)) + { + return Ok(()); + } + Err(source.into()) + } + } + } +} diff --git a/crates/cellule-runtime/src/control/authority/lineage/preparation_tests.rs b/crates/cellule-runtime/src/control/authority/lineage/preparation_tests.rs new file mode 100644 index 00000000..536e93dc --- /dev/null +++ b/crates/cellule-runtime/src/control/authority/lineage/preparation_tests.rs @@ -0,0 +1,333 @@ +use super::super::tests::FaultStore; +use super::*; +use crate::publication::CellPublisher; +use cellule_ltx::{CaptureBatch, CellReplica, Db, Limits}; +use cellule_store::Store; +use object_store::path::Path; +use std::sync::{Arc, atomic::Ordering}; +use std::time::Duration; + +struct NativePreparation { + store: Arc, + authority: CellAuthority, + publisher: CellPublisher, + database: Db, + cuts: CaptureBatch, + _directory: tempfile::TempDir, +} +impl NativePreparation { + async fn new() -> Self { + let directory = tempfile::tempdir().unwrap(); + let mut database = + Db::open(&directory.path().join("native.sqlite"), Limits::default()).unwrap(); + database + .transaction(|transaction| { + transaction.execute_batch( + "CREATE TABLE events(value INTEGER); INSERT INTO events VALUES(7)", + )?; + Ok(()) + }) + .unwrap(); + let cuts = database.capture_deferred().unwrap(); + let store = Arc::new(FaultStore::default()); + let authority = CellAuthority::new(CellStorageLayout::new( + Store::new(store.clone()), + Path::from("native-preparation"), + [3; 16], + )); + let cell = CellId::from_bytes([1; 32]); + let initial = crate::control::Control::initial( + cell, + IncarnationId::from_bytes([2; 16]), + crate::control::Owner { + session: crate::identity::SessionId::from_bytes([4; 16]), + endpoint: "https://publisher.internal".into(), + }, + crate::identity::Digest::from_bytes([5; 32]), + 1, + ) + .unwrap(); + authority + .layout + .store() + .create_strict( + &authority.layout.control_path(cell.as_bytes()), + Bytes::from(initial.encode().unwrap()), + ) + .await + .unwrap(); + let replica = CellReplica::new( + authority.layout.clone(), + [1; 32], + [2; 16], + Limits::default(), + ) + .unwrap(); + let observed = authority.load(cell).await.unwrap().unwrap(); + let publisher = CellPublisher::new( + replica, + authority.clone(), + observed, + directory.path().to_owned(), + ); + Self { + store, + authority, + publisher, + database, + cuts, + _directory: directory, + } + } + async fn assert_unpublished(&self) { + assert!( + self.authority + .load(CellId::from_bytes([1; 32])) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .is_none() + ); + } +} + +#[tokio::test] +async fn lineage_and_native_uploads_overlap_but_ready_and_authority_wait_for_both() { + let mut fixture = NativePreparation::new().await; + fixture.store.fault.store(3, Ordering::SeqCst); + let prepared = { + let preparation = fixture.publisher.prepare_initial(&fixture.cuts); + tokio::pin!(preparation); + tokio::time::timeout(Duration::from_secs(5), async { + tokio::select! { + result = &mut preparation => panic!("preparation escaped paused metadata: {}", result.is_ok()), + _ = async { + fixture.store.entered.notified().await; + while fixture.store.root_writes.load(Ordering::SeqCst) < 2 { + fixture.store.root_written.notified().await; + } + } => {} + } + }).await.unwrap(); + assert!( + fixture + .authority + .load(CellId::from_bytes([1; 32])) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .is_none() + ); + fixture.store.resume.notify_one(); + preparation.await.unwrap() + }; + assert_eq!(fixture.store.lineage_writes.load(Ordering::SeqCst), 1); + assert_eq!(fixture.store.lineage_reads.load(Ordering::SeqCst), 0); + fixture + .publisher + .publish_prepared(&prepared, None) + .await + .unwrap(); + assert_eq!( + fixture.store.lineage_writes.load(Ordering::SeqCst), + 1, + "publication must reuse its exact retained preparation" + ); + assert_eq!(fixture.store.lineage_reads.load(Ordering::SeqCst), 0); + let replica = CellReplica::new( + fixture.authority.layout.clone(), + [1; 32], + [2; 16], + Limits::default(), + ) + .unwrap(); + fixture + .authority + .verify_root_prefix(prepared.root(), prepared.root(), &replica, 64) + .await + .unwrap(); + fixture.database.close().unwrap(); +} + +#[tokio::test] +async fn retained_lineage_cannot_publish_before_native_uploads_finish() { + let mut fixture = NativePreparation::new().await; + fixture.store.root_fault.store(3, Ordering::SeqCst); + let prepared = { + let preparation = fixture.publisher.prepare_initial(&fixture.cuts); + tokio::pin!(preparation); + tokio::time::timeout(Duration::from_secs(5), async { + tokio::select! { + result = &mut preparation => panic!("preparation escaped paused native upload: {}", result.is_ok()), + _ = async { + fixture.store.entered.notified().await; + while fixture.store.lineage_writes.load(Ordering::SeqCst) < 1 { + tokio::task::yield_now().await; + } + } => {} + } + }).await.unwrap(); + assert!( + fixture + .authority + .load(CellId::from_bytes([1; 32])) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .is_none() + ); + fixture.store.resume.notify_one(); + preparation.await.unwrap() + }; + fixture + .publisher + .publish_prepared(&prepared, None) + .await + .unwrap(); + assert_eq!(fixture.store.lineage_writes.load(Ordering::SeqCst), 1); + fixture.database.close().unwrap(); +} + +#[tokio::test] +async fn native_preparation_preserves_original_metadata_errors_and_adopts_only_exact_lost_replies() +{ + for fault in [1, 2] { + let mut fixture = NativePreparation::new().await; + fixture.store.fault.store(fault, Ordering::SeqCst); + let result = fixture.publisher.prepare_initial(&fixture.cuts).await; + fixture.assert_unpublished().await; + let prepared = if fault == 1 { + assert!(matches!( + result, + Err(Error::Storage(StorageError::NotSupported { .. })) + )); + fixture + .publisher + .prepare_initial(&fixture.cuts) + .await + .unwrap() + } else { + result.unwrap() + }; + fixture + .publisher + .publish_prepared(&prepared, None) + .await + .unwrap(); + assert!( + fixture + .authority + .root_lineage(prepared.root()) + .await + .unwrap() + .is_some() + ); + fixture.database.close().unwrap(); + } +} + +#[tokio::test] +async fn failed_native_upload_cannot_select_root_even_with_retained_metadata() { + let mut fixture = NativePreparation::new().await; + fixture.store.root_fault.store(1, Ordering::SeqCst); + assert!(matches!( + fixture.publisher.prepare_initial(&fixture.cuts).await, + Err(Error::Ltx(_)) + )); + fixture.assert_unpublished().await; + let prepared = fixture + .publisher + .prepare_initial(&fixture.cuts) + .await + .unwrap(); + fixture + .publisher + .publish_prepared(&prepared, None) + .await + .unwrap(); + fixture.database.close().unwrap(); +} + +#[tokio::test] +async fn native_compaction_records_final_predecessor_without_a_second_conflict_write() { + let mut fixture = NativePreparation::new().await; + let initial = fixture + .publisher + .prepare_initial(&fixture.cuts) + .await + .unwrap(); + fixture + .publisher + .publish_prepared(&initial, None) + .await + .unwrap(); + for sequence in 1..=40 { + fixture + .database + .transaction(|transaction| { + transaction.execute("INSERT INTO events VALUES (?1)", [sequence])?; + Ok(()) + }) + .unwrap(); + let cuts = fixture.database.capture_deferred().unwrap(); + let original = fixture + .authority + .load(CellId::from_bytes([1; 32])) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .unwrap(); + let prepared = fixture + .publisher + .prepare_batch(&cuts, sequence) + .await + .unwrap(); + assert_eq!(prepared.predecessor(), Some(original)); + fixture + .publisher + .publish_prepared(&prepared, None) + .await + .unwrap(); + } + assert_eq!( + fixture.store.lineage_reads.load(Ordering::SeqCst), + 0, + "native compaction must retain its final authority predecessor before Ready escapes" + ); + assert_eq!( + fixture.store.lineage_writes.load(Ordering::SeqCst), + 41, + "one exact lineage write per complete proposal" + ); + let replica = CellReplica::new( + fixture.authority.layout.clone(), + [1; 32], + [2; 16], + Limits::default(), + ) + .unwrap(); + let latest = fixture + .authority + .load(CellId::from_bytes([1; 32])) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .unwrap(); + assert!(replica.open_root(&latest).await.unwrap().segment_count() < 32); + fixture + .authority + .verify_root_prefix(initial.root(), latest, &replica, 64) + .await + .unwrap(); + fixture.database.close().unwrap(); +} diff --git a/crates/cellule-runtime/src/control/authority/lineage/tests.rs b/crates/cellule-runtime/src/control/authority/lineage/tests.rs new file mode 100644 index 00000000..b0e2ed83 --- /dev/null +++ b/crates/cellule-runtime/src/control/authority/lineage/tests.rs @@ -0,0 +1,487 @@ +use super::super::tests::FaultStore; +use super::*; +use cellule_store::Store; +use object_store::{ObjectStoreExt, memory::InMemory, path::Path}; +use std::sync::{Arc, atomic::Ordering}; + +fn root(digest: u8, sequence: u64) -> RootRef { + RootRef { + cell: [1; 32], + incarnation: [2; 16], + digest: [digest; 32], + position: cellule_ltx::Position { + txid: sequence + 1, + checksum: cellule_ltx::types::CHECKSUM_FLAG | sequence, + }, + commit_sequence: sequence, + } +} +fn authority(store: Arc) -> CellAuthority { + CellAuthority::new(CellStorageLayout::new( + Store::new(store), + Path::from("root-lineage"), + [3; 16], + )) +} +#[test] +fn codec_bounds_canonical_links_and_all_corruption() { + let record = CellRootLineage { + root: root(80, 80), + predecessors: (1..=64).map(|byte| root(byte, u64::from(byte))).collect(), + }; + let body = record.encode().unwrap(); + assert!(body.len() < MAX_LINEAGE_BYTES as usize); + assert_eq!(CellRootLineage::decode(&body).unwrap(), record); + for length in 0..body.len() { + assert!(CellRootLineage::decode(&body[..length]).is_err()); + } + for index in 0..body.len() { + let mut corrupt = body.clone(); + corrupt[index] ^= 1; + assert!(CellRootLineage::decode(&corrupt).is_err()); + } + let mut extra = body; + extra.push(0); + assert!(CellRootLineage::decode(&extra).is_err()); + let mut record = record; + record.predecessors.push(root(65, 65)); + assert!(record.encode().is_err()); +} +#[test] +fn codec_refuses_scope_position_duplicates_and_order_changes() { + for change in 0..6 { + let mut record = CellRootLineage { + root: root(8, 8), + predecessors: vec![root(1, 1), root(2, 2)], + }; + match change { + 0 => record.predecessors[0].cell = [9; 32], + 1 => record.predecessors[0].incarnation = [9; 16], + 2 => record.predecessors[0] = record.root, + 3 => record.predecessors[0] = root(1, 9), + 4 => record.predecessors[1] = record.predecessors[0], + _ => record.predecessors.reverse(), + } + assert!(record.encode().is_err()); + } + let mut parent = root(1, 8); + parent.digest = [1; 32]; + let record = CellRootLineage { + root: root(8, 8), + predecessors: vec![parent], + }; + record.encode().unwrap(); // Representation-only compaction preserves position. + let mut changed = record; + changed.predecessors[0].position.checksum ^= 1; + assert!(changed.encode().is_err()); +} +#[tokio::test] +async fn links_accumulate_and_reconstruct_without_authority() { + let backend = Arc::new(InMemory::new()); + let first = authority(backend.clone()); + let child = root(8, 8); + first + .retain_verified_link(child, Some(root(2, 2))) + .await + .unwrap(); + first + .retain_verified_link(child, Some(root(1, 1))) + .await + .unwrap(); + first + .retain_verified_link(child, Some(root(2, 2))) + .await + .unwrap(); + let second = authority(backend); + let record = second.root_lineage(child).await.unwrap().unwrap(); + assert_eq!(record.predecessors(), &[root(1, 1), root(2, 2)]); + assert!( + second + .load(CellId::from_bytes(child.cell)) + .await + .unwrap() + .is_none() + ); + let mut different = root(2, 2); + different.commit_sequence += 1; + assert!( + second + .retain_verified_link(child, Some(different)) + .await + .is_err() + ); + assert_eq!(second.root_lineage(child).await.unwrap().unwrap(), record); +} +#[tokio::test] +async fn failed_and_lost_link_replies_do_not_invent_success() { + for fault in [1, 2] { + let store = Arc::new(FaultStore::default()); + let authority = authority(store.clone()); + store.fault.store(fault, Ordering::SeqCst); + let result = authority + .retain_verified_link(root(8, 8), Some(root(1, 1))) + .await; + if fault == 1 { + assert!(matches!(result, Err(Error::Storage(_)))); + assert!(authority.root_lineage(root(8, 8)).await.unwrap().is_none()); + } else { + result.unwrap(); + } + } +} +#[tokio::test] +async fn competing_distinct_links_cannot_erase_the_original() { + let store = Arc::new(FaultStore::default()); + let first = authority(store.clone()); + let second = authority(store.clone()); + store.fault.store(3, Ordering::SeqCst); + let task = tokio::spawn(async move { + first + .retain_verified_link(root(8, 8), Some(root(1, 1))) + .await + }); + store.entered.notified().await; + second + .retain_verified_link(root(8, 8), Some(root(2, 2))) + .await + .unwrap(); + store.resume.notify_one(); + task.await.unwrap().unwrap(); + // The delayed create conflicts, then its single ETag merge retains both + // native inputs before returning. Replay must preserve the complete set. + second + .retain_verified_link(root(8, 8), Some(root(1, 1))) + .await + .unwrap(); + assert_eq!( + second + .root_lineage(root(8, 8)) + .await + .unwrap() + .unwrap() + .predecessors(), + &[root(1, 1), root(2, 2)] + ); +} +#[tokio::test] +async fn publication_confirms_native_links_before_its_root_cas() { + for fault in [1, 2] { + let store = Arc::new(FaultStore::default()); + let authority = authority(store.clone()); + let directory = tempfile::tempdir().unwrap(); + let mut database = cellule_ltx::Db::open( + &directory.path().join("publication.sqlite"), + cellule_ltx::Limits::default(), + ) + .unwrap(); + database + .transaction(|transaction| { + transaction.execute_batch( + "CREATE TABLE values_seen(value INTEGER); INSERT INTO values_seen VALUES(7)", + )?; + Ok(()) + }) + .unwrap(); + let cell = CellId::from_bytes([1; 32]); + let incarnation = IncarnationId::from_bytes([2; 16]); + let initial = Control::initial( + cell, + incarnation, + crate::control::Owner { + session: crate::identity::SessionId::from_bytes([4; 16]), + endpoint: "https://publisher.internal".into(), + }, + crate::identity::Digest::from_bytes([5; 32]), + 1, + ) + .unwrap(); + authority + .layout + .store() + .create_strict( + &authority.layout.control_path(cell.as_bytes()), + Bytes::from(initial.encode().unwrap()), + ) + .await + .unwrap(); + let replica = CellReplica::new( + authority.layout.clone(), + *cell.as_bytes(), + *incarnation.as_bytes(), + cellule_ltx::Limits::default(), + ) + .unwrap(); + let prepared = replica + .prepare(None, &database.capture().unwrap(), 1, 1) + .await + .unwrap(); + let observed = authority.load(cell).await.unwrap().unwrap(); + let mut publisher = crate::publication::CellPublisher::new( + replica, + authority.clone(), + observed, + directory.path().to_owned(), + ); + store.fault.store(fault, Ordering::SeqCst); + let result = publisher.publish_prepared(&prepared, None).await; + if fault == 1 { + assert!(matches!(result, Err(Error::Storage(_)))); + assert_eq!( + authority.load(cell).await.unwrap().unwrap().value(), + &initial + ); + assert!( + authority + .root_lineage(prepared.root()) + .await + .unwrap() + .is_none() + ); + publisher.publish_prepared(&prepared, None).await.unwrap(); + } else { + result.unwrap(); + } + assert_eq!( + authority + .load(cell) + .await + .unwrap() + .unwrap() + .value() + .ltx_root(), + Some(prepared.root()) + ); + assert!( + authority + .root_lineage(prepared.root()) + .await + .unwrap() + .unwrap() + .predecessors() + .is_empty() + ); + database.close().unwrap(); + } +} + +#[tokio::test] +async fn corrupt_or_foreign_metadata_is_not_absence() { + let backend = Arc::new(InMemory::new()); + let authority = authority(backend.clone()); + let child = root(8, 8); + let path = authority + .layout + .root_lineage_path(&child.cell, &child.incarnation, &child.digest); + assert!(authority.root_lineage(child).await.unwrap().is_none()); + backend + .put(&path, Bytes::from_static(b"corrupt").into()) + .await + .unwrap(); + assert!(authority.root_lineage(child).await.is_err()); + assert!(matches!( + authority + .retain_verified_link(child, Some(root(1, 1))) + .await, + Err(Error::Control(_)) + )); + let foreign = CellRootLineage { + root: root(9, 9), + predecessors: vec![], + }; + backend + .put(&path, Bytes::from(foreign.encode().unwrap()).into()) + .await + .unwrap(); + assert!(authority.root_lineage(child).await.is_err()); + assert!(matches!( + authority + .retain_verified_link(child, Some(root(1, 1))) + .await, + Err(Error::Control(_)) + )); +} + +#[tokio::test] +async fn fresh_native_publications_do_not_read_absent_lineage() { + let store = Arc::new(FaultStore::default()); + let authority = authority(store.clone()); + let directory = tempfile::tempdir().unwrap(); + let mut database = cellule_ltx::Db::open( + &directory.path().join("fresh-publication.sqlite"), + cellule_ltx::Limits::default(), + ) + .unwrap(); + database + .transaction(|transaction| { + transaction.execute_batch( + "CREATE TABLE values_seen(value INTEGER); INSERT INTO values_seen VALUES(0)", + )?; + Ok(()) + }) + .unwrap(); + let cell = CellId::from_bytes([1; 32]); + let incarnation = IncarnationId::from_bytes([2; 16]); + let initial = Control::initial( + cell, + incarnation, + crate::control::Owner { + session: crate::identity::SessionId::from_bytes([4; 16]), + endpoint: "https://publisher.internal".into(), + }, + crate::identity::Digest::from_bytes([5; 32]), + 1, + ) + .unwrap(); + authority + .layout + .store() + .create_strict( + &authority.layout.control_path(cell.as_bytes()), + Bytes::from(initial.encode().unwrap()), + ) + .await + .unwrap(); + let replica = CellReplica::new( + authority.layout.clone(), + *cell.as_bytes(), + *incarnation.as_bytes(), + cellule_ltx::Limits::default(), + ) + .unwrap(); + let mut publisher = crate::publication::CellPublisher::new( + replica.clone(), + authority.clone(), + authority.load(cell).await.unwrap().unwrap(), + directory.path().to_owned(), + ); + let mut previous = None; + let mut original = None; + for sequence in 1..=16 { + database + .transaction(|transaction| { + transaction.execute("UPDATE values_seen SET value = value + 1", [])?; + Ok(()) + }) + .unwrap(); + let prepared = replica + .prepare(previous.as_ref(), &database.capture().unwrap(), sequence, 1) + .await + .unwrap(); + publisher.publish_prepared(&prepared, None).await.unwrap(); + original.get_or_insert(prepared.root()); + previous = Some(prepared.root()); + } + // This exercises the ordinary publisher with genuinely captured native + // roots. Fresh immutable identities need no absence GET before strict PUT. + assert_eq!(store.lineage_reads.load(Ordering::SeqCst), 0); + assert_eq!(store.lineage_writes.load(Ordering::SeqCst), 16); + let latest = authority + .load(cell) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .unwrap(); + assert_eq!(Some(latest), previous); + authority + .verify_root_prefix(original.unwrap(), latest, &replica, 16) + .await + .unwrap(); + assert!(store.lineage_reads.load(Ordering::SeqCst) > 0); + database.close().unwrap(); +} + +#[tokio::test] +async fn failed_and_lost_conflict_merges_preserve_original_inputs_and_errors() { + for fault in [1, 2] { + let store = Arc::new(FaultStore::default()); + let authority = authority(store.clone()); + let child = root(8, 8); + authority + .retain_verified_link(child, Some(root(2, 2))) + .await + .unwrap(); + store.lineage_update_fault.store(fault, Ordering::SeqCst); + let result = authority + .retain_verified_link(child, Some(root(1, 1))) + .await; + if fault == 1 { + assert!(matches!( + result, + Err(Error::Storage(StorageError::NotSupported { .. })) + )); + assert_eq!( + authority + .root_lineage(child) + .await + .unwrap() + .unwrap() + .predecessors(), + &[root(2, 2)] + ); + authority + .retain_verified_link(child, Some(root(1, 1))) + .await + .unwrap(); + } else { + result.unwrap(); + } + assert_eq!( + authority + .root_lineage(child) + .await + .unwrap() + .unwrap() + .predecessors(), + &[root(1, 1), root(2, 2)] + ); + } +} + +#[tokio::test] +async fn delayed_conflict_merge_cannot_erase_a_concurrent_extension() { + let store = Arc::new(FaultStore::default()); + let first = authority(store.clone()); + let second = authority(store.clone()); + let child = root(8, 8); + first + .retain_verified_link(child, Some(root(2, 2))) + .await + .unwrap(); + store.lineage_update_fault.store(3, Ordering::SeqCst); + let task = + tokio::spawn(async move { first.retain_verified_link(child, Some(root(1, 1))).await }); + store.entered.notified().await; + second + .retain_verified_link(child, Some(root(3, 3))) + .await + .unwrap(); + store.resume.notify_one(); + assert!(matches!( + task.await.unwrap(), + Err(Error::Storage(StorageError::StateConflict { .. })) + )); + assert_eq!( + second + .root_lineage(child) + .await + .unwrap() + .unwrap() + .predecessors(), + &[root(2, 2), root(3, 3)] + ); + second + .retain_verified_link(child, Some(root(1, 1))) + .await + .unwrap(); + assert_eq!( + second + .root_lineage(child) + .await + .unwrap() + .unwrap() + .predecessors(), + &[root(1, 1), root(2, 2), root(3, 3)] + ); +} diff --git a/crates/cellule-runtime/src/control/authority/lineage/verify.rs b/crates/cellule-runtime/src/control/authority/lineage/verify.rs new file mode 100644 index 00000000..2d5dc3c5 --- /dev/null +++ b/crates/cellule-runtime/src/control/authority/lineage/verify.rs @@ -0,0 +1,130 @@ +use super::*; + +/// Verified preparation path and complete origin dependency availability. +/// +/// This point observation proves derivation from an exact root through native +/// verified preparations, including compaction. It grants no authority, retention +/// pin, current native serving, original fleet scope or role settlement. Callers +/// must separately authenticate backend mappings and recheck selected authority, +/// boot/operation scope, recovery suffixes and current native work. +#[derive(Debug)] +pub struct VerifiedRootPrefix { + prefix: RootRef, + root: RootRef, + inspected: usize, + objects: usize, +} +impl VerifiedRootPrefix { + /// Exact required historical root. + #[must_use] + pub const fn prefix(&self) -> RootRef { + self.prefix + } + /// Exact successor whose origin graph was completely verified. + #[must_use] + pub const fn root(&self) -> RootRef { + self.root + } + /// Distinct lineage records read while finding the path. + #[must_use] + pub const fn inspected_roots(&self) -> usize { + self.inspected + } + /// Complete successor dependency count, including the root. + #[must_use] + pub const fn dependency_count(&self) -> usize { + self.objects + } +} + +impl CellAuthority { + /// Proves native verified derivation and current origin availability for one + /// exact successor. Limits all expanded and queued lineage roots; + /// The origin inventory is capped at 10,000 dependencies and refuses rather + /// than truncating. The caller owns memory admission and its finite deadline; + /// use the runtime wrapper to charge the shared node memory ledger. + pub async fn verify_root_prefix( + &self, + prefix: RootRef, + root: RootRef, + replica: &CellReplica, + limit: usize, + ) -> Result { + validate_root(&prefix)?; + validate_root(&root)?; + if limit == 0 || limit > MAX_LINEAGE_ROOTS { + return Err(Error::Capacity("invalid Cell root lineage traversal bound")); + } + if prefix.cell != root.cell + || prefix.incarnation != root.incarnation + || replica.scope() != (root.cell, root.incarnation) + { + return Err(Error::Control("Cell root prefix scope differs")); + } + let mut pending = vec![root]; + let mut seen = BTreeMap::from([(root.digest, root)]); + let mut missing = None; + let mut inspected = 0; + let mut reached = false; + while let Some(candidate) = pending.pop() { + if candidate == prefix { + reached = true; + break; + } + if candidate.commit_sequence < prefix.commit_sequence + || candidate.position.txid < prefix.position.txid + { + continue; + } + let Some(record) = self.root_lineage(candidate).await? else { + missing.get_or_insert(candidate); + continue; + }; + inspected += 1; + if record.predecessors.contains(&prefix) { + reached = true; + break; + } + for parent in record.predecessors { + match seen.entry(parent.digest) { + std::collections::btree_map::Entry::Occupied(entry) + if entry.get() != &parent => + { + return Err(Error::Control("Cell root lineage digest changed position")); + } + std::collections::btree_map::Entry::Occupied(_) => {} + std::collections::btree_map::Entry::Vacant(entry) => { + entry.insert(parent); + pending.push(parent); + } + } + if seen.len() > limit { + return Err(Error::Capacity( + "Cell root lineage traversal bound exceeded", + )); + } + } + } + if !reached { + return Err(missing.map_or( + Error::RootPrefixUnproven { + prefix: Box::new(prefix), + root: Box::new(root), + }, + |root| Error::RootLineageIncomplete { root }, + )); + } + // Never reuse the metadata cache as availability evidence: authenticate + // every currently stored byte/extent through the ordinary origin walk. + let objects = replica + .reachable_objects_bounded(&root, MAX_LINEAGE_ROOTS) + .await? + .len(); + Ok(VerifiedRootPrefix { + prefix, + root, + inspected, + objects, + }) + } +} diff --git a/crates/cellule-runtime/src/control/authority.rs b/crates/cellule-runtime/src/control/authority/mod.rs similarity index 90% rename from crates/cellule-runtime/src/control/authority.rs rename to crates/cellule-runtime/src/control/authority/mod.rs index 8ff2819b..3b1afcc3 100644 --- a/crates/cellule-runtime/src/control/authority.rs +++ b/crates/cellule-runtime/src/control/authority/mod.rs @@ -11,6 +11,17 @@ use crate::{Error, Result}; const MAX_CONTROL_BYTES: u64 = 8 * 1024; +mod acquisition; +pub use acquisition::{CellAcquisitionRecord, VerifiedRecoveryPrefix}; +mod history; +pub use history::CellOwnerHistory; +mod lineage; +pub(crate) use lineage::MAX_LINEAGE_ROOTS; +pub use lineage::{CellRootLineage, VerifiedRootPrefix}; + +#[cfg(test)] +mod tests; + /// One exact control observation and its conditional-write token. #[derive(Clone)] pub struct VersionedControl { @@ -166,6 +177,12 @@ impl CellAuthority { transition: Transition, ) -> Result { observed.value.validate_transition(&next, transition)?; + if observed.value.owner.is_some() && observed.value.owner != next.owner { + // Preserve the original full control before the sole authority CAS + // can erase its owner. A lost history reply cannot permit departure. + // Stale proposals may retain observations but grant no ownership. + self.retain_owner(&observed.value).await?; + } let path = self.layout.control_path(observed.value.cell.as_bytes()); let token = self .layout diff --git a/crates/cellule-runtime/src/control/authority/tests.rs b/crates/cellule-runtime/src/control/authority/tests.rs new file mode 100644 index 00000000..5a9fa4ae --- /dev/null +++ b/crates/cellule-runtime/src/control/authority/tests.rs @@ -0,0 +1,754 @@ +use super::*; +use crate::control::{ControlState, RootRef}; +use crate::identity::{Digest, SessionId}; +use cellule_store::Store; +use futures_util::stream::BoxStream; +use object_store::{ + CopyOptions, GetOptions, GetResult, ListResult, MultipartUpload, ObjectMeta, ObjectStore, + PutMultipartOptions, PutOptions, PutPayload, PutResult, memory::InMemory, path::Path, +}; +use std::sync::{ + Arc, + atomic::{AtomicUsize, Ordering}, +}; +use tokio::sync::Notify; + +#[derive(Debug, Default)] +pub(super) struct FaultStore { + inner: InMemory, + pub(super) fault: AtomicUsize, + pub(super) entered: Notify, + pub(super) resume: Notify, + history_writes: AtomicUsize, + control_writes: AtomicUsize, + pub(super) lineage_reads: AtomicUsize, + pub(super) lineage_writes: AtomicUsize, + pub(super) lineage_update_fault: AtomicUsize, + pub(super) root_writes: AtomicUsize, + pub(super) root_written: Notify, + pub(super) root_fault: AtomicUsize, +} + +impl std::fmt::Display for FaultStore { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str("owner-history-fault-store") + } +} + +fn denied(_path: &Path) -> object_store::Error { + object_store::Error::NotSupported { + source: Box::new(std::io::Error::new( + std::io::ErrorKind::PermissionDenied, + "original owner history failure", + )), + } +} + +#[async_trait::async_trait] +impl ObjectStore for FaultStore { + async fn put_opts( + &self, + path: &Path, + body: PutPayload, + opts: PutOptions, + ) -> object_store::Result { + let history = path.as_ref().contains("/owner-history/"); + let control = path.as_ref().ends_with("/control.json"); + if history { + self.history_writes.fetch_add(1, Ordering::SeqCst); + } + if control { + self.control_writes.fetch_add(1, Ordering::SeqCst); + } + if path.as_ref().contains("/root-lineage/") { + self.lineage_writes.fetch_add(1, Ordering::SeqCst); + } + let fault = if history + || path.as_ref().contains("/acquisitions/") + || path.as_ref().contains("/root-lineage/") + { + self.fault + .try_update(Ordering::SeqCst, Ordering::SeqCst, |fault| { + (1..=3).contains(&fault).then_some(0) + }) + .unwrap_or(0) + } else { + 0 + }; + let update_fault = if path.as_ref().contains("/root-lineage/") + && matches!(&opts.mode, object_store::PutMode::Update(_)) + { + self.lineage_update_fault.swap(0, Ordering::SeqCst) + } else { + 0 + }; + let fault = if update_fault == 0 { + fault + } else { + update_fault + }; + if fault == 1 { + return Err(denied(path)); + } + if fault == 3 { + self.entered.notify_one(); + self.resume.notified().await; + } + if path.as_ref().ends_with(".root") { + match self.root_fault.swap(0, Ordering::SeqCst) { + 1 => return Err(denied(path)), + 3 => { + self.entered.notify_one(); + self.resume.notified().await; + } + _ => {} + } + } + let result = self.inner.put_opts(path, body, opts).await?; + if path.as_ref().ends_with(".root") { + self.root_writes.fetch_add(1, Ordering::SeqCst); + self.root_written.notify_one(); + } + if fault == 2 { + return Err(denied(path)); + } + if control + && self + .fault + .compare_exchange(4, 0, Ordering::SeqCst, Ordering::SeqCst) + .is_ok() + { + return Err(denied(path)); + } + Ok(result) + } + async fn put_multipart_opts( + &self, + path: &Path, + opts: PutMultipartOptions, + ) -> object_store::Result> { + self.inner.put_multipart_opts(path, opts).await + } + async fn get_opts(&self, path: &Path, opts: GetOptions) -> object_store::Result { + if path.as_ref().contains("/root-lineage/") { + self.lineage_reads.fetch_add(1, Ordering::SeqCst); + } + if path.as_ref().contains("/acquisitions/") + && self + .fault + .compare_exchange(6, 0, Ordering::SeqCst, Ordering::SeqCst) + .is_ok() + { + return Err(denied(path)); + } + if path.as_ref().contains("/owner-history/") + && self + .fault + .compare_exchange(5, 0, Ordering::SeqCst, Ordering::SeqCst) + .is_ok() + { + self.entered.notify_one(); + self.resume.notified().await; + } + self.inner.get_opts(path, opts).await + } + fn delete_stream( + &self, + paths: BoxStream<'static, object_store::Result>, + ) -> BoxStream<'static, object_store::Result> { + self.inner.delete_stream(paths) + } + fn list(&self, prefix: Option<&Path>) -> BoxStream<'static, object_store::Result> { + self.inner.list(prefix) + } + async fn list_with_delimiter(&self, prefix: Option<&Path>) -> object_store::Result { + self.inner.list_with_delimiter(prefix).await + } + async fn copy_opts( + &self, + from: &Path, + to: &Path, + opts: CopyOptions, + ) -> object_store::Result<()> { + self.inner.copy_opts(from, to, opts).await + } +} + +struct Fixture { + store: Arc, + authority: CellAuthority, + original: VersionedControl, +} + +impl Fixture { + async fn new(published: bool) -> Self { + let store = Arc::new(FaultStore::default()); + let authority = CellAuthority::new(CellStorageLayout::new( + Store::new(store.clone()), + Path::from("owner-test"), + [7; 16], + )); + let mut control = Control::initial( + CellId::from_bytes([1; 32]), + IncarnationId::from_bytes([2; 16]), + owner(3), + Digest::from_bytes([4; 32]), + 1, + ) + .unwrap(); + if published { + control.root = Some(RootRef { + digest: Digest::from_bytes([8; 32]), + txid: 7, + checksum: cellule_ltx::types::CHECKSUM_FLAG | 7, + commit_sequence: 7, + }); + control.state = ControlState::Serving; + } + authority + .layout + .store() + .create_strict( + &authority.layout.control_path(control.cell.as_bytes()), + Bytes::from(control.encode().unwrap()), + ) + .await + .unwrap(); + let original = authority.load(control.cell).await.unwrap().unwrap(); + Self { + store, + authority, + original, + } + } + fn cell(&self) -> CellId { + self.original.value.cell + } + fn history_path(&self, epoch: u64) -> Path { + self.authority.layout.owner_observation_path( + self.cell().as_bytes(), + self.original.value.incarnation.as_bytes(), + epoch, + ) + } +} + +fn owner(byte: u8) -> Owner { + Owner { + session: SessionId::from_bytes([byte; 16]), + endpoint: format!("https://node-{byte}.test"), + } +} + +#[tokio::test] +async fn bounded_canonical_history_rejects_conflicting_or_malformed_observations() { + let f = Fixture::new(false).await; + assert!(matches!( + f.authority.owner_history(f.cell(), 0).await, + Err(Error::Capacity(_)) + )); + for body in [ + Bytes::from_static(b"{}"), + Bytes::from(vec![b' '; 8193]), + Bytes::from(format!( + "{}\n", + String::from_utf8(f.original.value.encode().unwrap()).unwrap() + )), + ] { + f.authority + .layout + .store() + .create_strict(&f.history_path(1), body) + .await + .unwrap(); + assert!( + f.authority + .owner_observation(f.cell(), f.original.value.incarnation, 1) + .await + .is_err() + ); + f.authority + .layout + .store() + .delete(&f.history_path(1)) + .await + .unwrap(); + } + let mut conflicting = f.original.value.clone(); + conflicting.owner = Some(owner(9)); + f.authority + .layout + .store() + .create_strict( + &f.history_path(1), + Bytes::from(conflicting.encode().unwrap()), + ) + .await + .unwrap(); + assert!(matches!( + f.authority + .transition( + &f.original, + f.original.value.takeover(owner(5)).unwrap(), + Transition::Takeover + ) + .await, + Err(Error::Control(_)) + )); + assert_eq!( + f.authority.load(f.cell()).await.unwrap().unwrap().value, + f.original.value + ); + assert_eq!(f.store.control_writes.load(Ordering::SeqCst), 1); +} + +fn tombstone(control: &Control) -> Control { + let mut next = control.clone(); + next.state = ControlState::Tombstoned; + next.owner = None; + next.epoch += 1; + next.revision += 1; + next.progress += 1; + next +} + +#[tokio::test] +async fn rootless_and_recovering_takeovers_retain_all_original_epochs() { + let f = Fixture::new(false).await; + let first = f + .authority + .transition( + &f.original, + f.original.value.takeover(owner(5)).unwrap(), + Transition::Takeover, + ) + .await + .unwrap(); + let second = f + .authority + .transition( + &first, + first.value.takeover(owner(6)).unwrap(), + Transition::Takeover, + ) + .await + .unwrap(); + let independent = CellAuthority::new(CellStorageLayout::new( + Store::new(f.store.clone()), + Path::from("owner-test"), + [7; 16], + )); + let history = independent.owner_history(f.cell(), 3).await.unwrap(); + assert_eq!(history.current(), second.value()); + assert_eq!( + history.owners(), + &[ + f.original.value.clone(), + first.value.clone(), + second.value.clone() + ] + ); + assert!( + history + .owners() + .iter() + .all(|control| control.root.is_none()) + ); + assert_eq!(f.store.history_writes.load(Ordering::SeqCst), 2); + assert!(matches!( + independent.owner_history(f.cell(), 2).await, + Err(Error::Capacity(_)) + )); +} + +#[tokio::test] +async fn release_idle_acquisition_and_tombstone_keep_object_covered_history() { + let f = Fixture::new(true).await; + let idle = f + .authority + .transition( + &f.original, + f.original.value.release().unwrap(), + Transition::Release, + ) + .await + .unwrap(); + let history = f.authority.owner_history(f.cell(), 1).await.unwrap(); + assert_eq!(history.owners(), std::slice::from_ref(&f.original.value)); + assert_eq!(history.current(), idle.value()); + let next = f + .authority + .transition( + &idle, + idle.value.takeover(owner(5)).unwrap(), + Transition::Takeover, + ) + .await + .unwrap(); + let retired = f + .authority + .transition(&next, tombstone(&next.value), Transition::Tombstone) + .await + .unwrap(); + let history = f.authority.owner_history(f.cell(), 2).await.unwrap(); + assert_eq!( + history.owners(), + &[f.original.value.clone(), next.value.clone()] + ); + assert_eq!(history.current(), retired.value()); + assert_eq!(f.store.history_writes.load(Ordering::SeqCst), 2); + assert!( + history + .owners() + .iter() + .all(|control| control.root == f.original.value.root) + ); +} + +#[tokio::test] +async fn idle_tombstone_does_not_invent_an_owner_epoch() { + let f = Fixture::new(true).await; + let idle = f + .authority + .transition( + &f.original, + f.original.value.release().unwrap(), + Transition::Release, + ) + .await + .unwrap(); + f.authority + .transition(&idle, tombstone(&idle.value), Transition::Tombstone) + .await + .unwrap(); + assert_eq!( + f.authority + .owner_history(f.cell(), 1) + .await + .unwrap() + .owners(), + std::slice::from_ref(&f.original.value) + ); +} + +#[tokio::test] +async fn same_owner_publication_and_renewal_do_not_write_history() { + let f = Fixture::new(false).await; + let mut published = f.original.value.clone(); + published.state = ControlState::Serving; + published.root = Some(RootRef { + digest: Digest::from_bytes([8; 32]), + txid: 1, + checksum: cellule_ltx::types::CHECKSUM_FLAG | 1, + commit_sequence: 1, + }); + published.revision += 1; + published.progress += 1; + let published = f + .authority + .transition(&f.original, published, Transition::Publish) + .await + .unwrap(); + let renewed = f + .authority + .transition( + &published, + published.value.renew().unwrap(), + Transition::Renew, + ) + .await + .unwrap(); + assert_eq!(f.store.history_writes.load(Ordering::SeqCst), 0); + assert_eq!( + f.authority + .owner_history(f.cell(), 1) + .await + .unwrap() + .owners(), + &[renewed.value] + ); +} + +#[tokio::test] +async fn history_failure_preserves_source_and_prevents_owner_departure() { + let f = Fixture::new(false).await; + f.store.fault.store(1, Ordering::SeqCst); + let error = f + .authority + .transition( + &f.original, + f.original.value.takeover(owner(5)).unwrap(), + Transition::Takeover, + ) + .await + .err() + .unwrap(); + let Error::Storage(source) = error else { + panic!("original storage failure expected"); + }; + let mut cause = std::error::Error::source(&source); + let mut original = None; + while let Some(error) = cause { + if let Some(error) = error.downcast_ref::() { + original = Some(error); + } + cause = error.source(); + } + let original = original.unwrap(); + assert_eq!(original.kind(), std::io::ErrorKind::PermissionDenied); + assert_eq!(original.to_string(), "original owner history failure"); + assert_eq!( + f.authority.load(f.cell()).await.unwrap().unwrap().value, + f.original.value + ); + assert_eq!(f.store.control_writes.load(Ordering::SeqCst), 1); +} + +#[tokio::test] +async fn lost_history_reply_is_confirmed_before_the_authority_cas() { + let f = Fixture::new(false).await; + f.store.fault.store(2, Ordering::SeqCst); + let moved = f + .authority + .transition( + &f.original, + f.original.value.takeover(owner(5)).unwrap(), + Transition::Takeover, + ) + .await + .unwrap(); + let history = f.authority.owner_history(f.cell(), 2).await.unwrap(); + assert_eq!(history.owners(), &[f.original.value.clone(), moved.value]); + assert_eq!(f.store.history_writes.load(Ordering::SeqCst), 1); +} + +#[tokio::test] +async fn a_delayed_original_proposal_cannot_replace_a_newer_observation() { + let f = Fixture::new(false).await; + f.store.fault.store(3, Ordering::SeqCst); + let original = f.original.clone(); + let authority = f.authority.clone(); + let delayed = tokio::spawn(async move { + authority + .transition( + &original, + original.value.takeover(owner(5)).unwrap(), + Transition::Takeover, + ) + .await + }); + f.store.entered.notified().await; + let renewed = f + .authority + .transition( + &f.original, + f.original.value.renew().unwrap(), + Transition::Renew, + ) + .await + .unwrap(); + let moved = f + .authority + .transition( + &renewed, + renewed.value.takeover(owner(6)).unwrap(), + Transition::Takeover, + ) + .await + .unwrap(); + f.store.resume.notify_one(); + assert!(matches!( + delayed.await.unwrap(), + Err(Error::Storage(StorageError::StateConflict { .. })) + )); + let history = f.authority.owner_history(f.cell(), 2).await.unwrap(); + assert_eq!(history.owners(), &[renewed.value.clone(), moved.value]); + assert_eq!( + f.authority + .owner_observation(f.cell(), renewed.value.incarnation, 1) + .await + .unwrap() + .unwrap(), + renewed.value + ); +} + +#[tokio::test] +async fn cancelled_history_waiter_cannot_remove_original_owner() { + let f = Fixture::new(false).await; + f.store.fault.store(3, Ordering::SeqCst); + let original = f.original.clone(); + let authority = f.authority.clone(); + let pending = tokio::spawn(async move { + authority + .transition( + &original, + original.value.takeover(owner(5)).unwrap(), + Transition::Takeover, + ) + .await + }); + f.store.entered.notified().await; + pending.abort(); + let Err(cancelled) = pending.await else { + panic!("waiter must be cancelled"); + }; + assert!(cancelled.is_cancelled()); + assert_eq!( + f.authority.load(f.cell()).await.unwrap().unwrap().value, + f.original.value + ); + let moved = f + .authority + .transition( + &f.original, + f.original.value.takeover(owner(6)).unwrap(), + Transition::Takeover, + ) + .await + .unwrap(); + assert_eq!( + f.authority + .owner_history(f.cell(), 2) + .await + .unwrap() + .owners(), + &[f.original.value.clone(), moved.value] + ); +} + +#[tokio::test] +async fn retained_proposal_supplies_no_departure_proof() { + let f = Fixture::new(false).await; + f.authority.retain_owner(&f.original.value).await.unwrap(); + let history = f.authority.owner_history(f.cell(), 1).await.unwrap(); + assert_eq!(history.current(), f.original.value()); + assert_eq!(history.owners(), std::slice::from_ref(&f.original.value)); + assert!( + f.authority + .owner_observation(f.cell(), f.original.value.incarnation, 1) + .await + .unwrap() + .is_some() + ); +} + +#[tokio::test] +async fn missing_legacy_history_and_invalid_rows_cannot_exclude_original_owners() { + let f = Fixture::new(false).await; + let moved = f + .authority + .transition( + &f.original, + f.original.value.takeover(owner(5)).unwrap(), + Transition::Takeover, + ) + .await + .unwrap(); + f.authority + .layout + .store() + .delete(&f.history_path(1)) + .await + .unwrap(); + assert!(matches!( + f.authority.owner_history(f.cell(), 2).await, + Err(Error::OwnerHistoryIncomplete { epoch: 1, .. }) + )); + for variant in 0..4 { + let mut foreign = f.original.value.clone(); + match variant { + 0 => foreign.cell = CellId::from_bytes([9; 32]), + 1 => foreign.incarnation = IncarnationId::from_bytes([9; 16]), + 2 => foreign.epoch = 2, + _ => { + foreign.state = ControlState::Tombstoned; + foreign.owner = None; + } + } + f.authority + .layout + .store() + .create_strict(&f.history_path(1), Bytes::from(foreign.encode().unwrap())) + .await + .unwrap(); + assert!(matches!( + f.authority.owner_history(f.cell(), 2).await, + Err(Error::Control(_)) + )); + f.authority + .layout + .store() + .delete(&f.history_path(1)) + .await + .unwrap(); + } + assert_eq!( + f.authority.load(f.cell()).await.unwrap().unwrap().value, + moved.value + ); + assert!( + f.authority + .owner_observation(f.cell(), f.original.value.incarnation, 0) + .await + .is_err() + ); +} + +#[tokio::test] +async fn lost_authority_reply_does_not_lose_original_history() { + let f = Fixture::new(true).await; + f.store.fault.store(4, Ordering::SeqCst); + let idle = f.original.value.release().unwrap(); + assert!(matches!( + f.authority + .transition(&f.original, idle.clone(), Transition::Release) + .await, + Err(Error::Storage(_)) + )); + let history = f.authority.owner_history(f.cell(), 1).await.unwrap(); + assert_eq!(history.current(), &idle); + assert_eq!(history.owners(), std::slice::from_ref(&f.original.value)); + assert_eq!(f.store.history_writes.load(Ordering::SeqCst), 1); + assert!(matches!( + f.authority + .transition(&f.original, idle, Transition::Release) + .await, + Err(Error::Storage(StorageError::StateConflict { .. })) + )); + assert_eq!(f.store.history_writes.load(Ordering::SeqCst), 1); +} + +#[tokio::test] +async fn history_read_refuses_authority_progress_during_collection() { + let f = Fixture::new(false).await; + let moved = f + .authority + .transition( + &f.original, + f.original.value.takeover(owner(5)).unwrap(), + Transition::Takeover, + ) + .await + .unwrap(); + f.store.fault.store(5, Ordering::SeqCst); + let authority = f.authority.clone(); + let cell = f.cell(); + let pending = tokio::spawn(async move { authority.owner_history(cell, 2).await }); + f.store.entered.notified().await; + let renewed = f + .authority + .transition(&moved, moved.value.renew().unwrap(), Transition::Renew) + .await + .unwrap(); + f.store.resume.notify_one(); + assert!(matches!(pending.await.unwrap(), Err(Error::Fenced))); + assert_eq!( + f.authority + .owner_history(f.cell(), 2) + .await + .unwrap() + .current(), + renewed.value() + ); +} diff --git a/crates/cellule-runtime/src/coordination/mod.rs b/crates/cellule-runtime/src/coordination/mod.rs index f36f7204..166f24cb 100644 --- a/crates/cellule-runtime/src/coordination/mod.rs +++ b/crates/cellule-runtime/src/coordination/mod.rs @@ -9,6 +9,14 @@ pub(crate) enum AdmissionKind { Query, Resolve, Migration, + LeaseCommand, + LeaseQuery, +} + +impl AdmissionKind { + pub(crate) const fn allowed_while_quiescing(self) -> bool { + matches!(self, Self::LeaseCommand | Self::LeaseQuery | Self::Resolve) + } } /// The adapter intent associated with one in-flight effect. @@ -72,6 +80,7 @@ pub(crate) enum CoordinationInput { fenced: bool, }, BeginDrain, + BeginMaintenanceQuiescence, BeginShutdown, BeginMigration, FinishMigration { @@ -120,6 +129,13 @@ pub(crate) enum CoordinationInput { refreshing: bool, lease_live: bool, }, + BeginMaintenanceInventory { + refreshing: bool, + finalizing: bool, + queue_empty: bool, + publisher_ready: bool, + lease_live: bool, + }, BeginTransferPreflight { queue_empty: bool, publication_idle: bool, @@ -191,6 +207,7 @@ pub(crate) struct CoordinationState { residency: Residency, shutdown_requested: bool, transfer_preparing: bool, + maintenance_quiescing: bool, next_effect_id: u64, pending_effects: BTreeMap, } @@ -213,6 +230,7 @@ impl CoordinationState { residency: Residency::Resident, shutdown_requested: false, transfer_preparing: false, + maintenance_quiescing: false, next_effect_id: 0, pending_effects: BTreeMap::new(), } @@ -231,6 +249,7 @@ impl CoordinationState { residency, shutdown_requested: false, transfer_preparing: false, + maintenance_quiescing: false, next_effect_id: 0, pending_effects: BTreeMap::new(), } @@ -256,6 +275,10 @@ impl CoordinationState { self.transfer_preparing } + pub(crate) fn is_maintenance_quiescing(&self) -> bool { + self.maintenance_quiescing + } + pub(crate) fn is_busy(&self) -> bool { self.busy } @@ -405,6 +428,18 @@ impl CoordinationState { CoordinationDecision::Ignored } } + CoordinationInput::BeginMaintenanceQuiescence => { + if self.is_fenced() { + CoordinationDecision::Reject(RejectReason::Fenced) + } else if self.is_draining() || self.transfer_preparing { + CoordinationDecision::Reject(RejectReason::Draining) + } else { + // Unlike idle preflight, this is sticky even while busy. + // Accepted work and exact lease completions keep scheduling. + self.maintenance_quiescing = true; + CoordinationDecision::Started + } + } CoordinationInput::BeginDrain => { if self.is_fenced() { CoordinationDecision::Reject(RejectReason::Fenced) @@ -483,7 +518,7 @@ impl CoordinationState { CoordinationDecision::Fence } else if self.is_fenced() || self.is_draining() - || self.transfer_preparing + || (self.transfer_preparing && !self.maintenance_quiescing) || self.renewing { CoordinationDecision::Reject(if self.is_fenced() { @@ -550,7 +585,10 @@ impl CoordinationState { CoordinationDecision::Fence } else if self.is_fenced() { CoordinationDecision::Reject(RejectReason::Fenced) - } else if self.is_draining() || self.transfer_preparing { + } else if self.is_draining() + || self.transfer_preparing + || self.maintenance_quiescing + { CoordinationDecision::Reject(RejectReason::Draining) } else if self.busy { CoordinationDecision::Reject(RejectReason::Busy) @@ -588,6 +626,33 @@ impl CoordinationState { CoordinationDecision::Started } } + CoordinationInput::BeginMaintenanceInventory { + refreshing, + finalizing, + queue_empty, + publisher_ready, + lease_live, + } => { + if !lease_live { + self.lifecycle = Lifecycle::Fenced; + CoordinationDecision::Fence + } else if self.is_fenced() { + CoordinationDecision::Reject(RejectReason::Fenced) + } else if self.is_draining() || !self.maintenance_quiescing { + CoordinationDecision::Reject(RejectReason::Draining) + } else if refreshing + || !publisher_ready + || (finalizing + && (!self.transfer_preparing || !queue_empty || !self.can_deactivate())) + { + CoordinationDecision::Ignored + } else { + // The worker serializes this read behind already executing SQL. + // Its owned Inventory effect prevents the next queued work from + // starving the first maintenance observation. + CoordinationDecision::Started + } + } CoordinationInput::BeginTransferPreflight { queue_empty: _, publication_idle: _, @@ -663,6 +728,7 @@ impl CoordinationState { } else if self.is_fenced() || self.is_draining() || self.transfer_preparing + || self.maintenance_quiescing || self.busy || self.renewing || !queue_empty @@ -729,6 +795,9 @@ impl CoordinationState { if self.transfer_preparing { return CoordinationDecision::Reject(RejectReason::Draining); } + if self.maintenance_quiescing && !kind.allowed_while_quiescing() { + return CoordinationDecision::Reject(RejectReason::Draining); + } // Migration swaps the admission capability before its durable cut is // published. The successor capability may queue work, but `busy` keeps // it from executing until `FinishMigration` returns to Serving. diff --git a/crates/cellule-runtime/src/coordination/tests/maintenance.rs b/crates/cellule-runtime/src/coordination/tests/maintenance.rs new file mode 100644 index 00000000..90933337 --- /dev/null +++ b/crates/cellule-runtime/src/coordination/tests/maintenance.rs @@ -0,0 +1,207 @@ +use super::*; + +#[test] +fn final_maintenance_inventory_requires_closed_transfer_admission() { + let mut state = CoordinationState::serving(true); + state.step(CoordinationInput::BeginMaintenanceQuiescence); + let input = CoordinationInput::BeginMaintenanceInventory { + refreshing: false, + finalizing: true, + queue_empty: true, + publisher_ready: true, + lease_live: true, + }; + assert_eq!(state.step(input), CoordinationDecision::Ignored); + state.step(CoordinationInput::BeginTransferPreflight { + queue_empty: true, + publication_idle: true, + lease_live: true, + }); + assert_eq!(state.step(input), CoordinationDecision::Started); +} + +#[test] +fn busy_maintenance_closes_foreground_and_preserves_exact_completion_schedule() { + let mut state = CoordinationState::serving(true); + assert_eq!( + state.step(CoordinationInput::BeginWork { + kind: AdmissionKind::Command, + publisher_ready: true + }), + CoordinationDecision::Started + ); + assert_eq!( + state.step(CoordinationInput::BeginMaintenanceQuiescence), + CoordinationDecision::Started + ); + assert!(state.is_maintenance_quiescing()); + assert!(state.is_busy()); + for kind in [ + AdmissionKind::Command, + AdmissionKind::Query, + AdmissionKind::Migration, + ] { + assert_eq!( + state.step(CoordinationInput::Admit { + kind, + admission_matches: true + }), + CoordinationDecision::Reject(RejectReason::Draining) + ); + } + for kind in [ + AdmissionKind::LeaseCommand, + AdmissionKind::LeaseQuery, + AdmissionKind::Resolve, + ] { + assert_eq!( + state.step(CoordinationInput::Admit { + kind, + admission_matches: true + }), + CoordinationDecision::Admit + ); + } + state.step(CoordinationInput::AbortTransfer); + assert!(state.is_maintenance_quiescing()); + state.step(CoordinationInput::FinishWork { fenced: false }); + assert_eq!( + state.step(CoordinationInput::BeginWork { + kind: AdmissionKind::LeaseCommand, + publisher_ready: true + }), + CoordinationDecision::Started + ); + state.step(CoordinationInput::FinishWork { fenced: false }); + state.step(CoordinationInput::Fence); + assert_eq!( + state.step(CoordinationInput::Admit { + kind: AdmissionKind::LeaseCommand, + admission_matches: true + }), + CoordinationDecision::Reject(RejectReason::Fenced) + ); +} + +#[test] +fn terminal_transfer_and_shutdown_close_even_native_completion_admission() { + for terminal in [ + CoordinationInput::BeginDrain, + CoordinationInput::BeginShutdown, + ] { + let mut state = CoordinationState::serving(true); + state.step(CoordinationInput::BeginMaintenanceQuiescence); + state.step(terminal); + assert_eq!( + state.step(CoordinationInput::Admit { + kind: AdmissionKind::LeaseCommand, + admission_matches: true + }), + CoordinationDecision::Reject(RejectReason::Draining) + ); + } + let mut state = CoordinationState::serving(true); + state.step(CoordinationInput::BeginMaintenanceQuiescence); + state.step(CoordinationInput::BeginTransferPreflight { + queue_empty: true, + publication_idle: true, + lease_live: true, + }); + assert_eq!( + state.step(CoordinationInput::Admit { + kind: AdmissionKind::LeaseCommand, + admission_matches: true + }), + CoordinationDecision::Reject(RejectReason::Draining) + ); +} + +#[test] +fn maintenance_inventory_has_a_serialized_slot_even_during_accepted_work() { + let mut state = CoordinationState::serving(true); + state.step(CoordinationInput::BeginWork { + kind: AdmissionKind::Command, + publisher_ready: true, + }); + let input = CoordinationInput::BeginMaintenanceInventory { + refreshing: false, + finalizing: false, + queue_empty: false, + publisher_ready: true, + lease_live: true, + }; + assert_eq!( + state.step(input), + CoordinationDecision::Reject(RejectReason::Draining) + ); + state.step(CoordinationInput::BeginMaintenanceQuiescence); + assert_eq!(state.step(input), CoordinationDecision::Started); + let effect = state.begin_effect(CoordinationEffect::Inventory); + assert!(!state.can_deactivate()); + state.step(CoordinationInput::CompleteEffect { + effect_id: effect, + effect: CoordinationEffect::Inventory, + }); + state.step(CoordinationInput::BeginTransferPreflight { + queue_empty: false, + publication_idle: true, + lease_live: true, + }); + assert_eq!( + state.step(CoordinationInput::BeginMaintenanceInventory { + refreshing: false, + finalizing: true, + queue_empty: false, + publisher_ready: true, + lease_live: true + }), + CoordinationDecision::Ignored + ); + state.step(CoordinationInput::FinishWork { fenced: false }); + assert_eq!( + state.step(CoordinationInput::BeginMaintenanceInventory { + finalizing: true, + queue_empty: true, + refreshing: false, + publisher_ready: true, + lease_live: true + }), + CoordinationDecision::Started + ); +} + +#[test] +fn quiesced_source_keeps_canonical_renewal_before_final_release() { + let mut state = CoordinationState::serving(true); + state.step(CoordinationInput::BeginMaintenanceQuiescence); + state.step(CoordinationInput::BeginTransferPreflight { + queue_empty: true, + publication_idle: true, + lease_live: true, + }); + assert_eq!( + state.step(CoordinationInput::BeginRenewal { + queue_empty: true, + publication_idle: true, + lease_live: true + }), + CoordinationDecision::Started + ); + assert_eq!( + state.step(CoordinationInput::ConfirmTransfer), + CoordinationDecision::Ignored + ); + state.step(CoordinationInput::FinishRenewal { fenced: false }); + assert_eq!( + state.step(CoordinationInput::ConfirmTransfer), + CoordinationDecision::ReadyToDeactivate + ); + assert!(matches!( + state.step(CoordinationInput::BeginRenewal { + queue_empty: true, + publication_idle: true, + lease_live: true + }), + CoordinationDecision::Reject(RejectReason::Draining) + )); +} diff --git a/crates/cellule-runtime/src/coordination/tests/mod.rs b/crates/cellule-runtime/src/coordination/tests/mod.rs index d61ce1b9..fc32c923 100644 --- a/crates/cellule-runtime/src/coordination/tests/mod.rs +++ b/crates/cellule-runtime/src/coordination/tests/mod.rs @@ -2,6 +2,7 @@ use super::*; mod effects; mod lifecycle; +mod maintenance; mod publication; mod scheduler; mod transfer; diff --git a/crates/cellule-runtime/src/error.rs b/crates/cellule-runtime/src/error.rs index 7dfe350c..91e20a4b 100644 --- a/crates/cellule-runtime/src/error.rs +++ b/crates/cellule-runtime/src/error.rs @@ -10,6 +10,40 @@ pub enum Error { /// A control record or transition failed validation. #[error("invalid Cell control record: {0}")] Control(&'static str), + /// A closed owner epoch lacks the original retained control observation. + #[error("Cell owner history is incomplete at epoch {epoch}")] + OwnerHistoryIncomplete { + /// Cell whose history cannot exclude an original failed owner. + cell: crate::identity::CellId, + /// Current incarnation whose history was requested. + incarnation: crate::identity::IncarnationId, + /// First missing ownership epoch; absence supplies no closure proof. + epoch: u64, + }, + /// A required canonical successful acquisition lacks its retained input. + #[error("Cell acquisition history is incomplete at epoch {epoch}")] + AcquisitionHistoryIncomplete { + /// Cell whose original recovery materialization must be proven. + cell: crate::identity::CellId, + /// Original incarnation, without substituting a new Cell lifetime. + incarnation: crate::identity::IncarnationId, + /// Exact missing successful acquisition epoch. + epoch: u64, + }, + /// A required root lacks its canonical preparation links. + #[error("Cell root lineage is incomplete at {root:?}")] + RootLineageIncomplete { + /// Exact root whose absence cannot establish an acknowledged prefix. + root: cellule_ltx::RootRef, + }, + /// Complete bounded preparation links did not reach the required root. + #[error("Cell root {root:?} has no verified preparation path from {prefix:?}")] + RootPrefixUnproven { + /// Required exact historical root. + prefix: Box, + /// Requested successor root, without any authority grant. + root: Box, + }, /// A catalog record, scan, or head failed validation. #[error("invalid Cell catalog: {0}")] Catalog(&'static str), @@ -74,6 +108,19 @@ pub enum Error { /// The peer is authenticated but not authorized for this operation. #[error("Cell peer authorization denied: {0}")] PeerAuthorization(&'static str), + /// A fleet receiver request failed its exact operation contract. + #[error("Cell fleet operation failed")] + FleetOperation(#[source] Box), + /// This exact source request stopped before canonical deactivation or + /// authority release began. It proves refusal, not another job's outcome. + #[error("Cell source release was refused before deactivation: {blocker:?}")] + CellReleaseRefused { + /// The condition which prevented this request's release. + blocker: crate::fleet::operations::DrainBlocker, + /// Original validation, admission, or inventory error. + #[source] + source: Box, + }, /// A node advertisement or directory record failed validation. #[error("invalid Cell node advertisement: {0}")] Node(&'static str), diff --git a/crates/cellule-runtime/src/fleet/admission/mod.rs b/crates/cellule-runtime/src/fleet/admission/mod.rs new file mode 100644 index 00000000..4985bc78 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/admission/mod.rs @@ -0,0 +1,232 @@ +//! Shared node role-admission reasons and the locally measured stable tier. +//! +//! Pressure recovery only clears pressure. Cordon and terminal drain stay +//! closed for this runtime; a return to service needs a validated new session. + +use std::sync::{Arc, RwLock}; + +use crate::node::{NodeMode, NodeOperationalSample, NodePressure}; +use crate::{Error, Result}; + +use super::pressure::PressureState; + +#[derive(Clone, Copy, Debug)] +struct State { + mode: NodeMode, + pressure: NodePressure, + sequence: u64, + observed_at_ms: Option, + startup: Startup, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum Startup { + Unconfigured, + Held, + Confirmed, +} + +impl State { + fn effective_mode(self) -> NodeMode { + if self.startup == Startup::Held && self.mode == NodeMode::Active { + NodeMode::Cordoned + } else { + self.mode + } + } +} + +/// One node's local writer, new-reader, and new-follower admission gate. +/// Clones share reasons; this state does not authorize or fence Cell ownership. +#[derive(Clone, Debug)] +pub struct NodeAdmission { + state: Arc>, +} + +impl Default for NodeAdmission { + fn default() -> Self { + Self { + state: Arc::new(RwLock::new(State { + mode: NodeMode::Active, + pressure: NodePressure::Normal, + sequence: 0, + observed_at_ms: None, + startup: Startup::Unconfigured, + })), + } + } +} + +impl NodeAdmission { + /// Reports whether boot confirmation still holds this runtime's startup. + /// Cordon and pressure remain separate; outbound replacement of existing + /// responsibilities may proceed after a maintenance boot is confirmed. + pub fn startup_held(&self) -> Result { + Ok(self + .state + .read() + .map_err(|_| Error::Node("node admission lock poisoned"))? + .startup + == Startup::Held) + } + /// Returns the lifecycle reason independent of pressure. + pub fn mode(&self) -> Result { + Ok(self + .state + .read() + .map_err(|_| Error::Node("node admission lock poisoned"))? + .effective_mode()) + } + + /// Checks current local admission; a stale peer advertisement cannot reopen it. + pub fn check_new_role(&self) -> Result<()> { + let state = self + .state + .read() + .map_err(|_| Error::Node("node admission lock poisoned"))?; + if state.effective_mode() != NodeMode::Active { + return Err(Error::CellDraining); + } + if state.pressure != NodePressure::Normal { + return Err(Error::Capacity("node pressure")); + } + Ok(()) + } + + /// Holds every new role before a fleet boot is enrolled. Install before + /// exposing the runtime or installing its required node lease. The hold + /// reports Cordoned until confirmation; it does not refresh measurements. + pub fn hold_startup(&self) -> Result<()> { + let mut state = self + .state + .write() + .map_err(|_| Error::Node("node admission lock poisoned"))?; + if state.startup != Startup::Unconfigured { + return Err(Error::Node("node startup admission already configured")); + } + let sequence = state + .sequence + .checked_add(1) + .ok_or(Error::Node("node admission sequence overflow"))?; + state.startup = Startup::Held; + state.sequence = sequence; + Ok(()) + } + + /// Confirms application-checked durable boot enrollment in its retained + /// mode. This only removes the startup hold: cordon, drain and pressure + /// remain independent. Returning to service requires a new runtime/session. + pub fn confirm_startup(&self, mode: NodeMode) -> Result<()> { + let mut state = self + .state + .write() + .map_err(|_| Error::Node("node admission lock poisoned"))?; + if state.startup == Startup::Unconfigured { + return Err(Error::Node("node startup admission is not configured")); + } + let mode = match (state.mode, mode) { + (NodeMode::Draining, _) | (_, NodeMode::Draining) => NodeMode::Draining, + (NodeMode::Cordoned, _) | (_, NodeMode::Cordoned) => NodeMode::Cordoned, + _ => NodeMode::Active, + }; + if state.startup == Startup::Confirmed && state.mode == mode { + return Ok(()); + } + let sequence = state + .sequence + .checked_add(1) + .ok_or(Error::Node("node admission sequence overflow"))?; + state.mode = mode; + state.startup = Startup::Confirmed; + state.sequence = sequence; + Ok(()) + } + + /// Closes new role admission while preserving all existing obligations. + pub fn cordon(&self) -> Result<()> { + self.advance_mode(NodeMode::Cordoned) + } + + /// Closes new role admission for terminal lifecycle drain. + pub fn begin_drain(&self) -> Result<()> { + self.advance_mode(NodeMode::Draining) + } + + fn advance_mode(&self, requested: NodeMode) -> Result<()> { + let mut state = self + .state + .write() + .map_err(|_| Error::Node("node admission lock poisoned"))?; + if state.mode == requested || state.mode == NodeMode::Draining { + return Ok(()); + } + let sequence = state + .sequence + .checked_add(1) + .ok_or(Error::Node("node admission sequence overflow"))?; + state.mode = requested; + state.sequence = sequence; + // A lifecycle change does not refresh an old pressure measurement. + Ok(()) + } + + /// Returns the latest real sample, or unknown before the first observation. + /// Repeated reads do not advance its sequence or measurement timestamp. + pub fn sample(&self) -> Result> { + let state = self + .state + .read() + .map_err(|_| Error::Node("node admission lock poisoned"))?; + Ok(state + .observed_at_ms + .map(|observed_at_ms| NodeOperationalSample { + mode: state.effective_mode(), + pressure: state.pressure, + sequence: state.sequence, + observed_at_ms, + })) + } + + pub(crate) fn observe(&self, pressure: PressureState, at_ms: i64) -> Result<()> { + let pressure = NodePressure::try_from(pressure)?; + let mut state = self + .state + .write() + .map_err(|_| Error::Node("node admission lock poisoned"))?; + if at_ms < 0 + || state + .observed_at_ms + .is_some_and(|previous| at_ms < previous) + { + return Err(Error::Node("node admission sample time regressed")); + } + let sequence = state + .sequence + .checked_add(1) + .ok_or(Error::Node("node admission sequence overflow"))?; + state.pressure = pressure; + state.observed_at_ms = Some(at_ms); + state.sequence = sequence; + Ok(()) + } + + /// Performs a short synchronous admission effect atomically with cordon. + /// Used to enroll a new follower lane; existing lanes bypass this new-role + /// gate so their acknowledged tails can continue to settle under pressure. + pub(crate) fn admit(&self, effect: impl FnOnce() -> Result) -> Result { + let state = self + .state + .read() + .map_err(|_| Error::Node("node admission lock poisoned"))?; + if state.effective_mode() != NodeMode::Active { + return Err(Error::CellDraining); + } + if state.pressure != NodePressure::Normal { + return Err(Error::Capacity("node pressure")); + } + effect() + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-runtime/src/fleet/admission/tests.rs b/crates/cellule-runtime/src/fleet/admission/tests.rs new file mode 100644 index 00000000..1f14788c --- /dev/null +++ b/crates/cellule-runtime/src/fleet/admission/tests.rs @@ -0,0 +1,110 @@ +use super::*; + +#[test] +fn pressure_recovery_cannot_clear_cordon_or_shutdown() { + let gate = NodeAdmission::default(); + assert!(gate.check_new_role().is_ok()); + assert_eq!(gate.sample().unwrap(), None); + gate.observe(PressureState::Shedding, 100).unwrap(); + assert!(gate.check_new_role().is_err()); + gate.observe(PressureState::Normal, 200).unwrap(); + assert!(gate.check_new_role().is_ok()); + gate.cordon().unwrap(); + gate.observe(PressureState::Normal, 300).unwrap(); + assert_eq!(gate.mode().unwrap(), NodeMode::Cordoned); + assert!(gate.check_new_role().is_err()); + gate.begin_drain().unwrap(); + gate.cordon().unwrap(); + gate.observe(PressureState::Normal, 400).unwrap(); + assert_eq!(gate.mode().unwrap(), NodeMode::Draining); + assert!(gate.check_new_role().is_err()); +} + +#[test] +fn samples_are_shared_monotonic_and_not_refreshed_by_read_or_cordon() { + let gate = NodeAdmission::default(); + let clone = gate.clone(); + clone.observe(PressureState::Constrained, 100).unwrap(); + let old = gate.sample().unwrap().unwrap(); + assert_eq!(clone.sample().unwrap(), Some(old)); + assert!(gate.observe(PressureState::Normal, 99).is_err()); + assert!(gate.observe(PressureState::Recovering, 101).is_err()); + assert_eq!(gate.sample().unwrap(), Some(old)); + clone.cordon().unwrap(); + let cordoned = gate.sample().unwrap().unwrap(); + assert_eq!(cordoned.sequence, old.sequence + 1); + assert_eq!(cordoned.observed_at_ms, old.observed_at_ms); + gate.cordon().unwrap(); + assert_eq!(gate.sample().unwrap(), Some(cordoned)); +} + +#[test] +fn startup_hold_uses_shared_role_gate_and_preserves_measurement_time() { + let gate = NodeAdmission::default(); + gate.observe(PressureState::Normal, 100).unwrap(); + let before = gate.sample().unwrap().unwrap(); + gate.hold_startup().unwrap(); + let held = gate.sample().unwrap().unwrap(); + assert_eq!(held.mode, NodeMode::Cordoned); + assert_eq!(held.observed_at_ms, before.observed_at_ms); + assert_eq!(held.sequence, before.sequence + 1); + assert!(matches!(gate.check_new_role(), Err(Error::CellDraining))); + assert!(matches!( + gate.admit::<()>(|| panic!("startup admitted a follower")), + Err(Error::CellDraining) + )); + assert!(gate.hold_startup().is_err()); + gate.confirm_startup(NodeMode::Active).unwrap(); + assert!(gate.check_new_role().is_ok()); + let confirmed = gate.sample().unwrap().unwrap(); + assert_eq!(confirmed.mode, NodeMode::Active); + assert_eq!(confirmed.observed_at_ms, before.observed_at_ms); + assert_eq!(confirmed.sequence, held.sequence + 1); + gate.confirm_startup(NodeMode::Active).unwrap(); + assert_eq!(gate.sample().unwrap(), Some(confirmed)); +} + +#[test] +fn startup_confirmation_cannot_clear_a_racing_cordon_drain_or_pressure() { + for mode in [NodeMode::Cordoned, NodeMode::Draining] { + let gate = NodeAdmission::default(); + gate.hold_startup().unwrap(); + if mode == NodeMode::Cordoned { + gate.cordon().unwrap(); + } else { + gate.begin_drain().unwrap(); + } + gate.confirm_startup(NodeMode::Active).unwrap(); + assert_eq!(gate.mode().unwrap(), mode); + assert!(matches!(gate.check_new_role(), Err(Error::CellDraining))); + gate.observe(PressureState::Normal, 100).unwrap(); + gate.confirm_startup(NodeMode::Active).unwrap(); + assert_eq!(gate.mode().unwrap(), mode); + } + let gate = NodeAdmission::default(); + assert!(gate.confirm_startup(NodeMode::Active).is_err()); + gate.hold_startup().unwrap(); + gate.observe(PressureState::Shedding, 100).unwrap(); + gate.confirm_startup(NodeMode::Active).unwrap(); + assert_eq!(gate.mode().unwrap(), NodeMode::Active); + assert!(matches!( + gate.check_new_role(), + Err(Error::Capacity("node pressure")) + )); +} + +#[test] +fn startup_hold_is_distinct_from_confirmed_maintenance_mode() { + for mode in [NodeMode::Active, NodeMode::Cordoned, NodeMode::Draining] { + let gate = NodeAdmission::default(); + assert!(!gate.startup_held().unwrap()); + gate.hold_startup().unwrap(); + assert!(gate.startup_held().unwrap()); + gate.confirm_startup(mode).unwrap(); + assert!(!gate.startup_held().unwrap()); + assert_eq!(gate.mode().unwrap(), mode); + if mode != NodeMode::Active { + assert!(gate.check_new_role().is_err()); + } + } +} diff --git a/crates/cellule-runtime/src/fleet/mod.rs b/crates/cellule-runtime/src/fleet/mod.rs index 5a321073..803801db 100644 --- a/crates/cellule-runtime/src/fleet/mod.rs +++ b/crates/cellule-runtime/src/fleet/mod.rs @@ -1,6 +1,8 @@ //! Fleet placement, pressure, admission accounting, eviction, and scheduling. +pub mod admission; pub mod eviction; +pub mod operations; pub mod placement; pub mod pressure; pub mod resource; diff --git a/crates/cellule-runtime/src/fleet/operations/accepted.rs b/crates/cellule-runtime/src/fleet/operations/accepted.rs new file mode 100644 index 00000000..af4ccb72 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/accepted.rs @@ -0,0 +1,199 @@ +use crate::identity::{NodeId, SessionId}; + +use super::{ + FleetAction, FleetActionKind, FleetActionOutcome, FleetHead, MovementAction, OperationError, + Result, nonzero, +}; + +/// Immutable acceptance of one exact local fleet effect. +/// +/// Construct and publish this record in the same journal transaction that +/// checks the current head. Constructing or decoding it confers no authority +/// and does not authenticate a caller. Retain it across controller replacement. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct AcceptedFleetAction { + pub(super) action: FleetAction, + pub(super) node: NodeId, + pub(super) session: SessionId, + pub(super) accepted_at_ms: i64, +} + +impl AcceptedFleetAction { + /// Checks first acceptance against the exact head and executing endpoint. + /// The journal must atomically publish this check with acceptance. + pub fn new( + action: FleetAction, + head: &FleetHead, + node: NodeId, + session: SessionId, + now_ms: i64, + ) -> Result { + action.authorize_against(head, now_ms)?; + let accepted = Self { + action, + node, + session, + accepted_at_ms: now_ms, + }; + accepted.validate()?; + Ok(accepted) + } + + /// Returns the originally accepted envelope, including its authorization. + #[must_use] + pub const fn action(&self) -> &FleetAction { + &self.action + } + /// Returns the physical node bound at acceptance. + #[must_use] + pub const fn node(&self) -> NodeId { + self.node + } + /// Returns the boot session bound at acceptance. + #[must_use] + pub const fn session(&self) -> SessionId { + self.session + } + /// Returns the checked first-acceptance time. + #[must_use] + pub const fn accepted_at_ms(&self) -> i64 { + self.accepted_at_ms + } + + /// Checks replay identity while permitting refreshed controller authorization. + /// + /// The stable action key is an index, not a complete payload comparison. + /// An existing acceptance may finish after its original deadline or lease. + /// This check never authorizes first acceptance or another execution. + pub fn validate_replay( + &self, + action: &FleetAction, + node: NodeId, + session: SessionId, + ) -> Result<()> { + self.validate()?; + action.validate()?; + if self.node != node + || self.session != session + || self.action.scope() != action.scope() + || self.action.key()? != action.key()? + { + return Err(OperationError::Conflict); + } + self.action.validate_replay(action) + } + + /// Binds a result to the accepted endpoint and original execution inputs. + /// This validates shape; canonical execution and fresh serving checks remain + /// the host's responsibility. + pub fn validate_result(&self, result: &FleetActionOutcome) -> Result<()> { + self.validate()?; + result.validate_for(&self.action)?; + if let super::FleetOutcome::Recovered(evidence) = &result.outcome + && matches!( + self.action.kind(), + FleetActionKind::Movement { + action: MovementAction::Recover, + .. + } + ) + { + evidence.recovery.basis().validate_acceptance(self)?; + } + if result.node != self.node + || result.session != self.session + || result.observed_at_ms < self.accepted_at_ms + { + return Err(OperationError::Invalid("fleet result acceptance mismatch")); + } + Ok(()) + } + + pub(super) fn validate(&self) -> Result<()> { + self.action.validate_endpoint(self.node, self.session)?; + if self.accepted_at_ms < self.action.issued_at_ms() { + return Err(OperationError::Invalid("invalid fleet acceptance identity")); + } + self.action.check_admission_deadline(self.accepted_at_ms)?; + Ok(()) + } +} + +impl FleetAction { + /// Compares complete immutable execution inputs across authorization renewal. + /// This does not authorize a side effect or establish durable acceptance. + pub fn validate_replay(&self, action: &Self) -> Result<()> { + self.validate()?; + action.validate()?; + if self.scope() != action.scope() || self.key()? != action.key()? { + return Err(OperationError::Conflict); + } + let same = match (self.kind(), action.kind()) { + ( + FleetActionKind::Movement { + action: original, + attempt: a, + }, + FleetActionKind::Movement { + action: replay, + attempt: b, + }, + ) => original == replay && a.spec() == b.spec(), + ( + FleetActionKind::Maintenance { + action: original, + operation: a, + }, + FleetActionKind::Maintenance { + action: replay, + operation: b, + }, + ) => { + original == replay + && a.id == b.id + && a.request_digest == b.request_digest + && a.node == b.node + && a.session == b.session + && a.intent_revision == b.intent_revision + && a.created_at_ms == b.created_at_ms + && a.deadline_ms == b.deadline_ms + } + _ => false, + }; + if !same { + return Err(OperationError::Conflict); + } + Ok(()) + } + + /// Checks the exact local endpoint without conferring journal authorization. + pub fn validate_endpoint(&self, node: NodeId, session: SessionId) -> Result<()> { + self.validate()?; + if !nonzero(node.as_bytes()) || !nonzero(session.as_bytes()) { + return Err(OperationError::Invalid("invalid fleet acceptance identity")); + } + let endpoint = match self.kind() { + FleetActionKind::Movement { action, attempt } => { + let spec = attempt.spec(); + let source = node == spec.source_node && session == spec.source; + let receiver = node == spec.destination_node && session == spec.destination; + match action { + MovementAction::Release | MovementAction::ReleaseMaintenance => source, + MovementAction::Prepare + | MovementAction::Activate + | MovementAction::Cancel + | MovementAction::Recover => receiver, + MovementAction::Inspect => source || receiver, + MovementAction::Retire => false, + } + } + FleetActionKind::Maintenance { operation, .. } => { + node == operation.node() && session == operation.session() + } + }; + if !endpoint { + return Err(OperationError::Fenced); + } + Ok(()) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/acquisition.rs b/crates/cellule-runtime/src/fleet/operations/acquisition.rs new file mode 100644 index 00000000..11e5cf17 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/acquisition.rs @@ -0,0 +1,125 @@ +use crate::control::{Control, ControlState}; + +use super::{ + AcceptedFleetAction, FleetActionKind, FleetActionOutcome, FleetOutcome, MovementAction, + OperationError, PublishedPosition, Result, +}; + +/// Immutable checked Idle input retained before a prepared acquisition CAS. +/// +/// The host records its actual authority observation under the accepted action +/// before invoking canonical acquisition. This record proves neither CAS +/// success nor serving. It is distinct from failed-owner recovery evidence. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct AcquisitionBasis { + pub(super) accepted: AcceptedFleetAction, + pub(super) control: Control, + pub(super) observed_at_ms: i64, +} + +impl AcquisitionBasis { + /// Checks the observed Idle control against an accepted receiver activation. + pub fn new( + accepted: AcceptedFleetAction, + control: Control, + observed_at_ms: i64, + ) -> Result { + let basis = Self { + accepted, + control, + observed_at_ms, + }; + basis.validate()?; + Ok(basis) + } + + /// Returns the exact original accepted receiver action. + #[must_use] + pub const fn accepted(&self) -> &AcceptedFleetAction { + &self.accepted + } + /// Returns the canonical Idle control read before takeover. + #[must_use] + pub const fn control(&self) -> &Control { + &self.control + } + /// Returns the capture time without refreshing the historical observation. + #[must_use] + pub const fn observed_at_ms(&self) -> i64 { + self.observed_at_ms + } + + /// Returns the exact immutable position of the acquisition input. + pub fn position(&self) -> Result { + Ok(PublishedPosition { + incarnation: self.control.incarnation, + epoch: self.control.epoch, + root: self + .control + .root + .clone() + .ok_or(OperationError::Invalid("acquisition lacks root"))?, + }) + } + + /// Checks that a receiver's activation result follows this recorded basis. + /// Fresh authority and actor checks still establish current serving. + pub fn validate_result(&self, result: &FleetActionOutcome) -> Result<()> { + self.validate()?; + self.accepted.validate_result(result)?; + if let FleetOutcome::Activated(evidence) = &result.outcome { + let input = self.position()?; + if result.observed_at_ms < self.observed_at_ms + || evidence.position.epoch <= input.epoch + || !super::attempt::successor_position(&evidence.position, &input) + { + return Err(OperationError::Invalid( + "activation does not follow acquisition basis", + )); + } + } + Ok(()) + } + + pub(super) fn validate(&self) -> Result<()> { + self.accepted.validate()?; + self.control + .encode() + .map_err(|error| OperationError::Control(Box::new(error)))?; + let FleetActionKind::Movement { + action: MovementAction::Activate, + attempt, + } = self.accepted.action().kind() + else { + return Err(OperationError::Invalid( + "acquisition requires accepted activation", + )); + }; + let spec = attempt.spec(); + if self.observed_at_ms < self.accepted.accepted_at_ms() + || self.control.cell != spec.target.cell_id() + || self.control.incarnation != spec.incarnation + || self.control.state != ControlState::Idle + || self.control.owner.is_some() + || self.control.recovery.is_some() + || self.control.epoch < spec.source_epoch + { + return Err(OperationError::Invalid( + "acquisition control scope or state mismatch", + )); + } + let input = self.position()?; + input.validate()?; + let released = attempt + .released() + .ok_or(OperationError::Invalid("acquisition lacks source position"))?; + if !super::attempt::successor_position(&input, released) + || (input.epoch == released.epoch && input.root != released.root) + { + return Err(OperationError::Invalid( + "acquisition input precedes or changes release", + )); + } + Ok(()) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/actions.rs b/crates/cellule-runtime/src/fleet/operations/actions.rs new file mode 100644 index 00000000..6cd5e507 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/actions.rs @@ -0,0 +1,536 @@ +use crate::identity::{Digest, NodeId, SessionId}; + +use super::{ + ActivationEvidence, AttemptId, AttemptPhase, DrainBlocker, DrainEvidence, FleetHead, + FleetScope, MaintenanceOperation, MaintenancePhase, MoveAttempt, MovementAction, + OperationError, PublishedPosition, ReceiverReservation, RecoveredActivation, Result, nonzero, +}; + +/// Node lifecycle work recorded by one maintenance operation. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[repr(u8)] +pub enum MaintenanceAction { + /// Apply the persisted physical-node intent before new role admission. + Cordon = 1, + /// Reconcile reader replacements and foreign follower obligations. + SettleRoles = 2, + /// Complete the existing host drain after relocation is proven. + Finalize = 3, + /// Reconstruct actual progress without starting a new role transition. + Inspect = 4, +} + +/// Exact action payload cloned from the current committed journal state. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum FleetActionKind { + /// Execute or inspect one immutable session/generation-bound movement. + Movement { + /// Remote effect or inspection; retirement is journal-local. + action: MovementAction, + /// Current charged attempt and its dispatch/evidence state. + attempt: Box, + }, + /// Execute or inspect a physical-node maintenance operation. + Maintenance { + /// Lifecycle effect to reconcile. + action: MaintenanceAction, + /// Current committed operation and targeted session. + operation: Box, + }, +} + +/// Bounded journal-bound action envelope. It is not an authentication capability. +/// +/// The receiving application authenticates the caller and reads the journal +/// before `authorize_against`. Existing accepted actions may finish after a +/// controller change; accepting a new effect requires the current live epoch. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct FleetAction { + pub(super) scope: FleetScope, + pub(super) journal_revision: u64, + pub(super) controller: SessionId, + pub(super) controller_epoch: u64, + pub(super) issued_at_ms: i64, + pub(super) kind: FleetActionKind, +} + +impl FleetAction { + /// Returns the exact fleet/application authorization scope. + #[must_use] + pub const fn scope(&self) -> FleetScope { + self.scope + } + /// Returns the committed revision authorizing first acceptance. + #[must_use] + pub const fn journal_revision(&self) -> u64 { + self.journal_revision + } + /// Returns the controller boot identity covered by the journal lease. + #[must_use] + pub const fn controller(&self) -> SessionId { + self.controller + } + /// Returns the controller fencing epoch. + #[must_use] + pub const fn controller_epoch(&self) -> u64 { + self.controller_epoch + } + /// Returns the logical issuance time; receivers recheck their current time. + #[must_use] + pub const fn issued_at_ms(&self) -> i64 { + self.issued_at_ms + } + /// Returns the immutable exact target/effect payload. + #[must_use] + pub const fn kind(&self) -> &FleetActionKind { + &self.kind + } + + /// Returns a stable deduplication key across controller adoption and retries. + /// Authorization revision/time are deliberately excluded; the current + /// journal is still required to accept an effect for the first time. + pub fn key(&self) -> Result { + self.validate()?; + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-action-key.v1\0"); + hash.update(self.scope.fleet.as_bytes()); + hash.update(self.scope.application.as_bytes()); + match &self.kind { + FleetActionKind::Movement { action, attempt } => { + return Ok(movement_key(self.scope, *action, &attempt.spec)); + } + FleetActionKind::Maintenance { action, operation } => { + hash.update(&[2, *action as u8]); + hash.update(operation.id.as_bytes()); + hash.update(operation.node.as_bytes()); + hash.update(operation.session.as_bytes()); + hash.update(&operation.intent_revision.to_be_bytes()); + } + } + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) + } + + /// Verifies exact current journal state before first local acceptance. + /// Caller identity and backend authenticity are checked by the application. + pub fn authorize_against(&self, head: &FleetHead, now_ms: i64) -> Result<()> { + self.validate()?; + head.check_action_controller(now_ms)?; + if self.scope != head.scope + || self.journal_revision != head.revision + || self.issued_at_ms > now_ms + { + return Err(OperationError::Conflict); + } + let expected = match &self.kind { + FleetActionKind::Movement { action, attempt } => { + head.movement_action(attempt.spec.id, *action, self.issued_at_ms)? + } + FleetActionKind::Maintenance { action, .. } => { + head.maintenance_action(*action, self.issued_at_ms)? + } + }; + if *self != expected { + return Err(OperationError::Fenced); + } + self.check_admission_deadline(now_ms) + } + + pub(super) fn check_admission_deadline(&self, now_ms: i64) -> Result<()> { + if let FleetActionKind::Movement { action, attempt } = &self.kind { + if matches!( + action, + MovementAction::Prepare + | MovementAction::Release + | MovementAction::ReleaseMaintenance + ) && now_ms >= attempt.spec.deadline_ms + { + return Err(OperationError::Deadline); + } + if action.is_source_release() + && attempt + .reservation + .is_none_or(|r| r.expires_at_ms <= now_ms) + { + return Err(OperationError::Invalid("release reservation expired")); + } + } + Ok(()) + } + + pub(super) fn validate(&self) -> Result<()> { + self.scope.validate()?; + if self.journal_revision == 0 + || self.controller_epoch == 0 + || !nonzero(self.controller.as_bytes()) + || self.issued_at_ms < 0 + { + return Err(OperationError::Invalid( + "invalid fleet action authorization", + )); + } + match &self.kind { + FleetActionKind::Movement { action, attempt } => { + attempt.validate()?; + if attempt.spec.target.application() != self.scope.application + || !movement_allowed(attempt, *action) + { + return Err(OperationError::Invalid("movement dispatch phase mismatch")); + } + } + FleetActionKind::Maintenance { action, operation } => { + operation.validate()?; + if !maintenance_allowed(operation, *action) { + return Err(OperationError::Invalid( + "maintenance dispatch phase mismatch", + )); + } + } + } + self.check_admission_deadline(self.issued_at_ms) + } +} + +impl FleetHead { + fn check_action_controller(&self, now_ms: i64) -> Result<()> { + if now_ms < self.last_observed_ms { + return Err(OperationError::Invalid("action time regressed")); + } + if self + .controller + .is_none_or(|lease| now_ms >= lease.expires_at_ms) + { + return Err(OperationError::Fenced); + } + Ok(()) + } + + /// Builds a movement action only after its dispatch phase was CAS-published. + pub fn movement_action( + &self, + id: AttemptId, + action: MovementAction, + now_ms: i64, + ) -> Result { + let attempt = self + .attempts + .iter() + .find(|attempt| attempt.spec.id == id) + .ok_or(OperationError::NotFound)?; + if action == MovementAction::ReleaseMaintenance { + let operation = self.maintenance.as_ref().ok_or(OperationError::Invalid( + "busy release lacks maintenance intent", + ))?; + if attempt.spec.id.operation != operation.id + || attempt.spec.source_node != operation.node + || attempt.spec.source != operation.session + || operation.phase != MaintenancePhase::Evacuating + { + return Err(OperationError::Invalid( + "busy release maintenance identity mismatch", + )); + } + if now_ms >= operation.deadline_ms { + return Err(OperationError::Deadline); + } + } + self.make_action( + FleetActionKind::Movement { + action, + attempt: Box::new(attempt.clone()), + }, + now_ms, + ) + } + + /// Builds lifecycle work only from the current committed maintenance operation. + pub fn maintenance_action( + &self, + action: MaintenanceAction, + now_ms: i64, + ) -> Result { + let operation = self.maintenance.as_ref().ok_or(OperationError::NotFound)?; + self.make_action( + FleetActionKind::Maintenance { + action, + operation: Box::new(operation.clone()), + }, + now_ms, + ) + } + + fn make_action(&self, kind: FleetActionKind, now_ms: i64) -> Result { + self.check_action_controller(now_ms)?; + let controller = self.controller.ok_or(OperationError::Fenced)?; + let action = FleetAction { + scope: self.scope, + journal_revision: self.revision, + controller: controller.claimant, + controller_epoch: controller.epoch, + issued_at_ms: now_ms, + kind, + }; + action.validate()?; + Ok(action) + } +} + +fn movement_allowed(attempt: &MoveAttempt, action: MovementAction) -> bool { + if action == MovementAction::Inspect { + return true; + } + if attempt.blocker == Some(DrainBlocker::OutcomeUnknown) { + return false; + } + match action { + MovementAction::Prepare => attempt.phase == AttemptPhase::Preparing, + MovementAction::Release => attempt.phase == AttemptPhase::Releasing, + MovementAction::ReleaseMaintenance => attempt.phase == AttemptPhase::MaintenanceReleasing, + MovementAction::Activate => attempt.phase == AttemptPhase::Activating, + MovementAction::Recover => attempt.phase == AttemptPhase::Recovering, + MovementAction::Cancel => { + attempt.phase == AttemptPhase::Cancelling + || attempt.phase == AttemptPhase::CleaningReceiver + || (matches!( + attempt.phase, + AttemptPhase::Activated | AttemptPhase::Recovered + ) && !attempt.receiver_cleaned) + } + MovementAction::Inspect => true, + MovementAction::Retire => false, + } +} + +fn maintenance_allowed(operation: &MaintenanceOperation, action: MaintenanceAction) -> bool { + match action { + MaintenanceAction::Cordon => operation.phase != MaintenancePhase::Completed, + MaintenanceAction::SettleRoles => operation.phase == MaintenancePhase::Evacuating, + MaintenanceAction::Finalize => operation.phase == MaintenancePhase::Closing, + MaintenanceAction::Inspect => true, + } +} + +/// Bounded remote status; transport/source errors remain on the adapter result. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum FleetOutcome { + /// Exact preferred receiver resources have been admitted. + Reserved(ReceiverReservation), + /// Exact source release and object-covered position are proven. + Released(PublishedPosition), + /// A current owner and actor-backed serving position are proven. + Activated(ActivationEvidence), + /// Failed-source recovery, distinct from a clean release followed by activation. + Recovered(Box), + /// No effect was accepted; this is distinct from an ambiguous transport error. + Rejected(DrainBlocker), + /// Accepted work or an obligation still blocks progress; retain its permit. + Blocked(DrainBlocker), + /// Accepted remote work has no confirmed result yet. + Unknown, + /// Receiver work joined and its unused reservation no longer exists. + ReceiverCleaned, + /// The exact local node session has applied its persistent admission closure. + Cordoned, + /// Complete role inventory at the named barrier reports zero obligations. + RolesSettled { + /// Digest of the authoritative inventory used at the final barrier. + inventory: Digest, + }, + /// Complete relocation, runtime/facility shutdown, and withdrawal evidence. + Stopped(DrainEvidence), +} + +/// Canonical bounded action result for an authenticated origin session. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct FleetActionOutcome { + /// Exact fleet/application scope checked by the receiving adapter. + pub scope: FleetScope, + /// Stable action identity, independent of controller adoption. + pub action_key: Digest, + /// Authenticated physical node reporting the result. + pub node: NodeId, + /// Exact boot session that performed or inspected the effect. + pub session: SessionId, + /// Logical time the evidence was inspected, not merely republished. + pub observed_at_ms: i64, + /// Typed bounded status; this record itself grants no Cell authority. + pub outcome: FleetOutcome, +} + +impl FleetActionOutcome { + pub(super) fn validate(&self) -> Result<()> { + self.scope.validate()?; + if !nonzero(self.action_key.as_bytes()) + || !nonzero(self.node.as_bytes()) + || !nonzero(self.session.as_bytes()) + || self.observed_at_ms < 0 + { + return Err(OperationError::Invalid("invalid fleet result identity")); + } + match &self.outcome { + FleetOutcome::Reserved(r) + if r.session != self.session || r.expires_at_ms <= self.observed_at_ms => + { + Err(OperationError::Invalid( + "reservation result is not admitted", + )) + } + FleetOutcome::Released(p) => p.validate(), + FleetOutcome::Activated(e) => { + e.position.validate()?; + if e.node != self.node || e.session != self.session { + return Err(OperationError::Invalid("activation result origin mismatch")); + } + Ok(()) + } + FleetOutcome::Recovered(e) => { + e.validate()?; + if e.serving.node != self.node + || e.serving.session != self.session + || e.recovery.recorded_at_ms() > self.observed_at_ms + { + return Err(OperationError::Invalid( + "recovery result origin or time mismatch", + )); + } + Ok(()) + } + FleetOutcome::RolesSettled { inventory } if !nonzero(inventory.as_bytes()) => { + Err(OperationError::Invalid("role result lacks inventory")) + } + FleetOutcome::Stopped(e) + if e.node != self.node + || e.session != self.session + || !e.ready_to_close() + || !e.facilities_closed + || !e.stopped + || !e.withdrawn => + { + Err(OperationError::Invalid( + "stopped result lacks terminal proof", + )) + } + FleetOutcome::Rejected(DrainBlocker::OutcomeUnknown) + | FleetOutcome::Blocked(DrainBlocker::OutcomeUnknown) => Err(OperationError::Invalid( + "an unknown result must use the unknown outcome", + )), + _ => Ok(()), + } + } + + /// Binds a reply to the exact issued action and authenticated local endpoint. + /// Readiness/fresh authority checks remain required before counting relocation. + pub fn validate_for(&self, action: &FleetAction) -> Result<()> { + self.validate()?; + if self.scope != action.scope || self.action_key != action.key()? { + return Err(OperationError::Invalid("fleet reply action mismatch")); + } + let permitted = match &action.kind { + FleetActionKind::Movement { + action: kind, + attempt, + } => { + let source = + self.node == attempt.spec.source_node && self.session == attempt.spec.source; + let receiver = self.node == attempt.spec.destination_node + && self.session == attempt.spec.destination; + match &self.outcome { + FleetOutcome::Reserved(_) => { + receiver + && matches!(kind, MovementAction::Prepare | MovementAction::Inspect) + } + FleetOutcome::Released(p) => { + source + && matches!( + kind, + MovementAction::Release + | MovementAction::ReleaseMaintenance + | MovementAction::Inspect + ) + && p.incarnation == attempt.spec.incarnation + && p.epoch == attempt.spec.source_epoch + } + FleetOutcome::Activated(e) => { + matches!(kind, MovementAction::Activate | MovementAction::Inspect) + && e.position.incarnation == attempt.spec.incarnation + && e.position.epoch > attempt.spec.source_epoch + && e.session != attempt.spec.source + && (e.node != attempt.spec.destination_node || receiver) + } + FleetOutcome::Recovered(e) => { + receiver + && matches!(kind, MovementAction::Recover | MovementAction::Inspect) + && e.recovery.basis().spec() == attempt.spec() + && e.recovery.basis().scope == action.scope + } + FleetOutcome::ReceiverCleaned => { + receiver && matches!(kind, MovementAction::Cancel | MovementAction::Inspect) + } + FleetOutcome::Rejected(_) + | FleetOutcome::Blocked(_) + | FleetOutcome::Unknown => match kind { + MovementAction::Release | MovementAction::ReleaseMaintenance => source, + MovementAction::Prepare + | MovementAction::Activate + | MovementAction::Cancel + | MovementAction::Recover => receiver, + MovementAction::Inspect => source || receiver, + MovementAction::Retire => false, + }, + _ => false, + } + } + FleetActionKind::Maintenance { + action: kind, + operation, + } => { + self.node == operation.node + && self.session == operation.session + && match self.outcome { + FleetOutcome::Cordoned => { + matches!(kind, MaintenanceAction::Cordon | MaintenanceAction::Inspect) + } + FleetOutcome::RolesSettled { .. } => matches!( + kind, + MaintenanceAction::SettleRoles | MaintenanceAction::Inspect + ), + FleetOutcome::Stopped(_) => matches!( + kind, + MaintenanceAction::Finalize | MaintenanceAction::Inspect + ), + FleetOutcome::Rejected(_) + | FleetOutcome::Blocked(_) + | FleetOutcome::Unknown => true, + _ => false, + } + } + }; + if !permitted { + return Err(OperationError::Invalid( + "fleet reply effect or target mismatch", + )); + } + Ok(()) + } +} + +// Preserve the existing action index bytes for all prior movement effects. +// Recovery basis decoders use this same function, not an independent key format. +pub(super) fn movement_key( + scope: FleetScope, + action: MovementAction, + spec: &super::MoveAttemptSpec, +) -> Digest { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-action-key.v1\0"); + hash.update(scope.fleet.as_bytes()); + hash.update(scope.application.as_bytes()); + hash.update(&[1, action as u8]); + hash.update(spec.id.operation.as_bytes()); + hash.update(&spec.id.sequence.to_be_bytes()); + hash.update(spec.target.cell_id().as_bytes()); + hash.update(spec.incarnation.as_bytes()); + hash.update(spec.source.as_bytes()); + hash.update(&spec.generation.to_be_bytes()); + hash.update(spec.destination.as_bytes()); + Digest::from_bytes(*hash.finalize().as_bytes()) +} diff --git a/crates/cellule-runtime/src/fleet/operations/attempt.rs b/crates/cellule-runtime/src/fleet/operations/attempt.rs new file mode 100644 index 00000000..98b59935 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/attempt.rs @@ -0,0 +1,743 @@ +use crate::control::RootRef; +use crate::identity::{CellTarget, Digest, IncarnationId, NodeId, SessionId}; + +use super::{AttemptId, DrainBlocker, OperationError, RecoveredActivation, Result, nonzero}; + +/// Conservative resources charged before a planned receive begins. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct TransferCost { + /// Reserved native memory and retained buffers in bytes. + pub memory_bytes: u64, + /// Conservative local restore and scratch disk demand in bytes. + pub disk_bytes: u64, + /// File descriptors needed by the incoming Cell and restoration. + pub file_descriptors: u32, + /// Worker/restore job credits needed by this attempt. + pub job_credits: u32, +} + +impl TransferCost { + /// Rejects unknown cost represented as zero. + pub fn validate(self) -> Result { + if self.memory_bytes == 0 + || self.disk_bytes == 0 + || self.file_descriptors == 0 + || self.job_credits == 0 + { + return Err(OperationError::Invalid("unknown transfer cost")); + } + Ok(self) + } +} + +/// Immutable exact identity and resource demand allocated by the journal. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct MoveAttemptSpec { + /// Never-reused operation/sequence identity. + pub id: AttemptId, + /// Verified tenant/application/namespace partition target. + pub target: CellTarget, + /// Cell incarnation observed at the source. + pub incarnation: IncarnationId, + /// Physical donor node. + pub source_node: NodeId, + /// Exact source boot session; a reboot cannot adopt its actor handle. + pub source: SessionId, + /// Exact activation generation checked by source release. + pub generation: u64, + /// Source ownership epoch checked by release evidence. + pub source_epoch: u64, + /// Preferred physical receiver. + pub destination_node: NodeId, + /// Exact preferred receiver boot session. + pub destination: SessionId, + /// Cost reserved by both the fleet permit and local receiver. + pub cost: TransferCost, + /// Canonical digest of the planner's observation inputs. + pub snapshot_digest: Digest, + /// Latest time at which new source release may be dispatched. + pub deadline_ms: i64, +} + +impl MoveAttemptSpec { + /// Validates all immutable attempt fields before allocating resources. + pub fn validate(&self) -> Result<()> { + self.id.validate()?; + self.cost.validate()?; + if !nonzero(self.incarnation.as_bytes()) + || !nonzero(self.source_node.as_bytes()) + || !nonzero(self.destination_node.as_bytes()) + || !nonzero(self.source.as_bytes()) + || !nonzero(self.destination.as_bytes()) + || !nonzero(self.snapshot_digest.as_bytes()) + || !nonzero(self.target.tenant().as_bytes()) + || !nonzero(self.target.application().as_bytes()) + || !nonzero(self.target.namespace().as_bytes()) + || self.source == self.destination + || self.source_node == self.destination_node + || self.generation == 0 + || self.source_epoch == 0 + || self.deadline_ms < 0 + { + return Err(OperationError::Invalid("invalid movement identity")); + } + Ok(()) + } +} + +/// Exact object-covered release or actor-backed serving position. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct PublishedPosition { + /// Cell incarnation this position belongs to. + pub incarnation: IncarnationId, + /// Cell authority epoch associated with the evidence. + pub epoch: u64, + /// Exact authoritative immutable root and sequence. + pub root: RootRef, +} + +impl PublishedPosition { + pub(super) fn validate(&self) -> Result<()> { + if !nonzero(self.incarnation.as_bytes()) + || self.epoch == 0 + || !nonzero(self.root.digest.as_bytes()) + || self.root.commit_sequence > i64::MAX as u64 + || self.root.checksum & cellule_ltx::types::CHECKSUM_FLAG == 0 + { + return Err(OperationError::Invalid( + "invalid published movement position", + )); + } + Ok(()) + } +} + +/// Admitted receiver reservation; it provides resources, never Cell authority. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ReceiverReservation { + /// Exact session owning the reservation for this attempt. + pub session: SessionId, + /// Time after which local cancellation may begin; not proof of cleanup. + pub expires_at_ms: i64, +} + +/// Verified current owner and actor readiness after normal acquisition. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct ActivationEvidence { + /// Actual serving node, which may differ from the preferred receiver. + pub node: NodeId, + /// Current serving boot session. + pub session: SessionId, + /// Current verified root and actor position. + pub position: PublishedPosition, +} + +/// Durable movement phase. Dispatch is recorded before starting remote work. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[repr(u8)] +pub enum AttemptPhase { + /// Fleet count/byte permit is allocated, with no remote effect yet. + Planned = 1, + /// Preparation may be accepted remotely; inspect before reallocation. + Preparing = 2, + /// Exact receiver resources have been admitted. + Reserved = 3, + /// Source release may be accepted and its outcome needs reconciliation. + Releasing = 4, + /// Source release is proven; serving elsewhere is not yet established. + Released = 5, + /// Ordinary receiver activation may be in progress. + Activating = 6, + /// Current actor-backed successor evidence is proven. + Activated = 7, + /// Remote cancellation/cleanup is requested and still charged. + Cancelling = 8, + /// No source release and no receiver work remain for this attempt. + Cancelled = 9, + /// Source release is proven; unused receiver cleanup is requested. + /// This phase cannot turn relocation into pre-release cancellation. + CleaningReceiver = 10, + /// Canonical failed-source recovery is accepted or awaiting inspection. + Recovering = 11, + /// Recovery and current serving are proved; receiver cleanup remains separate. + Recovered = 12, + /// Explicit busy maintenance release dispatch; ordinary release remains separate. + MaintenanceReleasing = 13, +} + +/// Advisory next action selected from durable state. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[repr(u8)] +pub enum MovementAction { + /// Prepare an exact-session admitted resource reservation. + Prepare = 1, + /// Quiesce and release the exact source generation through its actor. + Release = 2, + /// Activate by ordinary acquisition, consuming receiver reservation. + Activate = 3, + /// Inspect accepted work or authority before deciding what happened. + Inspect = 4, + /// Cancel and join unused receiver reservation/work. + Cancel = 5, + /// Terminal evidence permits removing this attempt from the active budget. + Retire = 6, + /// Recover an unresolved release using canonical failed-session proof. + Recover = 7, + /// Quiesce busy work under retained maintenance intent, then release canonically. + ReleaseMaintenance = 8, +} + +impl MovementAction { + /// True for either exact source release policy; all other effects route to a receiver. + #[must_use] + pub const fn is_source_release(self) -> bool { + matches!(self, Self::Release | Self::ReleaseMaintenance) + } +} + +/// Replayable result or dispatch intent; input identity is supplied by the journal. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum AttemptEvent { + /// Persist preparation dispatch before a receiver side effect. + BeginPrepare, + /// Confirm exact receiver resource admission. + Reserved(ReceiverReservation), + /// Persist source release dispatch before a source side effect. + BeginRelease, + /// Persist explicit busy maintenance release before closing foreground work. + BeginMaintenanceRelease, + /// The source definitively refused without accepting a release. + ReleaseRefused(DrainBlocker), + /// Confirm exact source release and final authoritative root. + Released(PublishedPosition), + /// Persist ordinary activation dispatch. + BeginActivate, + /// Persist failed-source recovery dispatch without claiming a clean release. + BeginRecover, + /// Confirm pinned recovery and actor-backed serving independently of release. + Recovered(Box), + /// Confirm current actor-backed serving state at or after the release. + Activated(ActivationEvidence), + /// An accepted action needs observation; keep its phase and permit. + OutcomeUnknown, + /// Begin cancellation only before any source release could be accepted. + /// After a proven release, request independent receiver cleanup instead. + BeginCancel, + /// Confirm receiver work is joined, reservation freed, and source unreleased. + Cancelled, + /// Confirm cleanup of an unused preferred reservation after another node won. + ReceiverCleaned, +} + +/// One durably charged attempt. Its private fields can change only by transition. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct MoveAttempt { + pub(super) spec: MoveAttemptSpec, + pub(super) phase: AttemptPhase, + pub(super) reservation: Option, + pub(super) released: Option, + pub(super) activated: Option, + pub(super) recovered: Option>, + pub(super) receiver_cleaned: bool, + pub(super) blocker: Option, + pub(super) completed_at_ms: Option, +} + +impl MoveAttempt { + pub(super) fn new(spec: MoveAttemptSpec) -> Result { + spec.validate()?; + Ok(Self { + spec, + phase: AttemptPhase::Planned, + reservation: None, + released: None, + activated: None, + recovered: None, + receiver_cleaned: false, + blocker: None, + completed_at_ms: None, + }) + } + + /// Returns the immutable attempt scope and cost. + #[must_use] + pub const fn spec(&self) -> &MoveAttemptSpec { + &self.spec + } + /// Returns the persisted phase. + #[must_use] + pub const fn phase(&self) -> AttemptPhase { + self.phase + } + /// Returns a temporary blocker or unknown-outcome marker. + #[must_use] + pub const fn blocker(&self) -> Option { + self.blocker + } + /// Returns exact source release evidence, when proven. + #[must_use] + pub const fn released(&self) -> Option<&PublishedPosition> { + self.released.as_ref() + } + /// Returns actual successor serving evidence, when proven. + #[must_use] + pub const fn activated(&self) -> Option<&ActivationEvidence> { + self.activated.as_ref() + } + /// Returns failed-source recovery evidence, never a clean release position. + #[must_use] + pub fn recovered(&self) -> Option<&RecoveredActivation> { + self.recovered.as_deref() + } + /// Returns receiver admission evidence, when proven. + #[must_use] + pub const fn reservation(&self) -> Option { + self.reservation + } + + /// Returns committed receiver cleanup/consumption evidence. Expiry alone + /// never establishes that the receiver's resource owners have joined. + #[must_use] + pub const fn receiver_resources_settled(&self) -> bool { + self.receiver_cleaned + } + + /// Checks current actor-backed serving evidence against this exact release. + /// This checks shape and required position, not authority or resource cleanup. + pub fn validate_activation(&self, evidence: &ActivationEvidence) -> Result<()> { + evidence.position.validate()?; + let release = self + .released + .as_ref() + .ok_or(OperationError::Invalid("activation without release"))?; + if !nonzero(evidence.node.as_bytes()) + || !nonzero(evidence.session.as_bytes()) + || evidence.node == self.spec.source_node + || evidence.session == self.spec.source + || evidence.position.incarnation != release.incarnation + || evidence.position.epoch <= release.epoch + || !successor_position(&evidence.position, release) + || (evidence.session == self.spec.destination + && evidence.node != self.spec.destination_node) + { + return Err(OperationError::Invalid("successor activation mismatch")); + } + Ok(()) + } + + /// Returns the first proven activation or cancellation time for history. + /// A duplicate reply cannot refresh the ordinary movement cooldown. + #[must_use] + pub const fn completed_at_ms(&self) -> Option { + self.completed_at_ms + } + + /// Selects work without assuming that an expired lease cancelled an effect. + #[must_use] + pub fn next_action(&self) -> MovementAction { + if self.blocker == Some(DrainBlocker::OutcomeUnknown) { + return MovementAction::Inspect; + } + match self.phase { + AttemptPhase::Planned => MovementAction::Prepare, + AttemptPhase::Reserved => MovementAction::Release, + AttemptPhase::Released => MovementAction::Activate, + AttemptPhase::Preparing + | AttemptPhase::Releasing + | AttemptPhase::MaintenanceReleasing + | AttemptPhase::Activating + | AttemptPhase::Recovering => MovementAction::Inspect, + AttemptPhase::Cancelling | AttemptPhase::CleaningReceiver => MovementAction::Cancel, + AttemptPhase::Activated | AttemptPhase::Recovered if !self.receiver_cleaned => { + MovementAction::Cancel + } + AttemptPhase::Activated | AttemptPhase::Recovered | AttemptPhase::Cancelled => { + MovementAction::Retire + } + } + } + + pub(super) fn can_retire(&self) -> bool { + self.phase == AttemptPhase::Cancelled + || (matches!( + self.phase, + AttemptPhase::Activated | AttemptPhase::Recovered + ) && self.receiver_cleaned) + } + + pub(super) fn resolve_unaccepted(&mut self, effect: MovementAction, now_ms: i64) -> Result<()> { + let expected = match self.phase { + AttemptPhase::Preparing => MovementAction::Prepare, + AttemptPhase::Releasing => MovementAction::Release, + AttemptPhase::MaintenanceReleasing => MovementAction::ReleaseMaintenance, + AttemptPhase::Activating => MovementAction::Activate, + AttemptPhase::Recovering => MovementAction::Recover, + AttemptPhase::Cancelling | AttemptPhase::CleaningReceiver => MovementAction::Cancel, + AttemptPhase::Activated | AttemptPhase::Recovered if !self.receiver_cleaned => { + MovementAction::Cancel + } + _ => return Err(OperationError::Invalid("phase has no unresolved dispatch")), + }; + if effect != expected + || self + .blocker + .is_some_and(|b| b != DrainBlocker::OutcomeUnknown) + { + return Err(OperationError::Invalid( + "unaccepted effect disagrees with phase", + )); + } + self.blocker = None; + match self.phase { + AttemptPhase::Preparing if now_ms >= self.spec.deadline_ms => { + self.phase = AttemptPhase::Cancelling + } + AttemptPhase::Releasing | AttemptPhase::MaintenanceReleasing + if now_ms >= self.spec.deadline_ms + || self.reservation.is_none_or(|r| r.expires_at_ms <= now_ms) => + { + // Atomic absence proves no source effect was accepted. Keep + // the receiver charged until its independent cancellation joins. + self.phase = AttemptPhase::Reserved; + self.blocker = Some(DrainBlocker::Deadline); + } + _ => {} + } + self.validate() + } + + pub(super) fn apply(&mut self, event: AttemptEvent, now_ms: i64) -> Result<()> { + let require_admission = || { + if now_ms >= self.spec.deadline_ms { + Err(OperationError::Deadline) + } else { + Ok(()) + } + }; + match event { + AttemptEvent::BeginPrepare if self.phase == AttemptPhase::Planned => { + require_admission()?; + self.phase = AttemptPhase::Preparing; + } + AttemptEvent::Reserved(reservation) + if matches!(self.phase, AttemptPhase::Preparing | AttemptPhase::Reserved) => + { + if reservation.session != self.spec.destination + || reservation.expires_at_ms <= now_ms + || self.reservation.is_some_and(|old| old != reservation) + { + return Err(OperationError::Invalid("receiver reservation mismatch")); + } + self.reservation = Some(reservation); + self.phase = AttemptPhase::Reserved; + self.blocker = None; + } + event @ (AttemptEvent::BeginRelease | AttemptEvent::BeginMaintenanceRelease) + if self.phase == AttemptPhase::Reserved + && self.blocker != Some(DrainBlocker::OutcomeUnknown) => + { + require_admission()?; + if self.reservation.is_none_or(|r| r.expires_at_ms <= now_ms) { + return Err(OperationError::Invalid("receiver reservation expired")); + } + self.phase = if matches!(event, AttemptEvent::BeginMaintenanceRelease) { + AttemptPhase::MaintenanceReleasing + } else { + AttemptPhase::Releasing + }; + self.blocker = None; + } + AttemptEvent::ReleaseRefused(blocker) + if matches!( + self.phase, + AttemptPhase::Releasing | AttemptPhase::MaintenanceReleasing + ) && blocker != DrainBlocker::OutcomeUnknown => + { + self.phase = AttemptPhase::Reserved; + self.blocker = Some(blocker); + } + AttemptEvent::Released(position) + if matches!( + self.phase, + AttemptPhase::Releasing + | AttemptPhase::MaintenanceReleasing + | AttemptPhase::Released + ) => + { + position.validate()?; + if position.incarnation != self.spec.incarnation + || position.epoch != self.spec.source_epoch + || self.released.as_ref().is_some_and(|old| old != &position) + { + return Err(OperationError::Invalid("source release position mismatch")); + } + self.released = Some(position); + self.phase = AttemptPhase::Released; + self.blocker = None; + } + AttemptEvent::BeginRecover + if matches!( + self.phase, + AttemptPhase::Releasing | AttemptPhase::MaintenanceReleasing + ) => + { + self.phase = AttemptPhase::Recovering; + self.blocker = None; + } + AttemptEvent::Recovered(evidence) + if matches!( + self.phase, + AttemptPhase::Recovering | AttemptPhase::Recovered + ) => + { + evidence.validate()?; + if evidence.recovery.basis().spec() != &self.spec + || evidence.recovery.recorded_at_ms() > now_ms + || self.recovered.as_ref().is_some_and(|old| old != &evidence) + { + return Err(OperationError::Invalid("recovery movement input mismatch")); + } + self.recovered = Some(evidence); + self.phase = AttemptPhase::Recovered; + self.completed_at_ms.get_or_insert(now_ms); + self.blocker = None; + } + AttemptEvent::BeginActivate if self.phase == AttemptPhase::Released => { + // Recovery of an accepted release is allowed beyond its admission deadline. + self.phase = AttemptPhase::Activating; + self.blocker = None; + } + AttemptEvent::Activated(evidence) + if matches!( + self.phase, + AttemptPhase::Activating + | AttemptPhase::Activated + | AttemptPhase::CleaningReceiver + ) => + { + self.validate_activation(&evidence)?; + if self.activated.as_ref().is_some_and(|old| old != &evidence) { + return Err(OperationError::Invalid("successor activation mismatch")); + } + // Serving on the preferred session may be ordinary acquisition. + // Resource consumption/cleanup needs independent executor proof. + self.activated = Some(evidence); + self.phase = AttemptPhase::Activated; + self.completed_at_ms.get_or_insert(now_ms); + self.blocker = None; + } + AttemptEvent::OutcomeUnknown + if matches!( + self.phase, + AttemptPhase::Preparing + | AttemptPhase::Releasing + | AttemptPhase::MaintenanceReleasing + | AttemptPhase::Recovering + | AttemptPhase::Activating + | AttemptPhase::Cancelling + | AttemptPhase::CleaningReceiver + ) || (matches!( + self.phase, + AttemptPhase::Activated | AttemptPhase::Recovered + ) && !self.receiver_cleaned) => + { + self.blocker = Some(DrainBlocker::OutcomeUnknown); + } + AttemptEvent::BeginCancel + if matches!( + self.phase, + AttemptPhase::Released + | AttemptPhase::Activating + | AttemptPhase::CleaningReceiver + ) && !self.receiver_cleaned => + { + self.phase = AttemptPhase::CleaningReceiver; + self.blocker = None; + } + AttemptEvent::BeginCancel + if matches!( + self.phase, + AttemptPhase::Planned + | AttemptPhase::Preparing + | AttemptPhase::Reserved + | AttemptPhase::Cancelling + ) => + { + self.phase = AttemptPhase::Cancelling; + self.blocker = None; + } + AttemptEvent::Cancelled + if matches!( + self.phase, + AttemptPhase::Cancelling | AttemptPhase::Cancelled + ) => + { + self.phase = AttemptPhase::Cancelled; + self.completed_at_ms.get_or_insert(now_ms); + self.receiver_cleaned = true; + self.blocker = None; + } + AttemptEvent::ReceiverCleaned + if matches!( + self.phase, + AttemptPhase::Activated | AttemptPhase::Recovered + ) => + { + self.receiver_cleaned = true; + self.blocker = None; + } + AttemptEvent::ReceiverCleaned if self.phase == AttemptPhase::CleaningReceiver => { + self.receiver_cleaned = true; + self.phase = AttemptPhase::Released; + self.blocker = None; + } + _ => return Err(OperationError::Invalid("unproven movement transition")), + } + self.validate() + } + + pub(super) fn validate(&self) -> Result<()> { + self.spec.validate()?; + if matches!( + self.phase, + AttemptPhase::Activated | AttemptPhase::Recovered | AttemptPhase::Cancelled + ) != self.completed_at_ms.is_some() + || self.completed_at_ms.is_some_and(|at| at < 0) + { + return Err(OperationError::Invalid( + "movement completion time disagrees with phase", + )); + } + if matches!(self.phase, AttemptPhase::Planned | AttemptPhase::Preparing) + && self.reservation.is_some() + { + return Err(OperationError::Invalid("reservation precedes admission")); + } + if let Some(r) = self.reservation + && (r.session != self.spec.destination || r.expires_at_ms < 0) + { + return Err(OperationError::Invalid("invalid stored reservation")); + } + if matches!( + self.phase, + AttemptPhase::Reserved + | AttemptPhase::Releasing + | AttemptPhase::MaintenanceReleasing + | AttemptPhase::Released + | AttemptPhase::Activating + | AttemptPhase::Activated + | AttemptPhase::CleaningReceiver + | AttemptPhase::Recovering + | AttemptPhase::Recovered + ) && self.reservation.is_none() + { + return Err(OperationError::Invalid("movement lost reservation")); + } + if matches!( + self.phase, + AttemptPhase::Released + | AttemptPhase::Activating + | AttemptPhase::Activated + | AttemptPhase::CleaningReceiver + ) != self.released.is_some() + { + return Err(OperationError::Invalid( + "movement release evidence disagrees with phase", + )); + } + if let Some(p) = &self.released { + p.validate()?; + if p.incarnation != self.spec.incarnation || p.epoch != self.spec.source_epoch { + return Err(OperationError::Invalid("stored release scope mismatch")); + } + } + if (self.phase == AttemptPhase::Activated) != self.activated.is_some() { + return Err(OperationError::Invalid( + "movement activation evidence disagrees with phase", + )); + } + if let Some(e) = &self.activated { + e.position.validate()?; + let p = self + .released + .as_ref() + .ok_or(OperationError::Invalid("missing release evidence"))?; + if !nonzero(e.node.as_bytes()) + || !nonzero(e.session.as_bytes()) + || e.node == self.spec.source_node + || e.session == self.spec.source + || e.position.incarnation != p.incarnation + || e.position.epoch <= p.epoch + || !successor_position(&e.position, p) + || (e.session == self.spec.destination && e.node != self.spec.destination_node) + { + return Err(OperationError::Invalid("invalid stored activation")); + } + } + if (self.phase == AttemptPhase::Recovered) != self.recovered.is_some() { + return Err(OperationError::Invalid( + "recovery evidence disagrees with phase", + )); + } + if let Some(evidence) = &self.recovered { + evidence.validate()?; + if evidence.recovery.basis().spec() != &self.spec { + return Err(OperationError::Invalid("stored recovery input mismatch")); + } + if self + .completed_at_ms + .is_none_or(|at| at < evidence.recovery.recorded_at_ms()) + { + return Err(OperationError::Invalid( + "recovery completion precedes evidence", + )); + } + } + if self.receiver_cleaned + && !matches!( + self.phase, + AttemptPhase::Released + | AttemptPhase::Activating + | AttemptPhase::Activated + | AttemptPhase::Cancelled + | AttemptPhase::Recovered + ) + { + return Err(OperationError::Invalid("premature reservation cleanup")); + } + if self.phase == AttemptPhase::Cancelled && !self.receiver_cleaned { + return Err(OperationError::Invalid("cancelled work lacks cleanup")); + } + if self.blocker == Some(DrainBlocker::OutcomeUnknown) + && !matches!( + self.phase, + AttemptPhase::Preparing + | AttemptPhase::Releasing + | AttemptPhase::MaintenanceReleasing + | AttemptPhase::Recovering + | AttemptPhase::Activating + | AttemptPhase::Cancelling + | AttemptPhase::CleaningReceiver + ) + && !(matches!( + self.phase, + AttemptPhase::Activated | AttemptPhase::Recovered + ) && !self.receiver_cleaned) + { + return Err(OperationError::Invalid( + "unknown outcome has no pending action", + )); + } + Ok(()) + } +} + +pub(super) fn successor_position( + current: &PublishedPosition, + released: &PublishedPosition, +) -> bool { + current.root.commit_sequence >= released.root.commit_sequence + && current.root.txid >= released.root.txid + && (current.root.commit_sequence != released.root.commit_sequence + || current.root == released.root) +} diff --git a/crates/cellule-runtime/src/fleet/operations/codec/accepted.rs b/crates/cellule-runtime/src/fleet/operations/codec/accepted.rs new file mode 100644 index 00000000..4e364bd0 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/codec/accepted.rs @@ -0,0 +1,32 @@ +use super::*; + +impl AcceptedFleetAction { + /// Encodes the immutable acceptance within the ordinary record bound. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder(ACCEPTED)?; + e.write_bytes(&self.action.to_bytes()?)?; + e.write_bytes(self.node.as_bytes())?; + e.write_bytes(self.session.as_bytes())?; + e.write_i64(self.accepted_at_ms)?; + Ok(e.finish()) + } + + /// Decodes a complete bounded acceptance without trusting its origin. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, ACCEPTED)?; + let action = FleetAction::from_bytes(d.read_bytes()?)?; + let node = NodeId::from_bytes(fixed(&mut d)?); + let session = SessionId::from_bytes(fixed(&mut d)?); + let accepted_at_ms = d.read_i64()?; + d.finish()?; + let accepted = Self { + action, + node, + session, + accepted_at_ms, + }; + accepted.validate()?; + Ok(accepted) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/codec/acquisition.rs b/crates/cellule-runtime/src/fleet/operations/codec/acquisition.rs new file mode 100644 index 00000000..b4b525dd --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/codec/acquisition.rs @@ -0,0 +1,29 @@ +use super::*; + +impl AcquisitionBasis { + /// Encodes a bounded immutable basis without changing Cell control format. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder(ACQUISITION)?; + e.write_bytes(&self.accepted.to_bytes()?)?; + e.write_bytes( + &self + .control + .encode() + .map_err(|error| OperationError::Control(Box::new(error)))?, + )?; + e.write_i64(self.observed_at_ms)?; + Ok(e.finish()) + } + + /// Decodes shape only; the journal and checked host supply provenance. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, ACQUISITION)?; + let accepted = AcceptedFleetAction::from_bytes(d.read_bytes()?)?; + let control = crate::control::Control::decode(d.read_bytes()?) + .map_err(|error| OperationError::Control(Box::new(error)))?; + let observed_at_ms = d.read_i64()?; + d.finish()?; + Self::new(accepted, control, observed_at_ms) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/codec/actions.rs b/crates/cellule-runtime/src/fleet/operations/codec/actions.rs new file mode 100644 index 00000000..17e1e501 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/codec/actions.rs @@ -0,0 +1,185 @@ +use super::*; + +fn movement(d: &mut BoundedDecoder<'_>) -> Result { + match d.read_u8()? { + 1 => Ok(MovementAction::Prepare), + 2 => Ok(MovementAction::Release), + 3 => Ok(MovementAction::Activate), + 4 => Ok(MovementAction::Inspect), + 5 => Ok(MovementAction::Cancel), + 7 => Ok(MovementAction::Recover), + 8 => Ok(MovementAction::ReleaseMaintenance), + _ => Err(OperationError::Invalid("unknown remote movement action")), + } +} + +fn maintenance(d: &mut BoundedDecoder<'_>) -> Result { + match d.read_u8()? { + 1 => Ok(MaintenanceAction::Cordon), + 2 => Ok(MaintenanceAction::SettleRoles), + 3 => Ok(MaintenanceAction::Finalize), + 4 => Ok(MaintenanceAction::Inspect), + _ => Err(OperationError::Invalid("unknown maintenance action")), + } +} + +impl FleetAction { + /// Encodes a bounded canonical envelope for an authenticated management adapter. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder(ACTION)?; + write_scope(&mut e, self.scope)?; + e.write_u64(self.journal_revision)?; + e.write_bytes(self.controller.as_bytes())?; + e.write_u64(self.controller_epoch)?; + e.write_i64(self.issued_at_ms)?; + match &self.kind { + FleetActionKind::Movement { action, attempt } => { + e.write_u8(1)?; + e.write_u8(*action as u8)?; + write_attempt(&mut e, attempt)?; + } + FleetActionKind::Maintenance { action, operation } => { + e.write_u8(2)?; + e.write_u8(*action as u8)?; + write_maintenance(&mut e, operation)?; + } + } + Ok(e.finish()) + } + + /// Decodes shape only. Receiving applications still authenticate and call + /// `authorize_against` with a fresh journal head before accepting an effect. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, ACTION)?; + let scope = read_scope(&mut d)?; + let journal_revision = d.read_u64()?; + let controller = SessionId::from_bytes(fixed(&mut d)?); + let controller_epoch = d.read_u64()?; + let issued_at_ms = d.read_i64()?; + let kind = match d.read_u8()? { + 1 => FleetActionKind::Movement { + action: movement(&mut d)?, + attempt: Box::new(read_attempt(&mut d)?), + }, + 2 => FleetActionKind::Maintenance { + action: maintenance(&mut d)?, + operation: Box::new(read_maintenance(&mut d)?), + }, + _ => return Err(OperationError::Invalid("unknown fleet action family")), + }; + d.finish()?; + let action = Self { + scope, + journal_revision, + controller, + controller_epoch, + issued_at_ms, + kind, + }; + action.validate()?; + Ok(action) + } +} + +impl FleetActionOutcome { + /// Encodes a typed reply while source errors stay in correlated adapter diagnostics. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder(OUTCOME)?; + write_scope(&mut e, self.scope)?; + e.write_bytes(self.action_key.as_bytes())?; + e.write_bytes(self.node.as_bytes())?; + e.write_bytes(self.session.as_bytes())?; + e.write_i64(self.observed_at_ms)?; + match &self.outcome { + FleetOutcome::Reserved(r) => { + e.write_u8(1)?; + e.write_bytes(r.session.as_bytes())?; + e.write_i64(r.expires_at_ms)?; + } + FleetOutcome::Released(p) => { + e.write_u8(2)?; + write_position(&mut e, p)?; + } + FleetOutcome::Activated(a) => { + e.write_u8(3)?; + e.write_bytes(a.node.as_bytes())?; + e.write_bytes(a.session.as_bytes())?; + write_position(&mut e, &a.position)?; + } + FleetOutcome::Rejected(blocker) => { + e.write_u8(4)?; + write_blocker(&mut e, Some(*blocker))?; + } + FleetOutcome::Blocked(blocker) => { + e.write_u8(5)?; + write_blocker(&mut e, Some(*blocker))?; + } + FleetOutcome::Unknown => e.write_u8(6)?, + FleetOutcome::ReceiverCleaned => e.write_u8(7)?, + FleetOutcome::Cordoned => e.write_u8(8)?, + FleetOutcome::RolesSettled { inventory } => { + e.write_u8(9)?; + e.write_bytes(inventory.as_bytes())?; + } + FleetOutcome::Recovered(evidence) => { + e.write_u8(11)?; + super::recovery::write_recovered(&mut e, evidence)?; + } + FleetOutcome::Stopped(evidence) => { + e.write_u8(10)?; + write_drain_evidence(&mut e, *evidence)?; + } + } + Ok(e.finish()) + } + + /// Restores one complete bounded reply without trusting its claimed origin. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, OUTCOME)?; + let scope = read_scope(&mut d)?; + let action_key = Digest::from_bytes(fixed(&mut d)?); + let node = NodeId::from_bytes(fixed(&mut d)?); + let session = SessionId::from_bytes(fixed(&mut d)?); + let observed_at_ms = d.read_i64()?; + let outcome = match d.read_u8()? { + 1 => FleetOutcome::Reserved(ReceiverReservation { + session: SessionId::from_bytes(fixed(&mut d)?), + expires_at_ms: d.read_i64()?, + }), + 2 => FleetOutcome::Released(read_position(&mut d)?), + 3 => FleetOutcome::Activated(ActivationEvidence { + node: NodeId::from_bytes(fixed(&mut d)?), + session: SessionId::from_bytes(fixed(&mut d)?), + position: read_position(&mut d)?, + }), + 4 => FleetOutcome::Rejected( + read_blocker(&mut d)?.ok_or(OperationError::Invalid("rejection lacks a reason"))?, + ), + 5 => FleetOutcome::Blocked( + read_blocker(&mut d)?.ok_or(OperationError::Invalid("blocker lacks a reason"))?, + ), + 6 => FleetOutcome::Unknown, + 7 => FleetOutcome::ReceiverCleaned, + 8 => FleetOutcome::Cordoned, + 9 => FleetOutcome::RolesSettled { + inventory: Digest::from_bytes(fixed(&mut d)?), + }, + 10 => FleetOutcome::Stopped(read_drain_evidence(&mut d)?), + 11 => FleetOutcome::Recovered(Box::new(super::recovery::read_recovered(&mut d)?)), + _ => return Err(OperationError::Invalid("unknown fleet action outcome")), + }; + d.finish()?; + let result = Self { + scope, + action_key, + node, + session, + observed_at_ms, + outcome, + }; + result.validate()?; + Ok(result) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/codec/attempt.rs b/crates/cellule-runtime/src/fleet/operations/codec/attempt.rs new file mode 100644 index 00000000..f22b655a --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/codec/attempt.rs @@ -0,0 +1,149 @@ +use super::recovery::{read_recovered, write_recovered}; +use super::*; + +pub(super) fn write_spec(e: &mut BoundedEncoder, spec: &MoveAttemptSpec) -> Result<()> { + write_operation_id(e, spec.id.operation)?; + e.write_u64(spec.id.sequence)?; + e.write_bytes(spec.target.tenant().as_bytes())?; + e.write_bytes(spec.target.application().as_bytes())?; + e.write_bytes(spec.target.namespace().as_bytes())?; + e.write_bytes(spec.target.partition())?; + e.write_bytes(spec.incarnation.as_bytes())?; + e.write_bytes(spec.source_node.as_bytes())?; + e.write_bytes(spec.source.as_bytes())?; + e.write_u64(spec.generation)?; + e.write_u64(spec.source_epoch)?; + e.write_bytes(spec.destination_node.as_bytes())?; + e.write_bytes(spec.destination.as_bytes())?; + e.write_u64(spec.cost.memory_bytes)?; + e.write_u64(spec.cost.disk_bytes)?; + e.write_u32(spec.cost.file_descriptors)?; + e.write_u32(spec.cost.job_credits)?; + e.write_bytes(spec.snapshot_digest.as_bytes())?; + e.write_i64(spec.deadline_ms)?; + Ok(()) +} + +pub(super) fn read_spec(d: &mut BoundedDecoder<'_>) -> Result { + let id = AttemptId { + operation: read_operation_id(d)?, + sequence: d.read_u64()?, + }; + let tenant = TenantId::from_bytes(fixed(d)?); + let application = ApplicationId::from_bytes(fixed(d)?); + let namespace = NamespaceId::from_bytes(fixed(d)?); + let partition = d.read_bytes()?; + let target = CellTarget::new(tenant, application, namespace, partition) + .map_err(|source| OperationError::Identity(Box::new(source)))?; + Ok(MoveAttemptSpec { + id, + target, + incarnation: IncarnationId::from_bytes(fixed(d)?), + source_node: NodeId::from_bytes(fixed(d)?), + source: SessionId::from_bytes(fixed(d)?), + generation: d.read_u64()?, + source_epoch: d.read_u64()?, + destination_node: NodeId::from_bytes(fixed(d)?), + destination: SessionId::from_bytes(fixed(d)?), + cost: TransferCost { + memory_bytes: d.read_u64()?, + disk_bytes: d.read_u64()?, + file_descriptors: d.read_u32()?, + job_credits: d.read_u32()?, + }, + snapshot_digest: Digest::from_bytes(fixed(d)?), + deadline_ms: d.read_i64()?, + }) +} + +pub(super) fn write_attempt(e: &mut BoundedEncoder, attempt: &MoveAttempt) -> Result<()> { + write_spec(e, &attempt.spec)?; + e.write_u8(attempt.phase as u8)?; + e.write_bool(attempt.reservation.is_some())?; + if let Some(reservation) = attempt.reservation { + e.write_bytes(reservation.session.as_bytes())?; + e.write_i64(reservation.expires_at_ms)?; + } + e.write_bool(attempt.released.is_some())?; + if let Some(position) = &attempt.released { + write_position(e, position)?; + } + e.write_bool(attempt.activated.is_some())?; + if let Some(evidence) = &attempt.activated { + e.write_bytes(evidence.node.as_bytes())?; + e.write_bytes(evidence.session.as_bytes())?; + write_position(e, &evidence.position)?; + } + e.write_bool(attempt.receiver_cleaned)?; + write_blocker(e, attempt.blocker)?; + e.write_bool(attempt.completed_at_ms.is_some())?; + if let Some(at) = attempt.completed_at_ms { + e.write_i64(at)?; + } + if let Some(evidence) = &attempt.recovered { + write_recovered(e, evidence)?; + } + Ok(()) +} + +pub(super) fn read_attempt(d: &mut BoundedDecoder<'_>) -> Result { + let spec = read_spec(d)?; + let phase = match d.read_u8()? { + 1 => AttemptPhase::Planned, + 2 => AttemptPhase::Preparing, + 3 => AttemptPhase::Reserved, + 4 => AttemptPhase::Releasing, + 5 => AttemptPhase::Released, + 6 => AttemptPhase::Activating, + 7 => AttemptPhase::Activated, + 8 => AttemptPhase::Cancelling, + 9 => AttemptPhase::Cancelled, + 10 => AttemptPhase::CleaningReceiver, + 11 => AttemptPhase::Recovering, + 12 => AttemptPhase::Recovered, + 13 => AttemptPhase::MaintenanceReleasing, + _ => return Err(OperationError::Invalid("unknown movement phase")), + }; + let reservation = if d.read_bool()? { + Some(ReceiverReservation { + session: SessionId::from_bytes(fixed(d)?), + expires_at_ms: d.read_i64()?, + }) + } else { + None + }; + let released = if d.read_bool()? { + Some(read_position(d)?) + } else { + None + }; + let activated = if d.read_bool()? { + Some(ActivationEvidence { + node: NodeId::from_bytes(fixed(d)?), + session: SessionId::from_bytes(fixed(d)?), + position: read_position(d)?, + }) + } else { + None + }; + let mut attempt = MoveAttempt { + spec, + phase, + reservation, + released, + activated, + recovered: None, + receiver_cleaned: d.read_bool()?, + blocker: read_blocker(d)?, + completed_at_ms: if d.read_bool()? { + Some(d.read_i64()?) + } else { + None + }, + }; + if phase == AttemptPhase::Recovered { + attempt.recovered = Some(Box::new(read_recovered(d)?)); + } + attempt.validate()?; + Ok(attempt) +} diff --git a/crates/cellule-runtime/src/fleet/operations/codec/follower_evacuation.rs b/crates/cellule-runtime/src/fleet/operations/codec/follower_evacuation.rs new file mode 100644 index 00000000..4b3ddbb6 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/codec/follower_evacuation.rs @@ -0,0 +1,100 @@ +use super::*; +impl FollowerReplacementPolicy { + /// Canonical scoped policy; shape does not authorize its publication. + pub fn to_bytes(self) -> Result> { + self.validate()?; + let mut e = encoder(FOLLOWER_POLICY)?; + write_scope(&mut e, self.scope)?; + e.write_u64(self.revision)?; + e.write_u8(self.minimum_members)?; + Ok(e.finish()) + } + /// Rejects unknown envelope, invalid scope/count and trailing data. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, FOLLOWER_POLICY)?; + let value = Self::new(read_scope(&mut d)?, d.read_u64()?, d.read_u8()?)?; + d.finish()?; + Ok(value) + } +} +impl FollowerEvacuationRecord { + /// Bounded manifest retaining every original and replacement member. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder_limited(FOLLOWER_EVACUATION, MAX_PAGE_BYTES)?; + e.write_bytes(&self.operation.to_bytes()?)?; + e.write_bytes(self.head_digest.as_bytes())?; + e.write_bytes(&self.registry.to_bytes()?)?; + e.write_bytes(&self.policy.to_bytes()?)?; + e.write_bytes(self.original_key.as_bytes())?; + e.write_bytes(self.original_digest.as_bytes())?; + e.write_count(self.retired.len())?; + for row in &self.retired { + e.write_bytes(&row.to_bytes()?)?; + } + e.write_u64(self.covered_through)?; + e.write_bytes(self.source_boot.as_bytes())?; + e.write_u64(self.replacement_epoch)?; + e.write_bytes(self.replacement_evidence.as_bytes())?; + e.write_count(self.replacements.len())?; + for entry in &self.replacements { + e.write_bytes(&entry.enrollment.to_bytes()?)?; + e.write_bytes(entry.boot_identity.as_bytes())?; + } + e.write_i64(self.started_at_ms)?; + e.write_i64(self.finished_at_ms)?; + Ok(e.finish()) + } + /// Historical shape only; the journal and native verifier authenticate it. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder_limited(bytes, FOLLOWER_EVACUATION, MAX_PAGE_BYTES)?; + let operation = MaintenanceOperation::from_bytes(d.read_bytes()?)?; + let head_digest = Digest::from_bytes(fixed(&mut d)?); + let registry = RegistryVersion::from_bytes(d.read_bytes()?)?; + let policy = FollowerReplacementPolicy::from_bytes(d.read_bytes()?)?; + let original_key = Digest::from_bytes(fixed(&mut d)?); + let original_digest = Digest::from_bytes(fixed(&mut d)?); + let count = d.read_count()?; + if count == 0 || count > 2 { + return Err(CodecError::Limit.into()); + } + let mut retired = Vec::with_capacity(count); + for _ in 0..count { + retired.push(EnrollmentRecord::from_bytes(d.read_bytes()?)?); + } + let covered_through = d.read_u64()?; + let source_boot = Digest::from_bytes(fixed(&mut d)?); + let replacement_epoch = d.read_u64()?; + let replacement_evidence = Digest::from_bytes(fixed(&mut d)?); + let count = d.read_count()?; + if count == 0 || count > 2 { + return Err(CodecError::Limit.into()); + } + let mut replacements = Vec::with_capacity(count); + for _ in 0..count { + replacements.push(FollowerReplacementWitness { + enrollment: EnrollmentRecord::from_bytes(d.read_bytes()?)?, + boot_identity: Digest::from_bytes(fixed(&mut d)?), + }); + } + let record = Self { + operation, + head_digest, + registry, + policy, + original_key, + original_digest, + retired, + covered_through, + source_boot, + replacement_epoch, + replacement_evidence, + replacements, + started_at_ms: d.read_i64()?, + finished_at_ms: d.read_i64()?, + }; + d.finish()?; + record.validate()?; + Ok(record) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/codec/inspection.rs b/crates/cellule-runtime/src/fleet/operations/codec/inspection.rs new file mode 100644 index 00000000..70fb99db --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/codec/inspection.rs @@ -0,0 +1,49 @@ +use super::*; + +impl FleetInspectionRequest { + /// Encodes one complete bounded request; older record kinds are unchanged. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder(INSPECTION_REQUEST)?; + e.write_bytes(&self.action.to_bytes()?)?; + e.write_bytes(&self.registry.to_bytes()?)?; + e.write_bytes(self.nonce.as_bytes())?; + e.write_bytes(self.node.as_bytes())?; + e.write_bytes(self.session.as_bytes())?; + e.write_i64(self.deadline_ms)?; + Ok(e.finish()) + } + /// Decodes shape only; authentication and current journal checks are required. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, INSPECTION_REQUEST)?; + let action = FleetAction::from_bytes(d.read_bytes()?)?; + let registry = RegistryVersion::from_bytes(d.read_bytes()?)?; + let nonce = Digest::from_bytes(fixed(&mut d)?); + let node = NodeId::from_bytes(fixed(&mut d)?); + let session = SessionId::from_bytes(fixed(&mut d)?); + let deadline_ms = d.read_i64()?; + d.finish()?; + Self::new(action, registry, nonce, node, session, deadline_ms) + } +} + +impl FleetInspectionObservation { + /// Encodes original capture times; replay/delivery must not restamp them. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder(INSPECTION_OBSERVATION)?; + e.write_bytes(&self.request.to_bytes()?)?; + e.write_i64(self.capture_started_at_ms)?; + e.write_bytes(&self.outcome.to_bytes()?)?; + Ok(e.finish()) + } + /// Restores one complete bounded request-bound observation. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, INSPECTION_OBSERVATION)?; + let request = FleetInspectionRequest::from_bytes(d.read_bytes()?)?; + let started = d.read_i64()?; + let outcome = FleetActionOutcome::from_bytes(d.read_bytes()?)?; + d.finish()?; + Self::new(request, started, outcome) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/codec/mod.rs b/crates/cellule-runtime/src/fleet/operations/codec/mod.rs new file mode 100644 index 00000000..35a8d7d9 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/codec/mod.rs @@ -0,0 +1,347 @@ +//! Canonical bounded journal records. No serde defaults can invent missing proof. + +use crate::codec::{BoundedDecoder, BoundedEncoder, CodecError}; +use crate::control::RootRef; +use crate::identity::{ + ApplicationId, CellTarget, Digest, IncarnationId, NamespaceId, NodeId, SessionId, TenantId, +}; + +use super::*; + +mod accepted; +mod acquisition; +mod attempt; +mod follower_evacuation; +mod inspection; +mod reader_evacuation; +mod recovery; +mod registry; +mod writer_inventory; +use attempt::{read_attempt, write_attempt}; +mod actions; +mod pages; +use pages::{read_scope, write_scope}; + +const DOMAIN: &[u8] = b"cellule.fleet-operation\0"; +const HEAD: u8 = 1; +const ATTEMPT: u8 = 2; +const MAINTENANCE: u8 = 3; +const INTENT: u8 = 4; +const PROGRESS: u8 = 5; +const ACTION: u8 = 6; +const OUTCOME: u8 = 7; +const ACCEPTED: u8 = 8; +const ACQUISITION: u8 = 9; +const RECOVERY_BASIS: u8 = 10; +const RECOVERY_EVIDENCE: u8 = 11; +const REGISTRY_VERSION: u8 = 12; +const INTENT_PAGE: u8 = 13; +const ENROLLMENT: u8 = 14; +const ENROLLMENT_PAGE: u8 = 15; +const ENROLLMENT_SPEC: u8 = 16; +const INSPECTION_REQUEST: u8 = 17; +const INSPECTION_OBSERVATION: u8 = 18; +const READER_EVACUATION: u8 = 19; +const READER_EVACUATION_PAGE: u8 = 20; +const READER_EVACUATION_BASIS: u8 = 21; +const FOLLOWER_POLICY: u8 = 22; +const FOLLOWER_EVACUATION: u8 = 23; +const WRITER_INVENTORY: u8 = 24; +const WRITER_INVENTORY_PAGE: u8 = 25; +const WRITER_INVENTORY_BASIS: u8 = 26; + +fn encoder(kind: u8) -> Result { + encoder_limited(kind, MAX_RECORD_BYTES) +} + +fn encoder_limited(kind: u8, limit: u32) -> Result { + let mut e = BoundedEncoder::new(limit)?; + e.write_bytes(DOMAIN)?; + e.write_u8(FORMAT_VERSION)?; + e.write_u8(kind)?; + Ok(e) +} + +fn decoder(bytes: &[u8], kind: u8) -> Result> { + decoder_limited(bytes, kind, MAX_RECORD_BYTES) +} + +fn decoder_limited(bytes: &[u8], kind: u8, limit: u32) -> Result> { + let mut d = BoundedDecoder::new(bytes, limit)?; + if d.read_bytes()? != DOMAIN || d.read_u8()? != FORMAT_VERSION || d.read_u8()? != kind { + return Err(OperationError::Invalid("unsupported fleet record envelope")); + } + Ok(d) +} + +fn fixed(d: &mut BoundedDecoder<'_>) -> Result<[u8; N]> { + d.read_bytes()? + .try_into() + .map_err(|_| OperationError::Invalid("fleet identity width")) +} + +fn write_operation_id(e: &mut BoundedEncoder, id: OperationId) -> Result<()> { + e.write_bytes(id.as_bytes())?; + Ok(()) +} + +fn read_operation_id(d: &mut BoundedDecoder<'_>) -> Result { + OperationId::from_bytes(fixed(d)?) +} + +fn write_blocker(e: &mut BoundedEncoder, value: Option) -> Result<()> { + e.write_u8(value.map_or(0, |b| b as u8))?; + Ok(()) +} + +fn read_blocker(d: &mut BoundedDecoder<'_>) -> Result> { + Ok(Some(match d.read_u8()? { + 0 => return Ok(None), + 1 => DrainBlocker::IncompleteObservation, + 2 => DrainBlocker::StaleObservation, + 3 => DrainBlocker::IncompatibleRelease, + 4 => DrainBlocker::ReceiverCapacity, + 5 => DrainBlocker::BusyExecution, + 6 => DrainBlocker::ExternalLease, + 7 => DrainBlocker::PendingPublication, + 8 => DrainBlocker::FollowerObligation, + 9 => DrainBlocker::UnknownInventory, + 10 => DrainBlocker::MovementBudget, + 11 => DrainBlocker::OutcomeUnknown, + 12 => DrainBlocker::Deadline, + 13 => DrainBlocker::FacilityFailure, + 14 => DrainBlocker::ReaderObligation, + _ => return Err(OperationError::Invalid("unknown fleet blocker")), + })) +} + +fn write_position(e: &mut BoundedEncoder, position: &PublishedPosition) -> Result<()> { + e.write_bytes(position.incarnation.as_bytes())?; + e.write_u64(position.epoch)?; + e.write_bytes(position.root.digest.as_bytes())?; + e.write_u64(position.root.txid)?; + e.write_u64(position.root.checksum)?; + e.write_u64(position.root.commit_sequence)?; + Ok(()) +} + +fn read_position(d: &mut BoundedDecoder<'_>) -> Result { + let position = PublishedPosition { + incarnation: IncarnationId::from_bytes(fixed(d)?), + epoch: d.read_u64()?, + root: RootRef { + digest: Digest::from_bytes(fixed(d)?), + txid: d.read_u64()?, + checksum: d.read_u64()?, + commit_sequence: d.read_u64()?, + }, + }; + position.validate()?; + Ok(position) +} + +fn write_drain_evidence(e: &mut BoundedEncoder, evidence: DrainEvidence) -> Result<()> { + e.write_bytes(evidence.node.as_bytes())?; + e.write_bytes(evidence.session.as_bytes())?; + e.write_u64(evidence.remaining_cells)?; + e.write_u32(evidence.unresolved_attempts)?; + e.write_bool(evidence.relocated)?; + e.write_bool(evidence.readers_settled)?; + e.write_bool(evidence.followers_settled)?; + e.write_bool(evidence.facilities_closed)?; + e.write_bool(evidence.stopped)?; + e.write_bool(evidence.withdrawn)?; + Ok(()) +} + +fn read_drain_evidence(d: &mut BoundedDecoder<'_>) -> Result { + Ok(DrainEvidence { + node: NodeId::from_bytes(fixed(d)?), + session: SessionId::from_bytes(fixed(d)?), + remaining_cells: d.read_u64()?, + unresolved_attempts: d.read_u32()?, + relocated: d.read_bool()?, + readers_settled: d.read_bool()?, + followers_settled: d.read_bool()?, + facilities_closed: d.read_bool()?, + stopped: d.read_bool()?, + withdrawn: d.read_bool()?, + }) +} + +fn write_maintenance(e: &mut BoundedEncoder, operation: &MaintenanceOperation) -> Result<()> { + write_operation_id(e, operation.id)?; + e.write_bytes(operation.request_digest.as_bytes())?; + e.write_bytes(operation.node.as_bytes())?; + e.write_bytes(operation.session.as_bytes())?; + e.write_u64(operation.intent_revision)?; + e.write_i64(operation.created_at_ms)?; + e.write_i64(operation.deadline_ms)?; + e.write_u8(operation.phase as u8)?; + write_blocker(e, operation.blocker)?; + e.write_bool(operation.drain_evidence.is_some())?; + if let Some(evidence) = operation.drain_evidence { + write_drain_evidence(e, evidence)?; + } + Ok(()) +} + +fn read_maintenance(d: &mut BoundedDecoder<'_>) -> Result { + let id = read_operation_id(d)?; + let request_digest = Digest::from_bytes(fixed(d)?); + let node = NodeId::from_bytes(fixed(d)?); + let session = SessionId::from_bytes(fixed(d)?); + let intent_revision = d.read_u64()?; + let created_at_ms = d.read_i64()?; + let deadline_ms = d.read_i64()?; + let phase = match d.read_u8()? { + 1 => MaintenancePhase::Requested, + 2 => MaintenancePhase::Cordoned, + 3 => MaintenancePhase::Evacuating, + 4 => MaintenancePhase::Closing, + 5 => MaintenancePhase::Completed, + _ => return Err(OperationError::Invalid("unknown maintenance phase")), + }; + let operation = MaintenanceOperation { + id, + request_digest, + node, + session, + intent_revision, + created_at_ms, + deadline_ms, + phase, + blocker: read_blocker(d)?, + drain_evidence: if d.read_bool()? { + Some(read_drain_evidence(d)?) + } else { + None + }, + }; + operation.validate()?; + Ok(operation) +} + +impl FleetHead { + /// Encodes a validated bounded head for atomic publication by an adapter. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder(HEAD)?; + e.write_bytes(self.scope.fleet.as_bytes())?; + e.write_bytes(self.scope.application.as_bytes())?; + e.write_u64(self.revision)?; + e.write_i64(self.last_observed_ms)?; + e.write_bool(self.controller.is_some())?; + if let Some(lease) = self.controller { + e.write_bytes(lease.claimant.as_bytes())?; + e.write_u64(lease.epoch)?; + e.write_i64(lease.expires_at_ms)?; + } + e.write_bool(self.maintenance.is_some())?; + if let Some(operation) = &self.maintenance { + write_maintenance(&mut e, operation)?; + } + e.write_u64(self.next_sequence)?; + e.write_count(self.attempts.len())?; + for attempt in &self.attempts { + write_attempt(&mut e, attempt)?; + } + e.write_bool(self.progress.is_some())?; + if let Some(progress) = self.progress { + e.write_bytes(progress.digest.as_bytes())?; + e.write_u64(progress.sequence)?; + } + Ok(e.finish()) + } + + /// Restores a complete canonical head, rejecting trailing and oversized data. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, HEAD)?; + let scope = FleetScope { + fleet: Digest::from_bytes(fixed(&mut d)?), + application: ApplicationId::from_bytes(fixed(&mut d)?), + }; + let revision = d.read_u64()?; + let last_observed_ms = d.read_i64()?; + let controller = if d.read_bool()? { + Some(ControllerLease { + claimant: SessionId::from_bytes(fixed(&mut d)?), + epoch: d.read_u64()?, + expires_at_ms: d.read_i64()?, + }) + } else { + None + }; + let maintenance = if d.read_bool()? { + Some(read_maintenance(&mut d)?) + } else { + None + }; + let next_sequence = d.read_u64()?; + let count = d.read_count()?; + if count > MAX_ACTIVE_ATTEMPTS { + return Err(CodecError::Limit.into()); + } + let mut attempts = Vec::with_capacity(count); + for _ in 0..count { + attempts.push(read_attempt(&mut d)?); + } + let progress = if d.read_bool()? { + Some(ProgressHead { + digest: Digest::from_bytes(fixed(&mut d)?), + sequence: d.read_u64()?, + }) + } else { + None + }; + d.finish()?; + let head = Self { + scope, + revision, + last_observed_ms, + controller, + maintenance, + next_sequence, + attempts, + progress, + }; + head.validate()?; + Ok(head) + } +} + +impl MoveAttempt { + /// Encodes an exact attempt for immutable historical progress pages. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder(ATTEMPT)?; + write_attempt(&mut e, self)?; + Ok(e.finish()) + } + + /// Restores a validated attempt from its canonical bounded record. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, ATTEMPT)?; + let attempt = read_attempt(&mut d)?; + d.finish()?; + Ok(attempt) + } +} + +impl MaintenanceOperation { + /// Encodes the durable physical-node intent and lifecycle progress. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder(MAINTENANCE)?; + write_maintenance(&mut e, self)?; + Ok(e.finish()) + } + + /// Restores a canonical bounded maintenance record. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, MAINTENANCE)?; + let operation = read_maintenance(&mut d)?; + d.finish()?; + Ok(operation) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/codec/pages.rs b/crates/cellule-runtime/src/fleet/operations/codec/pages.rs new file mode 100644 index 00000000..2ff3c0c4 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/codec/pages.rs @@ -0,0 +1,107 @@ +use super::*; +use crate::node::NodeMode; + +pub(super) fn write_scope(e: &mut BoundedEncoder, scope: FleetScope) -> Result<()> { + e.write_bytes(scope.fleet.as_bytes())?; + e.write_bytes(scope.application.as_bytes())?; + Ok(()) +} + +pub(super) fn read_scope(d: &mut BoundedDecoder<'_>) -> Result { + let scope = FleetScope { + fleet: Digest::from_bytes(fixed(d)?), + application: ApplicationId::from_bytes(fixed(d)?), + }; + scope.validate()?; + Ok(scope) +} + +impl NodeIntent { + /// Encodes bounded physical-node intent for atomic journal publication. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder(INTENT)?; + write_scope(&mut e, self.scope)?; + e.write_bytes(self.node.as_bytes())?; + e.write_bytes(self.session.as_bytes())?; + e.write_u64(self.revision)?; + e.write_u8(match self.mode { + NodeMode::Active => 1, + NodeMode::Cordoned => 2, + NodeMode::Draining => 3, + })?; + e.write_bool(self.operation.is_some())?; + if let Some(operation) = self.operation { + write_operation_id(&mut e, operation)?; + } + Ok(e.finish()) + } + + /// Restores an exact canonical desired-mode record without reopening it. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, INTENT)?; + let intent = Self { + scope: read_scope(&mut d)?, + node: NodeId::from_bytes(fixed(&mut d)?), + session: SessionId::from_bytes(fixed(&mut d)?), + revision: d.read_u64()?, + mode: match d.read_u8()? { + 1 => NodeMode::Active, + 2 => NodeMode::Cordoned, + 3 => NodeMode::Draining, + _ => return Err(OperationError::Invalid("unknown desired node mode")), + }, + operation: if d.read_bool()? { + Some(read_operation_id(&mut d)?) + } else { + None + }, + }; + d.finish()?; + intent.validate()?; + Ok(intent) + } +} + +impl ProgressPage { + /// Encodes at most 128 terminal records within the one-MiB page limit. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder_limited(PROGRESS, MAX_PAGE_BYTES)?; + write_scope(&mut e, self.scope)?; + write_operation_id(&mut e, self.operation)?; + e.write_u64(self.sequence)?; + e.write_bool(self.previous.is_some())?; + if let Some(previous) = self.previous { + e.write_bytes(previous.as_bytes())?; + } + e.write_count(self.entries.len())?; + for attempt in &self.entries { + write_attempt(&mut e, attempt)?; + } + Ok(e.finish()) + } + + /// Restores a bounded complete page, rejecting an oversized count before allocation. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder_limited(bytes, PROGRESS, MAX_PAGE_BYTES)?; + let scope = read_scope(&mut d)?; + let operation = read_operation_id(&mut d)?; + let sequence = d.read_u64()?; + let previous = if d.read_bool()? { + Some(Digest::from_bytes(fixed(&mut d)?)) + } else { + None + }; + let count = d.read_count()?; + if count == 0 || count > MAX_PAGE_ENTRIES { + return Err(CodecError::Limit.into()); + } + let mut entries = Vec::with_capacity(count); + for _ in 0..count { + entries.push(read_attempt(&mut d)?); + } + d.finish()?; + Self::new(scope, operation, sequence, previous, entries) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/codec/reader_evacuation.rs b/crates/cellule-runtime/src/fleet/operations/codec/reader_evacuation.rs new file mode 100644 index 00000000..2a962884 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/codec/reader_evacuation.rs @@ -0,0 +1,139 @@ +use super::*; +use crate::control::Control; + +fn write_basis(e: &mut BoundedEncoder, record: &ReaderEvacuationRecord) -> Result<()> { + e.write_bytes(&record.operation.to_bytes()?)?; + e.write_bytes(record.head_digest.as_bytes())?; + e.write_bytes(&record.registry.to_bytes()?)?; + e.write_bytes(&record.retired.to_bytes()?)?; + e.write_bytes(record.original_digest.as_bytes())?; + e.write_bytes( + &record + .authority + .encode() + .map_err(|error| OperationError::Control(Box::new(error)))?, + )?; + e.write_bool(record.policy_revision.is_some())?; + if let Some(revision) = record.policy_revision { + e.write_u64(revision)?; + } + e.write_u32(u32::from(record.desired_readers))?; + e.write_u64(record.minimum_sequence)?; + e.write_i64(record.started_at_ms)?; + e.write_i64(record.finished_at_ms)?; + Ok(()) +} +impl ReaderEvacuationRecord { + /// Immutable capture basis, independent of its page digests. + pub fn basis_digest(&self) -> Result { + self.validate_basis()?; + let mut e = encoder_limited(READER_EVACUATION_BASIS, MAX_PAGE_BYTES)?; + write_basis(&mut e, self)?; + Ok(Digest::from_bytes(*blake3::hash(&e.finish()).as_bytes())) + } + /// Canonical bounded manifest. Complete page verification remains required. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder_limited(READER_EVACUATION, MAX_PAGE_BYTES)?; + write_basis(&mut e, self)?; + e.write_count(self.pages.len())?; + for digest in &self.pages { + e.write_bytes(digest.as_bytes())?; + } + Ok(e.finish()) + } + /// Decodes historical shape only; native provenance and freshness are separate. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder_limited(bytes, READER_EVACUATION, MAX_PAGE_BYTES)?; + let operation = MaintenanceOperation::from_bytes(d.read_bytes()?)?; + let head_digest = Digest::from_bytes(fixed(&mut d)?); + let registry = RegistryVersion::from_bytes(d.read_bytes()?)?; + let retired = EnrollmentRecord::from_bytes(d.read_bytes()?)?; + let original_digest = Digest::from_bytes(fixed(&mut d)?); + let authority = Control::decode(d.read_bytes()?) + .map_err(|error| OperationError::Control(Box::new(error)))?; + let policy_revision = if d.read_bool()? { + Some(d.read_u64()?) + } else { + None + }; + let desired_readers = u16::try_from(d.read_u32()?) + .map_err(|_| OperationError::Invalid("reader policy count width"))?; + let minimum_sequence = d.read_u64()?; + let started_at_ms = d.read_i64()?; + let finished_at_ms = d.read_i64()?; + let count = d.read_count()?; + if count > super::super::reader_evacuation::MAX_READER_PAGES { + return Err(CodecError::Limit.into()); + } + let mut pages = Vec::with_capacity(count); + for _ in 0..count { + pages.push(Digest::from_bytes(fixed(&mut d)?)); + } + d.finish()?; + let record = Self { + operation, + head_digest, + registry, + retired, + original_digest, + authority, + policy_revision, + desired_readers, + minimum_sequence, + started_at_ms, + finished_at_ms, + pages, + }; + record.validate()?; + Ok(record) + } +} +impl ReaderEvacuationPage { + /// Canonical page, bounded by the existing 128-entry and 64-KiB limits. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder(READER_EVACUATION_PAGE)?; + e.write_bytes(self.basis.as_bytes())?; + e.write_u32(self.ordinal)?; + e.write_count(self.entries.len())?; + for entry in &self.entries { + e.write_bytes(entry.node.as_bytes())?; + e.write_bytes(entry.session.as_bytes())?; + e.write_bytes(entry.boot_identity.as_bytes())?; + e.write_bytes(entry.enrollment_key.as_bytes())?; + e.write_bytes(entry.enrollment_digest.as_bytes())?; + e.write_u64(entry.commit_sequence)?; + } + Ok(e.finish()) + } + /// Rejects unknown envelopes, oversized/count/trailing data before allocation. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, READER_EVACUATION_PAGE)?; + let basis = Digest::from_bytes(fixed(&mut d)?); + let ordinal = d.read_u32()?; + let count = d.read_count()?; + if count == 0 || count > MAX_PAGE_ENTRIES { + return Err(CodecError::Limit.into()); + } + let mut entries = Vec::with_capacity(count); + for _ in 0..count { + entries.push(ReaderReplacementWitness { + node: NodeId::from_bytes(fixed(&mut d)?), + session: SessionId::from_bytes(fixed(&mut d)?), + boot_identity: Digest::from_bytes(fixed(&mut d)?), + enrollment_key: Digest::from_bytes(fixed(&mut d)?), + enrollment_digest: Digest::from_bytes(fixed(&mut d)?), + commit_sequence: d.read_u64()?, + }); + } + d.finish()?; + let page = Self { + basis, + ordinal, + entries, + }; + page.validate()?; + Ok(page) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/codec/recovery.rs b/crates/cellule-runtime/src/fleet/operations/codec/recovery.rs new file mode 100644 index 00000000..50861958 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/codec/recovery.rs @@ -0,0 +1,103 @@ +use super::attempt::{read_spec, write_spec}; +use super::*; + +fn write_control(e: &mut BoundedEncoder, control: &crate::control::Control) -> Result<()> { + e.write_bytes( + &control + .encode() + .map_err(|e| OperationError::Control(Box::new(e)))?, + )?; + Ok(()) +} +fn read_control(d: &mut BoundedDecoder<'_>) -> Result { + crate::control::Control::decode(d.read_bytes()?) + .map_err(|e| OperationError::Control(Box::new(e))) +} +fn write_basis(e: &mut BoundedEncoder, basis: &RecoveryBasis) -> Result<()> { + write_spec(e, &basis.spec)?; + write_scope(e, basis.scope)?; + e.write_bytes(basis.action_key.as_bytes())?; + e.write_bytes(basis.node.as_bytes())?; + e.write_bytes(basis.session.as_bytes())?; + e.write_i64(basis.accepted_at_ms)?; + write_control(e, &basis.control)?; + e.write_i64(basis.observed_at_ms)?; + Ok(()) +} +fn read_basis(d: &mut BoundedDecoder<'_>) -> Result { + let basis = RecoveryBasis { + spec: read_spec(d)?, + scope: read_scope(d)?, + action_key: Digest::from_bytes(fixed(d)?), + node: NodeId::from_bytes(fixed(d)?), + session: SessionId::from_bytes(fixed(d)?), + accepted_at_ms: d.read_i64()?, + control: read_control(d)?, + observed_at_ms: d.read_i64()?, + }; + basis.validate()?; + Ok(basis) +} +fn write_evidence(e: &mut BoundedEncoder, evidence: &RecoveryEvidence) -> Result<()> { + write_basis(e, &evidence.basis)?; + write_control(e, &evidence.restored)?; + e.write_i64(evidence.recorded_at_ms)?; + Ok(()) +} +fn read_evidence(d: &mut BoundedDecoder<'_>) -> Result { + RecoveryEvidence::new(read_basis(d)?, read_control(d)?, d.read_i64()?) +} +pub(super) fn write_recovered( + e: &mut BoundedEncoder, + evidence: &RecoveredActivation, +) -> Result<()> { + write_evidence(e, &evidence.recovery)?; + e.write_bytes(evidence.serving.node.as_bytes())?; + e.write_bytes(evidence.serving.session.as_bytes())?; + write_position(e, &evidence.serving.position) +} +pub(super) fn read_recovered(d: &mut BoundedDecoder<'_>) -> Result { + let evidence = RecoveredActivation { + recovery: read_evidence(d)?, + serving: ActivationEvidence { + node: NodeId::from_bytes(fixed(d)?), + session: SessionId::from_bytes(fixed(d)?), + position: read_position(d)?, + }, + }; + evidence.validate()?; + Ok(evidence) +} + +impl RecoveryBasis { + /// Encodes immutable recovery input separately from the accepted-action record. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder(RECOVERY_BASIS)?; + write_basis(&mut e, self)?; + Ok(e.finish()) + } + /// Checks bounded canonical shape; decoding cannot create takeover authority. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, RECOVERY_BASIS)?; + let basis = read_basis(&mut d)?; + d.finish()?; + Ok(basis) + } +} +impl RecoveryEvidence { + /// Encodes the exact recovered position retained before successor admission. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder(RECOVERY_EVIDENCE)?; + write_evidence(&mut e, self)?; + Ok(e.finish()) + } + /// Checks canonical transition shape; the journal provides execution provenance. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, RECOVERY_EVIDENCE)?; + let evidence = read_evidence(&mut d)?; + d.finish()?; + Ok(evidence) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/codec/registry.rs b/crates/cellule-runtime/src/fleet/operations/codec/registry.rs new file mode 100644 index 00000000..1f4ceaab --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/codec/registry.rs @@ -0,0 +1,303 @@ +use super::*; +use crate::node::NodeMode; + +fn write_version(e: &mut BoundedEncoder, version: RegistryVersion) -> Result<()> { + write_scope(e, version.scope)?; + e.write_u64(version.revision)?; + e.write_bool(version.bootstrap_revision.is_some())?; + if let Some(revision) = version.bootstrap_revision { + e.write_u64(revision)?; + } + e.write_bool(version.scheduling_enabled)?; + Ok(()) +} + +fn read_version(d: &mut BoundedDecoder<'_>) -> Result { + let version = RegistryVersion { + scope: read_scope(d)?, + revision: d.read_u64()?, + bootstrap_revision: if d.read_bool()? { + Some(d.read_u64()?) + } else { + None + }, + scheduling_enabled: d.read_bool()?, + }; + version.validate()?; + Ok(version) +} + +fn write_endpoint(e: &mut BoundedEncoder, endpoint: EnrollmentEndpoint) -> Result<()> { + e.write_bytes(endpoint.node.as_bytes())?; + e.write_bytes(endpoint.session.as_bytes())?; + e.write_u64(endpoint.intent_revision)?; + Ok(()) +} + +fn read_endpoint(d: &mut BoundedDecoder<'_>) -> Result { + let endpoint = EnrollmentEndpoint { + node: NodeId::from_bytes(fixed(d)?), + session: SessionId::from_bytes(fixed(d)?), + intent_revision: d.read_u64()?, + }; + endpoint.validate()?; + Ok(endpoint) +} + +fn write_optional_digest(e: &mut BoundedEncoder, value: Option) -> Result<()> { + e.write_bool(value.is_some())?; + if let Some(value) = value { + e.write_bytes(value.as_bytes())?; + } + Ok(()) +} + +fn read_optional_digest(d: &mut BoundedDecoder<'_>) -> Result> { + Ok(if d.read_bool()? { + Some(Digest::from_bytes(fixed(d)?)) + } else { + None + }) +} + +fn write_spec(e: &mut BoundedEncoder, spec: &EnrollmentSpec) -> Result<()> { + write_scope(e, spec.scope)?; + e.write_bytes(spec.request.as_bytes())?; + match &spec.role { + EnrollmentRole::Node { mode } => { + e.write_u8(1)?; + e.write_u8(match mode { + NodeMode::Active => 1, + NodeMode::Cordoned => 2, + NodeMode::Draining => 3, + })?; + } + EnrollmentRole::Reader { target, position } => { + e.write_u8(2)?; + e.write_bytes(target.tenant().as_bytes())?; + e.write_bytes(target.application().as_bytes())?; + e.write_bytes(target.namespace().as_bytes())?; + e.write_bytes(target.partition())?; + write_position(e, position)?; + } + EnrollmentRole::Follower { log_epoch } => { + e.write_u8(3)?; + e.write_u64(*log_epoch)?; + } + } + e.write_bool(spec.source.is_some())?; + if let Some(source) = spec.source { + write_endpoint(e, source)?; + } + write_endpoint(e, spec.target)?; + Ok(()) +} + +fn write_record(e: &mut BoundedEncoder, record: &EnrollmentRecord) -> Result<()> { + write_spec(e, &record.spec)?; + e.write_i64(record.accepted_at_ms)?; + e.write_i64(record.updated_at_ms)?; + e.write_u8(record.status as u8)?; + write_optional_digest(e, record.established)?; + write_optional_digest(e, record.settlement)?; + Ok(()) +} + +fn read_spec(d: &mut BoundedDecoder<'_>) -> Result { + let scope = read_scope(d)?; + let request = Digest::from_bytes(fixed(d)?); + let role = match d.read_u8()? { + 1 => EnrollmentRole::Node { + mode: match d.read_u8()? { + 1 => NodeMode::Active, + 2 => NodeMode::Cordoned, + 3 => NodeMode::Draining, + _ => return Err(OperationError::Invalid("unknown boot enrollment mode")), + }, + }, + 2 => { + let tenant = TenantId::from_bytes(fixed(d)?); + let application = ApplicationId::from_bytes(fixed(d)?); + let namespace = NamespaceId::from_bytes(fixed(d)?); + let target = CellTarget::new(tenant, application, namespace, d.read_bytes()?) + .map_err(|error| OperationError::Identity(Box::new(error)))?; + EnrollmentRole::Reader { + target, + position: read_position(d)?, + } + } + 3 => EnrollmentRole::Follower { + log_epoch: d.read_u64()?, + }, + _ => return Err(OperationError::Invalid("unknown enrollment role")), + }; + let source = if d.read_bool()? { + Some(read_endpoint(d)?) + } else { + None + }; + let target = read_endpoint(d)?; + let spec = EnrollmentSpec { + scope, + request, + role, + source, + target, + }; + spec.validate()?; + Ok(spec) +} + +fn read_record(d: &mut BoundedDecoder<'_>) -> Result { + let spec = read_spec(d)?; + let accepted_at_ms = d.read_i64()?; + let updated_at_ms = d.read_i64()?; + let status = match d.read_u8()? { + 1 => EnrollmentStatus::Pending, + 2 => EnrollmentStatus::Established, + 3 => EnrollmentStatus::Refused, + 4 => EnrollmentStatus::Retired, + _ => return Err(OperationError::Invalid("unknown enrollment status")), + }; + let record = EnrollmentRecord { + spec, + accepted_at_ms, + updated_at_ms, + status, + established: read_optional_digest(d)?, + settlement: read_optional_digest(d)?, + }; + record.validate()?; + Ok(record) +} + +impl EnrollmentSpec { + /// Encodes bounded immutable request inputs before journal acceptance. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder(ENROLLMENT_SPEC)?; + write_spec(&mut e, self)?; + Ok(e.finish()) + } + /// Restores a request without implying it has a pending permit or authority. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, ENROLLMENT_SPEC)?; + let spec = read_spec(&mut d)?; + d.finish()?; + Ok(spec) + } +} + +impl RegistryVersion { + /// Encodes the exact shared registry revision and coverage barrier. + pub fn to_bytes(self) -> Result> { + self.validate()?; + let mut e = encoder(REGISTRY_VERSION)?; + write_version(&mut e, self)?; + Ok(e.finish()) + } + /// Restores the strict record without inventing a bootstrap marker. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, REGISTRY_VERSION)?; + let version = read_version(&mut d)?; + d.finish()?; + Ok(version) + } +} + +impl EnrollmentRecord { + /// Encodes immutable inputs and checked original progress in 64 KiB. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder(ENROLLMENT)?; + write_record(&mut e, self)?; + Ok(e.finish()) + } + /// Restores a retained obligation, including failed or pending sessions. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder(bytes, ENROLLMENT)?; + let record = read_record(&mut d)?; + d.finish()?; + Ok(record) + } +} + +impl IntentPage { + /// Encodes at most 128 retained intents within one MiB. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder_limited(INTENT_PAGE, MAX_PAGE_BYTES)?; + write_version(&mut e, self.version)?; + e.write_bool(self.after.is_some())?; + if let Some(after) = self.after { + e.write_bytes(after.as_bytes())?; + } + e.write_count(self.entries.len())?; + for intent in &self.entries { + e.write_bytes(&intent.to_bytes()?)?; + } + e.write_bool(self.next.is_some())?; + if let Some(next) = self.next { + e.write_bytes(next.as_bytes())?; + } + Ok(e.finish()) + } + /// Rejects an oversized row count before allocating the page vector. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder_limited(bytes, INTENT_PAGE, MAX_PAGE_BYTES)?; + let version = read_version(&mut d)?; + let after = if d.read_bool()? { + Some(NodeId::from_bytes(fixed(&mut d)?)) + } else { + None + }; + let count = d.read_count()?; + if count > MAX_PAGE_ENTRIES { + return Err(CodecError::Limit.into()); + } + let mut entries = Vec::with_capacity(count); + for _ in 0..count { + entries.push(NodeIntent::from_bytes(d.read_bytes()?)?); + } + let next = if d.read_bool()? { + Some(NodeId::from_bytes(fixed(&mut d)?)) + } else { + None + }; + d.finish()?; + Self::new(version, after, entries, next) + } +} + +impl EnrollmentPage { + /// Encodes every included obligation without filtering failed-session rows. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder_limited(ENROLLMENT_PAGE, MAX_PAGE_BYTES)?; + write_version(&mut e, self.version)?; + write_optional_digest(&mut e, self.after)?; + e.write_count(self.entries.len())?; + for record in &self.entries { + write_record(&mut e, record)?; + } + write_optional_digest(&mut e, self.next)?; + Ok(e.finish()) + } + /// Restores a bounded sorted page, rejecting incomplete progress shapes. + pub fn from_bytes(bytes: &[u8]) -> Result { + let mut d = decoder_limited(bytes, ENROLLMENT_PAGE, MAX_PAGE_BYTES)?; + let version = read_version(&mut d)?; + let after = read_optional_digest(&mut d)?; + let count = d.read_count()?; + if count > MAX_PAGE_ENTRIES { + return Err(CodecError::Limit.into()); + } + let mut entries = Vec::with_capacity(count); + for _ in 0..count { + entries.push(read_record(&mut d)?); + } + let next = read_optional_digest(&mut d)?; + d.finish()?; + Self::new(version, after, entries, next) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/codec/writer_inventory.rs b/crates/cellule-runtime/src/fleet/operations/codec/writer_inventory.rs new file mode 100644 index 00000000..a321bb5c --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/codec/writer_inventory.rs @@ -0,0 +1,164 @@ +use super::*; +use crate::control::Control; + +fn write_basis(e: &mut BoundedEncoder, record: &OriginalWriterInventoryRecord) -> Result<()> { + let basis = record.basis(); + e.write_bytes(&basis.operation.to_bytes()?)?; + e.write_bytes(basis.head_digest.as_bytes())?; + e.write_bytes(&basis.registry.to_bytes()?)?; + e.write_bytes(&basis.boot.to_bytes()?)?; + for digest in [ + basis.process_request, + basis.process_witness, + basis.catalog_witness, + ] { + e.write_bytes(digest.as_bytes())?; + } + e.write_i64(basis.interval.0)?; + e.write_i64(basis.interval.1)?; + e.write_count(record.catalogs.len())?; + for row in &record.catalogs { + e.write_bytes(row.application.as_bytes())?; + e.write_bytes(row.tenant.as_bytes())?; + for digest in [row.source, row.heads, row.histories] { + e.write_bytes(digest.as_bytes())?; + } + e.write_u64(row.cells)?; + e.write_u64(row.owners)?; + } + e.write_count(record.count)?; + Ok(()) +} +impl OriginalWriterInventoryRecord { + /// Complete canonical basis, independent of its dependent page digests. + pub fn basis_digest(&self) -> Result { + self.validate_basis()?; + let mut e = encoder(WRITER_INVENTORY_BASIS)?; + write_basis(&mut e, self)?; + Ok(Digest::from_bytes(*blake3::hash(&e.finish()).as_bytes())) + } + /// Canonical manifest bounded by the existing 64-KiB record envelope. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder(WRITER_INVENTORY)?; + write_basis(&mut e, self)?; + e.write_count(self.pages.len())?; + for digest in &self.pages { + e.write_bytes(digest.as_bytes())?; + } + Ok(e.finish()) + } + /// Decodes historical metadata only; completion requires every verified page. + pub fn from_bytes(body: &[u8]) -> Result { + let mut d = decoder(body, WRITER_INVENTORY)?; + let operation = MaintenanceOperation::from_bytes(d.read_bytes()?)?; + let head_digest = Digest::from_bytes(fixed(&mut d)?); + let registry = RegistryVersion::from_bytes(d.read_bytes()?)?; + let boot = EnrollmentRecord::from_bytes(d.read_bytes()?)?; + let process_request = Digest::from_bytes(fixed(&mut d)?); + let process_witness = Digest::from_bytes(fixed(&mut d)?); + let catalog_witness = Digest::from_bytes(fixed(&mut d)?); + let interval = (d.read_i64()?, d.read_i64()?); + let count = d.read_count()?; + if count > MAX_ORIGINAL_CATALOGS { + return Err(CodecError::Limit.into()); + } + let mut catalogs = Vec::with_capacity(count); + for _ in 0..count { + catalogs.push(OriginalCatalogWitness { + application: ApplicationId::from_bytes(fixed(&mut d)?), + tenant: TenantId::from_bytes(fixed(&mut d)?), + source: Digest::from_bytes(fixed(&mut d)?), + heads: Digest::from_bytes(fixed(&mut d)?), + histories: Digest::from_bytes(fixed(&mut d)?), + cells: d.read_u64()?, + owners: d.read_u64()?, + }); + } + let count = d.read_count()?; + if count > MAX_ORIGINAL_WRITERS { + return Err(CodecError::Limit.into()); + } + let page_count = d.read_count()?; + if page_count > super::super::writer_inventory::MAX_WRITER_PAGES { + return Err(CodecError::Limit.into()); + } + let mut pages = Vec::with_capacity(page_count); + for _ in 0..page_count { + pages.push(Digest::from_bytes(fixed(&mut d)?)); + } + d.finish()?; + let record = Self { + basis: OriginalWriterInventoryBasis { + operation, + head_digest, + registry, + boot, + process_request, + process_witness, + catalog_witness, + interval, + }, + catalogs, + count, + pages, + }; + record.validate()?; + Ok(record) + } +} +impl OriginalWriterInventoryPage { + /// Full original Controls and targets, bounded by the one-MiB page envelope. + pub fn to_bytes(&self) -> Result> { + self.validate()?; + let mut e = encoder_limited(WRITER_INVENTORY_PAGE, MAX_PAGE_BYTES)?; + e.write_bytes(self.basis.as_bytes())?; + e.write_u32(self.ordinal)?; + e.write_count(self.entries.len())?; + for row in &self.entries { + e.write_bytes(row.target.tenant().as_bytes())?; + e.write_bytes(row.target.application().as_bytes())?; + e.write_bytes(row.target.namespace().as_bytes())?; + e.write_bytes(row.target.partition())?; + e.write_bytes( + &row.control + .encode() + .map_err(|source| OperationError::Control(Box::new(source)))?, + )?; + } + Ok(e.finish()) + } + /// Refuses counts, envelopes, malformed Controls and trailing data before use. + pub fn from_bytes(body: &[u8]) -> Result { + let mut d = decoder_limited(body, WRITER_INVENTORY_PAGE, MAX_PAGE_BYTES)?; + let basis = Digest::from_bytes(fixed(&mut d)?); + let ordinal = d.read_u32()?; + let count = d.read_count()?; + if count == 0 || count > super::super::writer_inventory::WRITERS_PER_PAGE { + return Err(CodecError::Limit.into()); + } + let mut entries = Vec::with_capacity(count); + for _ in 0..count { + let tenant = TenantId::from_bytes(fixed(&mut d)?); + let application = ApplicationId::from_bytes(fixed(&mut d)?); + let namespace = NamespaceId::from_bytes(fixed(&mut d)?); + let target = CellTarget::new(tenant, application, namespace, d.read_bytes()?) + .map_err(|source| OperationError::Identity(Box::new(source)))?; + let encoded = d.read_bytes()?; + if encoded.len() > 8 * 1024 { + return Err(CodecError::Limit.into()); + } + let control = Control::decode(encoded) + .map_err(|source| OperationError::Control(Box::new(source)))?; + entries.push(OriginalWriterObservation { target, control }); + } + d.finish()?; + let page = Self { + basis, + ordinal, + entries, + }; + page.validate()?; + Ok(page) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/enrollment.rs b/crates/cellule-runtime/src/fleet/operations/enrollment.rs new file mode 100644 index 00000000..9b3f1c2d --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/enrollment.rs @@ -0,0 +1,361 @@ +use crate::identity::{CellTarget, Digest, NodeId, SessionId}; +use crate::node::NodeMode; + +use super::{FleetScope, NodeIntent, OperationError, PublishedPosition, Result, nonzero}; + +/// Exact physical node and boot checked before an enrollment effect starts. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct EnrollmentEndpoint { + /// Physical identity used for retained cordon lookup. + pub node: NodeId, + /// Boot that will participate in the ordinary enrollment protocol. + pub session: SessionId, + /// Committed intent revision checked in the acceptance transaction. + pub intent_revision: u64, +} + +impl EnrollmentEndpoint { + pub(super) fn validate(self) -> Result<()> { + if !nonzero(self.node.as_bytes()) + || !nonzero(self.session.as_bytes()) + || self.intent_revision == 0 + { + return Err(OperationError::Invalid("invalid enrollment endpoint")); + } + Ok(()) + } + + fn check(self, scope: FleetScope, intent: &NodeIntent, receives_role: bool) -> Result<()> { + intent.validate()?; + if intent.scope() != scope + || intent.node() != self.node + || intent.session() != self.session + || intent.revision() != self.intent_revision + || (receives_role && intent.mode() != NodeMode::Active) + { + return Err(OperationError::Conflict); + } + Ok(()) + } +} + +/// Responsibility that ordinary authority or local lifecycle establishes. +/// Follower responsibility belongs to a node-log epoch, not an invented Cell lane. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum EnrollmentRole { + /// Enroll the target boot in its retained mode. A reboot under maintenance + /// may renew/withdraw its lease with acquisition closed; this is not readiness. + Node { + /// Exact retained mode the ordinary boot enrollment must honor. + mode: NodeMode, + }, + /// Open a reader at one checked Cell incarnation and published position. + Reader { + /// Catalog-validated Cell target. + target: CellTarget, + /// Position used by the existing reader opening protocol. + position: PublishedPosition, + }, + /// Enroll the target as follower of the source boot's exact log epoch. + Follower { + /// Nonzero node-log epoch; source boot is carried separately. + log_epoch: u64, + }, +} + +/// Immutable request identity and participating sessions for one responsibility. +/// The request digest is an idempotency index; duplicates compare this entire spec. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct EnrollmentSpec { + /// Shared journal and application namespace. + pub scope: FleetScope, + /// Application-assigned stable request identity. + pub request: Digest, + /// Ordinary protocol whose side effect follows pending acceptance. + pub role: EnrollmentRole, + /// Existing owner or log leader. Absent only for boot enrollment. + pub source: Option, + /// Boot receiving the new responsibility. + pub target: EnrollmentEndpoint, +} + +impl EnrollmentSpec { + /// Returns a stable index without conflating it with a full input comparison. + pub fn key(&self) -> Result { + self.validate()?; + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-enrollment-key.v1\0"); + hash.update(self.scope.fleet.as_bytes()); + hash.update(self.scope.application.as_bytes()); + hash.update(self.request.as_bytes()); + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) + } + + pub(super) fn validate(&self) -> Result<()> { + self.scope.validate()?; + self.target.validate()?; + if !nonzero(self.request.as_bytes()) { + return Err(OperationError::Invalid("zero enrollment request")); + } + if let Some(source) = self.source { + source.validate()?; + if source.node == self.target.node || source.session == self.target.session { + return Err(OperationError::Invalid("enrollment endpoints coincide")); + } + } + match &self.role { + EnrollmentRole::Node { .. } if self.source.is_none() => {} + EnrollmentRole::Reader { target, position } if self.source.is_some() => { + position.validate()?; + if target.application() != self.scope.application { + return Err(OperationError::Invalid( + "reader enrollment application differs", + )); + } + } + EnrollmentRole::Follower { log_epoch } if self.source.is_some() && *log_epoch != 0 => {} + _ => return Err(OperationError::Invalid("invalid enrollment role or source")), + } + Ok(()) + } +} + +/// Retained enrollment progress. Expiry never settles pending responsibility. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[repr(u8)] +pub enum EnrollmentStatus { + /// Accepted before the ordinary side effect; lost replies remain here. + Pending = 1, + /// Ordinary enrollment completion has been checked and retained. + Established = 2, + /// A checked definite refusal proves no responsibility was created. + Refused = 3, + /// Canonical closure or retirement proves the responsibility is settled. + Retired = 4, +} + +/// Checked ordinary-protocol result supplied by the trusted enrollment adapter. +/// Timeout, lease expiry and transport failure are not settlement events. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum EnrollmentEvent { + /// Canonical enrollment completion evidence. + Established(Digest), + /// Canonical definite refusal evidence. + Refused(Digest), + /// Canonical role closure or retirement evidence. + Retired(Digest), +} + +/// Durable advisory obligation, including failed boots and unknown outcomes. +/// Evidence digests identify application-retained canonical evidence. They do +/// not authenticate it or replace ordinary node-log/reader authority checks. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct EnrollmentRecord { + pub(super) spec: EnrollmentSpec, + pub(super) accepted_at_ms: i64, + pub(super) updated_at_ms: i64, + pub(super) status: EnrollmentStatus, + pub(super) established: Option, + pub(super) settlement: Option, +} + +impl EnrollmentRecord { + /// Creates an exclusion tombstone for joined work whose native enrollment + /// never started. Publish only in an atomic absent-or-Pending transaction + /// after checking the exact original request and nonexecution evidence. + /// This does not admit a role or require current receiver intent; delayed + /// acceptance must replay this terminal row before checking intent. + pub fn unexecuted_refusal(spec: EnrollmentSpec, evidence: Digest, now_ms: i64) -> Result { + let record = Self { + spec, + accepted_at_ms: now_ms, + updated_at_ms: now_ms, + status: EnrollmentStatus::Refused, + established: None, + settlement: Some(evidence), + }; + record.validate()?; + Ok(record) + } + + /// Applies one checked protocol event without changing immutable inputs. + pub fn apply(&self, event: EnrollmentEvent, now_ms: i64) -> Result { + match event { + EnrollmentEvent::Established(evidence) => self.establish(evidence, now_ms), + EnrollmentEvent::Refused(evidence) => self.refuse(evidence, now_ms), + EnrollmentEvent::Retired(evidence) => self.retire(evidence, now_ms), + } + } + /// Calculates first acceptance; publish atomically with both intent checks. + /// A draining source may enroll a replacement on an Active receiver. The + /// receiver cannot accept a new reader/follower role under a cordon. Boot + /// enrollment honors its exact retained mode without granting readiness. + /// No I/O is performed. + pub fn pending( + spec: EnrollmentSpec, + source_intent: Option<&NodeIntent>, + target_intent: &NodeIntent, + now_ms: i64, + ) -> Result { + spec.validate()?; + match (spec.source, source_intent) { + (Some(source), Some(intent)) => source.check(spec.scope, intent, false)?, + (None, None) => {} + _ => { + return Err(OperationError::Invalid( + "enrollment source intent is absent", + )); + } + } + let receives_role = match spec.role { + EnrollmentRole::Node { mode } => { + if mode != target_intent.mode() { + return Err(OperationError::Conflict); + } + mode == NodeMode::Active + } + _ => true, + }; + spec.target + .check(spec.scope, target_intent, receives_role)?; + let record = Self { + spec, + accepted_at_ms: now_ms, + updated_at_ms: now_ms, + status: EnrollmentStatus::Pending, + established: None, + settlement: None, + }; + record.validate()?; + Ok(record) + } + + /// Compares complete original inputs before returning an existing acceptance. + /// Do this before rechecking current intents: accepted work may finish after + /// cordon, and a changed boot or payload cannot reuse its request identity. + pub fn validate_replay(&self, spec: &EnrollmentSpec) -> Result<()> { + self.validate()?; + spec.validate()?; + if self.spec != *spec { + return Err(OperationError::Conflict); + } + Ok(()) + } + + /// Confirms ordinary enrollment; duplicate replies retain original time. + pub fn establish(&self, evidence: Digest, now_ms: i64) -> Result { + self.change(EnrollmentStatus::Established, evidence, now_ms) + } + + /// Records checked definite refusal, never a timeout or ambiguous reply. + pub fn refuse(&self, evidence: Digest, now_ms: i64) -> Result { + self.change(EnrollmentStatus::Refused, evidence, now_ms) + } + + /// Records canonical closure, including a pending enrollment whose reply + /// was lost. The adapter verifies that its evidence covers this exact spec. + pub fn retire(&self, evidence: Digest, now_ms: i64) -> Result { + self.change(EnrollmentStatus::Retired, evidence, now_ms) + } + + fn change(&self, status: EnrollmentStatus, evidence: Digest, now_ms: i64) -> Result { + self.validate()?; + if !nonzero(evidence.as_bytes()) || now_ms < self.updated_at_ms { + return Err(OperationError::Invalid( + "invalid enrollment evidence or time", + )); + } + if self.status == status { + let original = if status == EnrollmentStatus::Established { + self.established + } else { + self.settlement + }; + if original != Some(evidence) { + return Err(OperationError::Conflict); + } + return Ok(self.clone()); + } + if self.status != EnrollmentStatus::Pending + && !(self.status == EnrollmentStatus::Established + && status == EnrollmentStatus::Retired) + { + return Err(OperationError::Conflict); + } + let mut next = self.clone(); + next.status = status; + next.updated_at_ms = now_ms; + if status == EnrollmentStatus::Established { + next.established = Some(evidence); + } else { + next.settlement = Some(evidence); + } + next.validate()?; + Ok(next) + } + + /// Returns immutable accepted inputs. + #[must_use] + pub const fn spec(&self) -> &EnrollmentSpec { + &self.spec + } + /// Returns retained progress, independent of membership expiry. + #[must_use] + pub const fn status(&self) -> EnrollmentStatus { + self.status + } + /// Returns first acceptance time, or creation time for an exclusion tombstone. + #[must_use] + pub const fn accepted_at_ms(&self) -> i64 { + self.accepted_at_ms + } + /// Returns the last actual transition time; retries do not refresh it. + #[must_use] + pub const fn updated_at_ms(&self) -> i64 { + self.updated_at_ms + } + /// Returns checked original completion evidence, retained through retirement. + #[must_use] + pub const fn established_evidence(&self) -> Option { + self.established + } + /// Returns checked definite refusal or canonical closure evidence. + #[must_use] + pub const fn settlement_evidence(&self) -> Option { + self.settlement + } + /// Includes pending and failed-session work until canonical settlement. + #[must_use] + pub const fn unresolved(&self) -> bool { + matches!( + self.status, + EnrollmentStatus::Pending | EnrollmentStatus::Established + ) + } + + pub(super) fn validate(&self) -> Result<()> { + self.spec.validate()?; + let proof = |value: Option| value.is_none_or(|d| nonzero(d.as_bytes())); + let shape = match self.status { + EnrollmentStatus::Pending => { + self.established.is_none() + && self.settlement.is_none() + && self.updated_at_ms == self.accepted_at_ms + } + EnrollmentStatus::Established => { + self.established.is_some() && self.settlement.is_none() + } + EnrollmentStatus::Refused => self.established.is_none() && self.settlement.is_some(), + EnrollmentStatus::Retired => self.settlement.is_some(), + }; + if self.accepted_at_ms < 0 + || self.updated_at_ms < self.accepted_at_ms + || !proof(self.established) + || !proof(self.settlement) + || !shape + { + return Err(OperationError::Invalid("invalid retained enrollment")); + } + Ok(()) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/follower_evacuation/mod.rs b/crates/cellule-runtime/src/fleet/operations/follower_evacuation/mod.rs new file mode 100644 index 00000000..a11af581 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/follower_evacuation/mod.rs @@ -0,0 +1,196 @@ +//! Immutable follower replacement history; current authority remains separate. +use super::*; +use crate::identity::Digest; +use std::collections::HashSet; + +mod validation; + +/// Application redundancy policy in the fleet journal transaction domain. +/// The current native node-log protocol supports one or two members. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct FollowerReplacementPolicy { + pub(super) scope: FleetScope, + pub(super) revision: u64, + pub(super) minimum_members: u8, +} +impl FollowerReplacementPolicy { + /// Validates shape; the application authorizes durable policy changes. + pub fn new(scope: FleetScope, revision: u64, minimum_members: u8) -> Result { + let policy = Self { + scope, + revision, + minimum_members, + }; + policy.validate()?; + Ok(policy) + } + /// Journal/application namespace. + #[must_use] + pub const fn scope(self) -> FleetScope { + self.scope + } + /// Monotonic policy revision; equal counts can still name different policy. + #[must_use] + pub const fn revision(self) -> u64 { + self.revision + } + /// Required native member count outside the maintenance node. + #[must_use] + pub const fn minimum_members(self) -> u8 { + self.minimum_members + } +} + +/// Complete original Established replacement request and signed boot identity. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct FollowerReplacementWitness { + /// Original member enrollment, including acceptance and establishment history. + pub enrollment: EnrollmentRecord, + /// Immutable signed boot identity, excluding renewable measurements. + pub boot_identity: Digest, +} + +/// Historical live-owner rotation and complete replacement ensemble. +/// Decoding supplies no native retirement, authentication or finalization rights. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct FollowerEvacuationRecord { + pub(super) operation: MaintenanceOperation, + pub(super) head_digest: Digest, + pub(super) registry: RegistryVersion, + pub(super) policy: FollowerReplacementPolicy, + pub(super) original_key: Digest, + pub(super) original_digest: Digest, + pub(super) retired: Vec, + pub(super) covered_through: u64, + pub(super) source_boot: Digest, + pub(super) replacement_epoch: u64, + pub(super) replacement_evidence: Digest, + pub(super) replacements: Vec, + pub(super) started_at_ms: i64, + pub(super) finished_at_ms: i64, +} +impl FollowerEvacuationRecord { + /// Builds bounded history from trusted native capture; shape alone is advisory. + #[allow(clippy::too_many_arguments)] + pub fn new( + operation: MaintenanceOperation, + barrier: (Digest, RegistryVersion), + policy: FollowerReplacementPolicy, + original: (Digest, Digest), + mut retired: Vec, + covered_through: u64, + source_boot: Digest, + replacement: (u64, Digest), + mut replacements: Vec, + interval: (i64, i64), + ) -> Result { + retired.sort_by_key(|row| *row.spec().target.node.as_bytes()); + replacements.sort_by_key(|entry| *entry.enrollment.spec().target.node.as_bytes()); + let record = Self { + operation, + head_digest: barrier.0, + registry: barrier.1, + policy, + original_key: original.0, + original_digest: original.1, + retired, + covered_through, + source_boot, + replacement_epoch: replacement.0, + replacement_evidence: replacement.1, + replacements, + started_at_ms: interval.0, + finished_at_ms: interval.1, + }; + record.validate()?; + Ok(record) + } + /// Full original operation, including its session and deadline. + #[must_use] + pub fn operation(&self) -> &MaintenanceOperation { + &self.operation + } + /// Full original journal head identity. + #[must_use] + pub const fn head_digest(&self) -> Digest { + self.head_digest + } + /// Original complete registry barrier. + #[must_use] + pub const fn registry(&self) -> RegistryVersion { + self.registry + } + /// Authoritative application policy observed during this capture. + #[must_use] + pub const fn policy(&self) -> FollowerReplacementPolicy { + self.policy + } + /// Exact donor request key; latest pointers never conflate sibling members. + #[must_use] + pub const fn original_key(&self) -> Digest { + self.original_key + } + /// Digest of the original Established donor request, before retirement. + #[must_use] + pub const fn original_digest(&self) -> Digest { + self.original_digest + } + /// Complete original retired ensemble in physical member order. + #[must_use] + pub fn retired(&self) -> &[EnrollmentRecord] { + &self.retired + } + /// Original contiguous object-covered retirement watermark. + #[must_use] + pub const fn covered_through(&self) -> u64 { + self.covered_through + } + /// Immutable signed original leader boot identity. + #[must_use] + pub const fn source_boot(&self) -> Digest { + self.source_boot + } + /// Newer canonical ensemble epoch captured after original rotation. + #[must_use] + pub const fn replacement_epoch(&self) -> u64 { + self.replacement_epoch + } + /// Original canonical enrollment proof identity, without restamping. + #[must_use] + pub const fn replacement_evidence(&self) -> Digest { + self.replacement_evidence + } + /// Every original Established replacement and signed boot identity. + #[must_use] + pub fn replacements(&self) -> &[FollowerReplacementWitness] { + &self.replacements + } + /// Original source endpoint; decoding has already validated its presence. + pub fn source(&self) -> Result { + self.retired + .first() + .and_then(|row| row.spec().source) + .ok_or(OperationError::Invalid("follower source is absent")) + } + /// Original epoch; every original member belongs to this same epoch. + pub fn original_epoch(&self) -> Result { + match self.retired.first().map(|row| &row.spec().role) { + Some(EnrollmentRole::Follower { log_epoch }) => Ok(*log_epoch), + _ => Err(OperationError::Invalid("follower original epoch is absent")), + } + } + /// Original capture interval; replay cannot renew it. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + (self.started_at_ms, self.finished_at_ms) + } + /// Canonical immutable content identity. + pub fn digest(&self) -> Result { + Ok(Digest::from_bytes( + *blake3::hash(&self.to_bytes()?).as_bytes(), + )) + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-runtime/src/fleet/operations/follower_evacuation/tests.rs b/crates/cellule-runtime/src/fleet/operations/follower_evacuation/tests.rs new file mode 100644 index 00000000..a9fc9965 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/follower_evacuation/tests.rs @@ -0,0 +1,223 @@ +use super::*; +use crate::identity::{ApplicationId, NodeId, SessionId}; + +fn record() -> FollowerEvacuationRecord { + let scope = FleetScope { + fleet: Digest::from_bytes([100; 32]), + application: ApplicationId::from_bytes([9; 16]), + }; + let source = NodeIntent::initial( + scope, + NodeId::from_bytes([1; 16]), + SessionId::from_bytes([11; 16]), + ) + .unwrap(); + let member = |node, epoch| { + let intent = NodeIntent::initial( + scope, + NodeId::from_bytes([node; 16]), + SessionId::from_bytes([node + 10; 16]), + ) + .unwrap(); + EnrollmentRecord::pending( + EnrollmentSpec { + scope, + request: Digest::from_bytes([node + epoch as u8 * 10; 32]), + source: Some(EnrollmentEndpoint { + node: source.node(), + session: source.session(), + intent_revision: 1, + }), + target: EnrollmentEndpoint { + node: intent.node(), + session: intent.session(), + intent_revision: 1, + }, + role: EnrollmentRole::Follower { log_epoch: epoch }, + }, + Some(&source), + &intent, + 10, + ) + .unwrap() + .establish(Digest::from_bytes([70; 32]), 20) + .unwrap() + }; + let original = member(2, 1); + let retired = vec![ + original.retire(Digest::from_bytes([71; 32]), 30).unwrap(), + member(3, 1) + .retire(Digest::from_bytes([72; 32]), 30) + .unwrap(), + ]; + let mut operation = MaintenanceOperation::new( + OperationId::from_bytes([80; 16]).unwrap(), + Digest::from_bytes([81; 32]), + NodeId::from_bytes([2; 16]), + SessionId::from_bytes([12; 16]), + 2, + 0, + 60_000, + ) + .unwrap(); + operation.apply(MaintenanceEvent::Cordoned, 0).unwrap(); + operation + .apply(MaintenanceEvent::BeginEvacuation, 0) + .unwrap(); + FollowerEvacuationRecord::new( + operation, + ( + Digest::from_bytes([82; 32]), + RegistryVersion::new(scope).unwrap().bootstrap(0).unwrap(), + ), + FollowerReplacementPolicy::new(scope, 1, 2).unwrap(), + ( + original.spec().key().unwrap(), + Digest::from_bytes(*blake3::hash(&original.to_bytes().unwrap()).as_bytes()), + ), + retired, + 8, + Digest::from_bytes([83; 32]), + (2, Digest::from_bytes([84; 32])), + vec![ + FollowerReplacementWitness { + enrollment: member(3, 2), + boot_identity: Digest::from_bytes([85; 32]), + }, + FollowerReplacementWitness { + enrollment: member(4, 2), + boot_identity: Digest::from_bytes([86; 32]), + }, + ], + (40, 50), + ) + .unwrap() +} + +#[test] +fn follower_policy_record_roundtrips_complete_original_and_replacement_ensembles() { + let record = record(); + assert_eq!( + FollowerEvacuationRecord::from_bytes(&record.to_bytes().unwrap()).unwrap(), + record + ); + assert_eq!( + FollowerReplacementPolicy::from_bytes(&record.policy().to_bytes().unwrap()).unwrap(), + record.policy() + ); + assert_eq!(record.retired().len(), 2); + assert_eq!(record.replacements().len(), 2); + assert_ne!(record.original_digest(), record.digest().unwrap()); +} +#[test] +fn follower_policy_record_rejects_missing_duplicate_foreign_and_unretired_members() { + let original = record(); + for mutation in 0..7 { + let mut row = original.clone(); + match mutation { + 0 => { + row.retired.remove(0); + } + 1 => { + row.retired[1] = row.retired[0].clone(); + } + 2 => { + row.retired[0].status = EnrollmentStatus::Established; + row.retired[0].settlement = None; + } + 3 => { + row.retired[0].spec.source.as_mut().unwrap().session = + SessionId::from_bytes([88; 16]); + } + 4 => { + row.replacements[1] = row.replacements[0].clone(); + } + 5 => { + row.replacements[0].enrollment.spec.target.node = row.operation.node(); + } + _ => { + row.replacements[0].enrollment.spec.role = + EnrollmentRole::Follower { log_epoch: 1 }; + } + }; + assert!(row.to_bytes().is_err(), "mutation {mutation}"); + } +} +#[test] +fn follower_policy_record_rejects_stale_revision_scope_epoch_and_capture_times() { + let original = record(); + for mutation in 0..9 { + let mut row = original.clone(); + match mutation { + 0 => row.policy.revision = 0, + 1 => row.policy.minimum_members = 0, + 2 => row.policy.minimum_members = 3, + 3 => row.registry = RegistryVersion::new(row.policy.scope).unwrap(), + 4 => row.policy.scope.fleet = Digest::from_bytes([89; 32]), + 5 => row.replacement_epoch = 1, + 6 => row.head_digest = Digest::from_bytes([0; 32]), + 7 => row.finished_at_ms = 30_041, + _ => row.finished_at_ms = 39, + } + assert!(row.to_bytes().is_err(), "mutation {mutation}"); + } +} +#[test] +fn follower_policy_record_preserves_retirement_across_deadline_and_boot_adoption() { + let mut row = record(); + let retired = row.retired.clone(); + row.operation + .apply( + MaintenanceEvent::SessionReplaced(SessionId::from_bytes([90; 16])), + 40, + ) + .unwrap(); + row.operation.apply(MaintenanceEvent::Cordoned, 40).unwrap(); + row.operation + .apply(MaintenanceEvent::BeginEvacuation, 40) + .unwrap(); + row.operation + .apply(MaintenanceEvent::ExtendDeadline(70_000), 40) + .unwrap(); + row.operation + .apply( + MaintenanceEvent::ReadyToClose(DrainEvidence { + node: row.operation.node(), + session: row.operation.session(), + remaining_cells: 0, + unresolved_attempts: 0, + relocated: true, + readers_settled: true, + followers_settled: true, + facilities_closed: false, + stopped: false, + withdrawn: false, + }), + 40, + ) + .unwrap(); + assert_eq!( + FollowerEvacuationRecord::from_bytes(&row.to_bytes().unwrap()) + .unwrap() + .retired(), + retired + ); +} +#[test] +fn follower_policy_codecs_reject_truncation_trailing_unknown_and_oversized_envelopes() { + let record = record(); + let bytes = record.to_bytes().unwrap(); + for length in 0..bytes.len() { + assert!(FollowerEvacuationRecord::from_bytes(&bytes[..length]).is_err()); + } + let mut extra = bytes.clone(); + extra.push(0); + assert!(FollowerEvacuationRecord::from_bytes(&extra).is_err()); + let policy = record.policy.to_bytes().unwrap(); + assert!(FollowerEvacuationRecord::from_bytes(&policy).is_err()); + assert!(FollowerReplacementPolicy::from_bytes(&bytes).is_err()); + assert!(FollowerEvacuationRecord::from_bytes(&vec![0; MAX_PAGE_BYTES as usize + 1]).is_err()); + assert!( + FollowerReplacementPolicy::from_bytes(&vec![0; MAX_RECORD_BYTES as usize + 1]).is_err() + ); +} diff --git a/crates/cellule-runtime/src/fleet/operations/follower_evacuation/validation.rs b/crates/cellule-runtime/src/fleet/operations/follower_evacuation/validation.rs new file mode 100644 index 00000000..be7d34d3 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/follower_evacuation/validation.rs @@ -0,0 +1,119 @@ +use super::*; +impl FollowerReplacementPolicy { + pub(in crate::fleet::operations) fn validate(self) -> Result<()> { + self.scope.validate()?; + if self.revision == 0 || !(1..=2).contains(&self.minimum_members) { + return Err(OperationError::Invalid( + "invalid follower replacement policy", + )); + } + Ok(()) + } +} +impl FollowerEvacuationRecord { + pub(in crate::fleet::operations) fn validate(&self) -> Result<()> { + self.operation.validate()?; + self.registry.confirm(self.registry)?; + self.policy.validate()?; + if !matches!( + self.operation.phase(), + MaintenancePhase::Evacuating | MaintenancePhase::Closing + ) || self.registry.scope() != self.policy.scope + || self.retired.is_empty() + || self.retired.len() > 2 + || self.replacements.len() < usize::from(self.policy.minimum_members) + || self.replacements.len() > 2 + || [ + self.head_digest, + self.original_key, + self.original_digest, + self.source_boot, + self.replacement_evidence, + ] + .iter() + .any(|digest| !nonzero(digest.as_bytes())) + || self.replacement_epoch <= self.original_epoch()? + || self.covered_through == u64::MAX + || self.started_at_ms < self.operation.created_at_ms + || self.finished_at_ms < self.started_at_ms + || self.finished_at_ms - self.started_at_ms > 30_000 + || self.finished_at_ms >= self.operation.deadline_ms() + { + return Err(OperationError::Invalid( + "invalid follower evacuation capture", + )); + } + let source = self.source()?; + let epoch = self.original_epoch()?; + if source.node == self.operation.node() { + return Err(OperationError::Invalid( + "follower source is maintenance donor", + )); + } + let mut originals = HashSet::new(); + let mut sessions = HashSet::new(); + let mut after = None; + let mut donor = false; + for row in &self.retired { + row.validate_replay(row.spec())?; + let endpoint = row.spec().target; + let node = *endpoint.node.as_bytes(); + if row.status() != EnrollmentStatus::Retired + || row.established_evidence().is_none() + || row.spec().scope != self.policy.scope + || row.spec().source != Some(source) + || row.spec().role != (EnrollmentRole::Follower { log_epoch: epoch }) + || row.updated_at_ms() > self.finished_at_ms + || !originals.insert(row.spec().key()?) + || !sessions.insert(endpoint.session) + || after.is_some_and(|previous| node <= previous) + { + return Err(OperationError::Invalid( + "invalid original follower ensemble", + )); + } + if row.spec().key()? == self.original_key { + if endpoint.node != self.operation.node() { + return Err(OperationError::Invalid("original follower donor differs")); + } + donor = true; + } + after = Some(node); + } + if !donor { + return Err(OperationError::Invalid("original donor request is absent")); + } + let mut keys = HashSet::new(); + let mut sessions = HashSet::new(); + let mut after = None; + for entry in &self.replacements { + let row = &entry.enrollment; + row.validate_replay(row.spec())?; + let endpoint = row.spec().target; + let node = *endpoint.node.as_bytes(); + if row.status() != EnrollmentStatus::Established + || row.spec().scope != self.policy.scope + || row.spec().source.is_none_or(|current| { + current.node != source.node || current.session != source.session + }) + || row.spec().role + != (EnrollmentRole::Follower { + log_epoch: self.replacement_epoch, + }) + || endpoint.node == self.operation.node() + || endpoint.session == self.operation.session() + || row.updated_at_ms() > self.finished_at_ms + || !nonzero(entry.boot_identity.as_bytes()) + || !keys.insert(row.spec().key()?) + || !sessions.insert(endpoint.session) + || after.is_some_and(|previous| node <= previous) + { + return Err(OperationError::Invalid( + "invalid follower replacement ensemble", + )); + } + after = Some(node); + } + Ok(()) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/history.rs b/crates/cellule-runtime/src/fleet/operations/history.rs new file mode 100644 index 00000000..82f629cd --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/history.rs @@ -0,0 +1,128 @@ +use crate::identity::Digest; + +use super::{ + FleetScope, MAX_PAGE_ENTRIES, MoveAttempt, OperationError, OperationId, Result, nonzero, +}; + +/// Immutable progress chain reachable from the CAS-published fleet head. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ProgressHead { + /// Digest of the last canonical page published with permit retirement. + pub digest: Digest, + /// Monotonic page number across this fleet journal. + pub sequence: u64, +} + +impl ProgressHead { + pub(super) fn validate(self) -> Result<()> { + if self.sequence == 0 || !nonzero(self.digest.as_bytes()) { + return Err(OperationError::Invalid("invalid progress head")); + } + Ok(()) + } +} + +/// Bounded immutable terminal attempts, including incarnation and movement time. +/// +/// The adapter must persist this page before CAS-publishing the head that +/// retires its attempts. A page PUT alone is not committed progress. A failed +/// CAS leaves an unreachable page and retains all original movement permits. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct ProgressPage { + pub(super) scope: FleetScope, + pub(super) operation: OperationId, + pub(super) sequence: u64, + pub(super) previous: Option, + pub(super) entries: Vec, +} + +impl ProgressPage { + /// Constructs a canonical page containing only completed, cleaned attempts. + pub fn new( + scope: FleetScope, + operation: OperationId, + sequence: u64, + previous: Option, + entries: Vec, + ) -> Result { + let page = Self { + scope, + operation, + sequence, + previous, + entries, + }; + page.validate()?; + Ok(page) + } + + /// Returns the exact journal scope. + #[must_use] + pub const fn scope(&self) -> FleetScope { + self.scope + } + /// Returns the operation whose history this page contains. + #[must_use] + pub const fn operation(&self) -> OperationId { + self.operation + } + /// Returns the monotonically assigned fleet page number. + #[must_use] + pub const fn sequence(&self) -> u64 { + self.sequence + } + /// Returns the preceding committed page's digest. + #[must_use] + pub const fn previous(&self) -> Option { + self.previous + } + /// Returns bounded terminal attempts in ascending allocation order. + #[must_use] + pub fn entries(&self) -> &[MoveAttempt] { + &self.entries + } + /// Returns the digest the adapter must publish with the successor head. + pub fn digest(&self) -> Result { + Ok(Digest::from_bytes( + *blake3::hash(&self.to_bytes()?).as_bytes(), + )) + } + + pub(super) fn validate(&self) -> Result<()> { + self.scope.validate()?; + if !nonzero(self.operation.as_bytes()) + || self.sequence == 0 + || (self.sequence == 1) != self.previous.is_none() + || self + .previous + .is_some_and(|digest| !nonzero(digest.as_bytes())) + || self.entries.is_empty() + || self.entries.len() > MAX_PAGE_ENTRIES + { + return Err(OperationError::Invalid("invalid progress page")); + } + let mut previous = 0; + for attempt in &self.entries { + attempt.validate()?; + if attempt + .recovered() + .is_some_and(|e| e.recovery.basis().scope != self.scope) + { + return Err(OperationError::Invalid( + "recovery history belongs to another fleet", + )); + } + if !attempt.can_retire() + || attempt.spec.id.operation != self.operation + || attempt.spec.target.application() != self.scope.application + || attempt.spec.id.sequence <= previous + { + return Err(OperationError::Invalid( + "unproven or unordered historical attempt", + )); + } + previous = attempt.spec.id.sequence; + } + Ok(()) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/inspection.rs b/crates/cellule-runtime/src/fleet/operations/inspection.rs new file mode 100644 index 00000000..971d91ed --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/inspection.rs @@ -0,0 +1,297 @@ +use crate::identity::{Digest, NodeId, SessionId}; + +use super::{ + AttemptPhase, FleetAction, FleetActionKind, FleetActionOutcome, FleetHead, FleetOutcome, + MaintenanceAction, MovementAction, OperationError, RegistryVersion, Result, nonzero, +}; + +/// One caller-assigned nonce and bounded interval for a fresh, read-only check. +/// +/// This request confers no authority. Authenticate its origin and authorize it +/// against the current journal before gathering evidence. A new reconciliation +/// pass uses a new nonce; retrying a pass keeps its exact request. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct FleetInspectionRequest { + pub(super) action: FleetAction, + pub(super) registry: RegistryVersion, + pub(super) nonce: Digest, + pub(super) node: NodeId, + pub(super) session: SessionId, + pub(super) deadline_ms: i64, +} + +impl FleetInspectionRequest { + /// Binds a current Inspect envelope to an exact physical node and boot. + pub fn new( + action: FleetAction, + registry: RegistryVersion, + nonce: Digest, + node: NodeId, + session: SessionId, + deadline_ms: i64, + ) -> Result { + let request = Self { + action, + registry, + nonce, + node, + session, + deadline_ms, + }; + request.validate()?; + Ok(request) + } + + /// Returns the exact journal envelope whose state is being inspected. + #[must_use] + pub const fn action(&self) -> &FleetAction { + &self.action + } + /// Returns the exact retained intent/enrollment barrier for this capture. + #[must_use] + pub const fn registry(&self) -> RegistryVersion { + self.registry + } + /// Returns the caller's unique check identity, not an effect idempotency key. + #[must_use] + pub const fn nonce(&self) -> Digest { + self.nonce + } + /// Returns the authenticated physical endpoint required for the reply. + #[must_use] + pub const fn node(&self) -> NodeId { + self.node + } + /// Returns the endpoint's exact boot. + #[must_use] + pub const fn session(&self) -> SessionId { + self.session + } + /// Returns the exclusive end of the requested capture interval. + #[must_use] + pub const fn deadline_ms(&self) -> i64 { + self.deadline_ms + } + + /// Checks the exact read-only endpoint, including a successor after release. + /// This cannot authorize an effect on that endpoint. The adapter separately + /// authenticates its boot and checks the retained registry and journal head. + pub fn validate_endpoint(&self, node: NodeId, session: SessionId) -> Result<()> { + self.validate()?; + if node != self.node || session != self.session { + return Err(OperationError::Fenced); + } + Ok(()) + } + + /// Identifies all request inputs, including nonce, endpoint and authorization. + /// Different passes cannot reuse the stable effect key as fresh evidence. + pub fn key(&self) -> Result { + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule.fleet-inspection-key.v1\0"); + hash.update(&self.to_bytes()?); + Ok(Digest::from_bytes(*hash.finalize().as_bytes())) + } + + /// Checks the current head and capture interval in a journal read transaction. + /// No acceptance/result record is created and no effect is authorized. + pub fn authorize_against( + &self, + head: &FleetHead, + registry: RegistryVersion, + now_ms: i64, + ) -> Result<()> { + self.validate()?; + registry.validate()?; + if registry != self.registry { + return Err(OperationError::Conflict); + } + if now_ms >= self.deadline_ms { + return Err(OperationError::Deadline); + } + self.action.authorize_against(head, now_ms) + } + + pub(super) fn validate(&self) -> Result<()> { + self.registry.validate()?; + if self.registry.scope() != self.action.scope() { + return Err(OperationError::Conflict); + } + self.action.validate()?; + if !matches!( + self.action.kind(), + FleetActionKind::Movement { + action: MovementAction::Inspect, + .. + } | FleetActionKind::Maintenance { + action: MaintenanceAction::Inspect, + .. + } + ) || !nonzero(self.nonce.as_bytes()) + || self.deadline_ms <= self.action.issued_at_ms() + { + return Err(OperationError::Invalid("invalid fresh inspection request")); + } + if self + .action + .validate_endpoint(self.node, self.session) + .is_err() + { + let FleetActionKind::Movement { + action: MovementAction::Inspect, + attempt, + } = self.action.kind() + else { + return Err(OperationError::Fenced); + }; + let spec = attempt.spec(); + // A clean release permits observation of an ordinary-acquisition + // winner. Unknown release and failed-source recovery keep their + // separate proof paths; another actor cannot establish either. + if attempt.released().is_none() + || !matches!( + attempt.phase(), + AttemptPhase::Released + | AttemptPhase::Activating + | AttemptPhase::Activated + | AttemptPhase::CleaningReceiver + ) + || !nonzero(self.node.as_bytes()) + || !nonzero(self.session.as_bytes()) + || self.node == spec.source_node + || self.session == spec.source + || (self.session == spec.destination && self.node != spec.destination_node) + { + return Err(OperationError::Fenced); + } + } + Ok(()) + } + + fn validate_outcome(&self, outcome: &FleetActionOutcome) -> Result<()> { + if self + .action + .validate_endpoint(self.node, self.session) + .is_ok() + { + outcome.validate_for(&self.action)?; + } else { + // Keep ordinary effect/result validation strict. Only this exact + // request-bound read can report a different serving endpoint. + outcome.validate()?; + if outcome.scope != self.action.scope() || outcome.action_key != self.action.key()? { + return Err(OperationError::Conflict); + } + if !matches!( + outcome.outcome, + FleetOutcome::Activated(_) + | FleetOutcome::Unknown + | FleetOutcome::Blocked(_) + | FleetOutcome::Rejected(_) + ) { + return Err(OperationError::Invalid( + "successor inspection cannot prove receiver effects", + )); + } + } + if let FleetOutcome::Activated(evidence) = &outcome.outcome + && let FleetActionKind::Movement { attempt, .. } = self.action.kind() + && attempt.released().is_some() + { + attempt.validate_activation(evidence)?; + } + Ok(()) + } +} + +/// Checked observation captured for one exact request, separate from effect history. +/// +/// A decoder verifies shape and binding only. The trusted node adapter must +/// actually inspect current authority and actor readiness within this interval. +/// Retained Released/cleanup facts remain historical; current serving requires +/// a fresh actor-backed Activated/Recovered observation. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct FleetInspectionObservation { + pub(super) request: FleetInspectionRequest, + pub(super) capture_started_at_ms: i64, + pub(super) outcome: FleetActionOutcome, +} + +impl FleetInspectionObservation { + /// Retains the actual interval and checked endpoint response. + pub fn new( + request: FleetInspectionRequest, + capture_started_at_ms: i64, + outcome: FleetActionOutcome, + ) -> Result { + let observation = Self { + request, + capture_started_at_ms, + outcome, + }; + observation.validate()?; + Ok(observation) + } + /// Returns the exact request, including nonce and committed state. + #[must_use] + pub const fn request(&self) -> &FleetInspectionRequest { + &self.request + } + /// Returns when this node began the current evidence capture. + #[must_use] + pub const fn capture_started_at_ms(&self) -> i64 { + self.capture_started_at_ms + } + /// Returns the original completed capture time; delivery cannot refresh it. + #[must_use] + pub const fn capture_finished_at_ms(&self) -> i64 { + self.outcome.observed_at_ms + } + /// Returns the bounded observation with independently checked effect facts. + #[must_use] + pub const fn outcome(&self) -> &FleetActionOutcome { + &self.outcome + } + + /// Checks full request identity, absence of future time, and a finite age bound. + /// The caller also authenticates the response origin and rechecks its journal + /// generation before publishing dependent decisions. + pub fn validate_for( + &self, + request: &FleetInspectionRequest, + now_ms: i64, + max_age_ms: i64, + ) -> Result<()> { + self.validate()?; + request.validate()?; + if self.request != *request { + return Err(OperationError::Conflict); + } + if max_age_ms <= 0 || now_ms < self.capture_finished_at_ms() { + return Err(OperationError::Invalid( + "invalid inspection consumption time", + )); + } + // Bound the entire capture interval, not just a freshly stamped delivery. + if now_ms - self.capture_started_at_ms > max_age_ms || now_ms >= request.deadline_ms { + return Err(OperationError::Deadline); + } + Ok(()) + } + + pub(super) fn validate(&self) -> Result<()> { + self.request.validate()?; + self.request.validate_outcome(&self.outcome)?; + if self.outcome.node != self.request.node + || self.outcome.session != self.request.session + || self.capture_started_at_ms < self.request.action.issued_at_ms() + || self.outcome.observed_at_ms < self.capture_started_at_ms + || self.outcome.observed_at_ms >= self.request.deadline_ms + { + return Err(OperationError::Invalid( + "inspection origin or interval mismatch", + )); + } + Ok(()) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/intent.rs b/crates/cellule-runtime/src/fleet/operations/intent.rs new file mode 100644 index 00000000..9ae90547 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/intent.rs @@ -0,0 +1,169 @@ +use crate::identity::{NodeId, SessionId}; +use crate::node::NodeMode; + +use super::{ + FleetScope, MaintenanceOperation, MaintenancePhase, OperationError, OperationId, Result, + nonzero, +}; + +/// Application-journal intent keyed by physical NodeId, retained across reboots. +/// +/// The adapter commits maintenance intent with the fleet-head CAS, keeps older +/// nodes' intent after a later operation starts, and checks it before runtime +/// acquisition or readiness opens. An intent never grants Cell authority. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct NodeIntent { + pub(super) scope: FleetScope, + pub(super) node: NodeId, + pub(super) session: SessionId, + pub(super) revision: u64, + pub(super) mode: NodeMode, + pub(super) operation: Option, +} + +impl NodeIntent { + /// Constructs the first authorized serving intent in an empty node registry. + /// Existing registry entries must be conditionally updated, never replaced + /// with this constructor to erase a maintenance cordon. + pub fn initial(scope: FleetScope, node: NodeId, session: SessionId) -> Result { + let intent = Self { + scope, + node, + session, + revision: 1, + mode: NodeMode::Active, + operation: None, + }; + intent.validate()?; + Ok(intent) + } + + /// Derives the durable desired drain mode independently of local progress. + pub fn maintenance(scope: FleetScope, operation: &MaintenanceOperation) -> Result { + operation.validate()?; + let intent = Self { + scope, + node: operation.node, + session: operation.session, + revision: operation.intent_revision, + mode: NodeMode::Draining, + operation: Some(operation.id), + }; + intent.validate()?; + Ok(intent) + } + + /// Advances a retained intent from the operation in the same journal CAS. + /// The adapter must load this physical-node row inside that transaction. + /// Progress alone does not change desired mode or reopen a completed node. + pub fn advance_maintenance(&self, operation: &MaintenanceOperation) -> Result { + self.validate()?; + let next = Self::maintenance(self.scope, operation)?; + if next == *self { + return Ok(self.clone()); + } + if next.node != self.node + || next.revision <= self.revision + || (self.mode != NodeMode::Active && self.operation != next.operation) + || (self.mode == NodeMode::Active && next.session != self.session) + { + return Err(OperationError::Conflict); + } + Ok(next) + } + + /// Rebinds an already Active physical node to a separately validated boot. + /// The application must prove old-session withdrawal or complete failed-boot + /// closure (including process/accepted external work joining and all original + /// roles settled), and new-session enrollment. Expiry/takeover is insufficient. + /// This cannot reopen a retained cordon after a reboot. + pub fn rebind_active(&self, session: SessionId, revision: u64) -> Result { + self.validate()?; + if self.mode != NodeMode::Active || session == self.session || revision <= self.revision { + return Err(OperationError::Conflict); + } + let mut next = self.clone(); + next.session = session; + next.revision = revision; + next.validate()?; + Ok(next) + } + + /// Calculates an authorized newer Active intent after confirmed maintenance. + /// Returning to service uses a different validated boot session. Publishing + /// this record and enrolling that session remain application responsibilities. + pub fn return_to_service( + &self, + completed: &MaintenanceOperation, + new_session: SessionId, + revision: u64, + ) -> Result { + self.validate()?; + completed.validate()?; + if self.mode == NodeMode::Active + || self.operation != Some(completed.id) + || self.node != completed.node + || self.session != completed.session + || self.revision != completed.intent_revision + || completed.phase != MaintenancePhase::Completed + || revision <= self.revision + || new_session == self.session + { + return Err(OperationError::Invalid("return to service is not proven")); + } + let next = Self { + scope: self.scope, + node: self.node, + session: new_session, + revision, + mode: NodeMode::Active, + operation: None, + }; + next.validate()?; + Ok(next) + } + + /// Returns the journal authorization scope. + #[must_use] + pub const fn scope(&self) -> FleetScope { + self.scope + } + /// Returns the physical identity used for startup cordon lookup. + #[must_use] + pub const fn node(&self) -> NodeId { + self.node + } + /// Returns the session observed by the latest committed intent. + #[must_use] + pub const fn session(&self) -> SessionId { + self.session + } + /// Returns the monotonic desired-mode revision. + #[must_use] + pub const fn revision(&self) -> u64 { + self.revision + } + /// Returns the mode a new boot must honor before readiness. + #[must_use] + pub const fn mode(&self) -> NodeMode { + self.mode + } + /// Returns the maintenance operation retaining the cordon. + #[must_use] + pub const fn operation(&self) -> Option { + self.operation + } + + pub(super) fn validate(&self) -> Result<()> { + self.scope.validate()?; + if !nonzero(self.node.as_bytes()) + || !nonzero(self.session.as_bytes()) + || self.revision == 0 + || (self.mode == NodeMode::Active) != self.operation.is_none() + || self.operation.is_some_and(|id| !nonzero(id.as_bytes())) + { + return Err(OperationError::Invalid("invalid physical-node intent")); + } + Ok(()) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/journal.rs b/crates/cellule-runtime/src/fleet/operations/journal.rs new file mode 100644 index 00000000..dea63f2d --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/journal.rs @@ -0,0 +1,512 @@ +use crate::identity::SessionId; + +use super::{ + AttemptEvent, AttemptId, FleetProfile, FleetScope, MAX_ACTIVE_ATTEMPTS, MAX_RESTORE_BYTES, + MaintenanceEvent, MaintenanceOperation, MaintenancePhase, MoveAttempt, MoveAttemptSpec, + MovementAction, NodeIntent, OperationError, ProgressHead, ProgressPage, Result, nonzero, +}; + +/// Journal controller lease; it never confers Cell ownership. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ControllerLease { + /// Exact controller process that owns the journal epoch. + pub claimant: SessionId, + /// Monotonic fencing epoch, including same-process reacquisition after expiry. + pub epoch: u64, + /// Exclusive expiration time in adapter-supplied logical milliseconds. + pub expires_at_ms: i64, +} + +/// One CAS-published deterministic state change. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum JournalTransition { + /// Atomically persist maintenance intent before cordoning its node. + BeginMaintenance(MaintenanceOperation), + /// Advance or diagnose the current maintenance operation. + Maintenance(MaintenanceEvent), + /// Allocate one exact attempt and its fleet count/byte permit. + Allocate(MoveAttemptSpec), + /// Persist a dispatch intent or confirmed exact-attempt outcome. + Attempt { + /// Attempt receiving this result or dispatch intent. + id: AttemptId, + /// Validated movement transition. + event: AttemptEvent, + }, + /// In the same CAS transaction, prove no acceptance exists for this exact + /// effect, endpoint and attempt. Fence delayed old envelopes by advancing + /// the head revision, then retry or cancel when admission has expired. + /// A separate lookup cannot satisfy this precondition. Never use this to + /// erase an accepted effect or reclaim its permit. + ResolveUnaccepted { + /// Still-charged exact movement attempt. + id: AttemptId, + /// Effect whose original acceptance is atomically proved absent. + effect: MovementAction, + }, + /// Publish exact terminal history and free its permits in the same CAS. + /// The adapter persists this immutable page before publishing the head. + Retire { + /// All retired attempts must exactly match the current active records. + progress: ProgressPage, + }, +} + +/// Bounded atomically published controller state and unresolved movement permits. +/// +/// History pages are persisted by the adapter before CAS publication. All +/// unresolved attempts live in this head so controller failover cannot forget +/// remote work or oversubscribe the fleet's budget. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct FleetHead { + pub(super) scope: FleetScope, + pub(super) revision: u64, + pub(super) last_observed_ms: i64, + pub(super) controller: Option, + pub(super) maintenance: Option, + pub(super) next_sequence: u64, + pub(super) attempts: Vec, + pub(super) progress: Option, +} + +impl FleetHead { + /// Creates an empty journal; the adapter installs it with create-if-absent. + pub fn new(scope: FleetScope, now_ms: i64) -> Result { + scope.validate()?; + if now_ms < 0 { + return Err(OperationError::Invalid("negative journal time")); + } + Ok(Self { + scope, + revision: 0, + last_observed_ms: now_ms, + controller: None, + maintenance: None, + next_sequence: 1, + attempts: Vec::new(), + progress: None, + }) + } + + /// Returns the exact authorization scope. + #[must_use] + pub const fn scope(&self) -> FleetScope { + self.scope + } + /// Returns the revision an adapter must compare when publishing. + #[must_use] + pub const fn revision(&self) -> u64 { + self.revision + } + /// Returns the current journal controller lease. + #[must_use] + pub const fn controller(&self) -> Option { + self.controller + } + /// Returns the durable physical-node maintenance intent and progress. + #[must_use] + pub const fn maintenance(&self) -> Option<&MaintenanceOperation> { + self.maintenance.as_ref() + } + /// Returns all unresolved attempts, including unknown remote outcomes. + #[must_use] + pub fn attempts(&self) -> &[MoveAttempt] { + &self.attempts + } + /// Returns the next sequence the current controller may allocate once. + #[must_use] + pub const fn next_sequence(&self) -> u64 { + self.next_sequence + } + /// Returns the committed terminal-history chain, retained on controller expiry. + #[must_use] + pub const fn progress(&self) -> Option { + self.progress + } + + /// Derives the current maintenance intent for atomic publication by the adapter. + /// Older nodes' intent must remain in the adapter's physical-node registry. + pub fn node_intent(&self) -> Result> { + self.maintenance + .as_ref() + .map(|operation| NodeIntent::maintenance(self.scope, operation)) + .transpose() + } + + /// Builds one immutable retirement page without changing the current head. + /// The adapter writes it first, then CAS-publishes `JournalTransition::Retire`. + pub fn retirement_page(&self, ids: &[AttemptId]) -> Result { + if ids.is_empty() || ids.len() > MAX_ACTIVE_ATTEMPTS { + return Err(OperationError::Invalid("invalid retirement batch")); + } + let entries = self + .attempts + .iter() + .filter(|attempt| ids.contains(&attempt.spec.id)) + .cloned() + .collect::>(); + if entries.len() != ids.len() { + return Err(OperationError::NotFound); + } + let operation = entries[0].spec.id.operation; + ProgressPage::new( + self.scope, + operation, + self.next_progress_sequence()?, + self.progress.map(|head| head.digest), + entries, + ) + } + + fn next_progress_sequence(&self) -> Result { + self.progress.map_or(Ok(1), |head| { + head.sequence + .checked_add(1) + .ok_or(OperationError::Invalid("progress sequence overflow")) + }) + } + /// Returns restore demand still charged to unresolved planned attempts. + #[must_use] + pub fn reserved_restore_bytes(&self) -> u64 { + self.attempts.iter().map(|a| a.spec.cost.disk_bytes).sum() + } + + /// Calculates an exact CAS successor for acquiring or renewing the controller. + /// Expiry never removes an attempt or frees its resource permit. + pub fn claim( + &self, + profile: FleetProfile, + expected_revision: u64, + claimant: SessionId, + now_ms: i64, + ) -> Result { + let profile = profile.validate()?; + self.check_revision_and_time(expected_revision, now_ms)?; + if !nonzero(claimant.as_bytes()) { + return Err(OperationError::Invalid("zero controller claimant")); + } + let epoch = match self.controller { + Some(current) if now_ms < current.expires_at_ms => { + if current.claimant != claimant { + return Err(OperationError::Fenced); + } + current.epoch + } + Some(current) => current + .epoch + .checked_add(1) + .ok_or(OperationError::Invalid("controller epoch overflow"))?, + None => 1, + }; + let mut next = self.clone(); + next.controller = Some(ControllerLease { + claimant, + epoch, + expires_at_ms: now_ms + .checked_add(profile.controller_lease_ms) + .ok_or(OperationError::Invalid("controller lease time overflow"))?, + }); + next.finish_transition(now_ms)?; + Ok(next) + } + + /// Calculates a pure successor; the adapter must CAS the exact current head. + /// An event is accepted only by the live controller epoch and exact attempt. + pub fn transition( + &self, + profile: FleetProfile, + expected_revision: u64, + controller_epoch: u64, + now_ms: i64, + transition: JournalTransition, + ) -> Result { + let profile = profile.validate()?; + self.check_revision_and_time(expected_revision, now_ms)?; + let controller = self.controller.ok_or(OperationError::Fenced)?; + if controller.epoch != controller_epoch || now_ms >= controller.expires_at_ms { + return Err(OperationError::Fenced); + } + // Absence resolution must fence delayed authorizations even when the + // attempt already has no Unknown marker and otherwise stays identical. + let fence_unaccepted = matches!(transition, JournalTransition::ResolveUnaccepted { .. }); + let mut next = self.clone(); + match transition { + JournalTransition::BeginMaintenance(operation) => { + operation.validate()?; + if operation.phase != MaintenancePhase::Requested + || operation.created_at_ms > now_ms + { + return Err(OperationError::Invalid( + "maintenance request is not initial", + )); + } + if let Some(old) = &self.maintenance { + if old.id == operation.id { + if old.request_digest != operation.request_digest + || old.node != operation.node + || old.created_at_ms != operation.created_at_ms + { + return Err(OperationError::Conflict); + } + return Ok(self.clone()); + } + if old.phase != MaintenancePhase::Completed { + return Err(OperationError::Busy); + } + } + if now_ms >= operation.deadline_ms { + return Err(OperationError::Deadline); + } + next.maintenance = Some(operation); + } + JournalTransition::Maintenance(event) => { + let operation = next.maintenance.as_mut().ok_or(OperationError::NotFound)?; + if matches!( + event, + MaintenanceEvent::ReadyToClose(_) | MaintenanceEvent::Stopped(_) + ) && next.attempts.iter().any(|a| { + a.spec.source_node == operation.node + || a.spec.destination_node == operation.node + }) { + return Err(OperationError::Invalid( + "node still has unresolved movement", + )); + } + operation.apply(event, now_ms)?; + } + JournalTransition::Allocate(spec) => { + spec.validate()?; + if spec.target.application() != self.scope.application + || spec.id.sequence != self.next_sequence + { + return Err(OperationError::Invalid( + "attempt sequence or application mismatch", + )); + } + if now_ms >= spec.deadline_ms { + return Err(OperationError::Deadline); + } + if let Some(operation) = &self.maintenance { + if spec.destination_node == operation.node { + return Err(OperationError::Invalid( + "receiver has durable maintenance intent", + )); + } + if spec.source_node == operation.node { + if spec.id.operation != operation.id + || spec.source != operation.session + || operation.phase != MaintenancePhase::Evacuating + { + return Err(OperationError::Invalid( + "maintenance source identity or phase mismatch", + )); + } + if now_ms >= operation.deadline_ms { + return Err(OperationError::Deadline); + } + } + } + if self + .attempts + .iter() + .any(|a| a.spec.target.cell_id() == spec.target.cell_id()) + { + return Err(OperationError::Busy); + } + let bytes = self + .reserved_restore_bytes() + .checked_add(spec.cost.disk_bytes) + .ok_or(OperationError::Budget)?; + if self.attempts.len() >= profile.max_inflight || bytes > profile.max_restore_bytes + { + return Err(OperationError::Budget); + } + next.next_sequence = self + .next_sequence + .checked_add(1) + .ok_or(OperationError::Invalid("attempt sequence overflow"))?; + next.attempts.push(MoveAttempt::new(spec)?); + } + JournalTransition::Attempt { id, event } => { + if matches!(event, AttemptEvent::BeginMaintenanceRelease) { + let operation = self.maintenance.as_ref().ok_or(OperationError::Invalid( + "busy release lacks maintenance intent", + ))?; + let attempt = self + .attempts + .iter() + .find(|attempt| attempt.spec.id == id) + .ok_or(OperationError::NotFound)?; + if attempt.spec.id.operation != operation.id + || attempt.spec.source_node != operation.node + || attempt.spec.source != operation.session + || operation.phase != MaintenancePhase::Evacuating + { + return Err(OperationError::Invalid( + "busy release maintenance identity mismatch", + )); + } + if now_ms >= operation.deadline_ms { + return Err(OperationError::Deadline); + } + } + let attempt = next + .attempts + .iter_mut() + .find(|a| a.spec.id == id) + .ok_or(OperationError::NotFound)?; + attempt.apply(event, now_ms)?; + } + JournalTransition::ResolveUnaccepted { id, effect } => { + let attempt = next + .attempts + .iter_mut() + .find(|a| a.spec.id == id) + .ok_or(OperationError::NotFound)?; + attempt.resolve_unaccepted(effect, now_ms)?; + } + JournalTransition::Retire { progress } => { + progress.validate()?; + if progress.scope != self.scope + || progress.sequence != self.next_progress_sequence()? + || progress.previous != self.progress.map(|head| head.digest) + || progress.entries.len() > MAX_ACTIVE_ATTEMPTS + { + return Err(OperationError::Invalid( + "progress chain does not extend the current head", + )); + } + for entry in &progress.entries { + let current = self + .attempts + .iter() + .find(|attempt| attempt.spec.id == entry.spec.id) + .ok_or(OperationError::NotFound)?; + if current != entry { + return Err(OperationError::Invalid( + "historical attempt differs from active permit", + )); + } + } + next.progress = Some(ProgressHead { + digest: progress.digest()?, + sequence: progress.sequence, + }); + next.attempts.retain(|attempt| { + !progress + .entries + .iter() + .any(|entry| entry.spec.id == attempt.spec.id) + }); + } + } + if next == *self && !fence_unaccepted { + return Ok(next); + } + next.finish_transition(now_ms)?; + Ok(next) + } + + fn check_revision_and_time(&self, expected_revision: u64, now_ms: i64) -> Result<()> { + if expected_revision != self.revision { + return Err(OperationError::Conflict); + } + if now_ms < self.last_observed_ms { + return Err(OperationError::Invalid( + "journal observation time regressed", + )); + } + Ok(()) + } + + fn finish_transition(&mut self, now_ms: i64) -> Result<()> { + self.revision = self + .revision + .checked_add(1) + .ok_or(OperationError::Invalid("journal revision overflow"))?; + self.last_observed_ms = now_ms; + self.validate() + } + + pub(super) fn validate(&self) -> Result<()> { + self.scope.validate()?; + if self.last_observed_ms < 0 + || self.next_sequence == 0 + || self.attempts.len() > MAX_ACTIVE_ATTEMPTS + { + return Err(OperationError::Invalid("invalid fleet head")); + } + if let Some(lease) = self.controller { + if !nonzero(lease.claimant.as_bytes()) + || lease.epoch == 0 + || lease.expires_at_ms <= self.last_observed_ms + || self.revision == 0 + { + return Err(OperationError::Invalid("invalid stored controller lease")); + } + } else if self.revision != 0 + || !self.attempts.is_empty() + || self.maintenance.is_some() + || self.next_sequence != 1 + || self.progress.is_some() + { + return Err(OperationError::Invalid( + "journal state lacks controller epoch", + )); + } + if let Some(progress) = self.progress { + progress.validate()?; + if progress.sequence >= self.next_sequence || progress.sequence > self.revision { + return Err(OperationError::Invalid( + "progress chain is ahead of journal allocations", + )); + } + } + if let Some(operation) = &self.maintenance { + operation.validate()?; + if operation.created_at_ms > self.last_observed_ms { + return Err(OperationError::Invalid( + "maintenance record is ahead of its head", + )); + } + } + let mut bytes = 0_u64; + let mut previous_sequence = 0; + for (i, attempt) in self.attempts.iter().enumerate() { + attempt.validate()?; + if attempt + .recovered() + .is_some_and(|e| e.recovery.basis().scope != self.scope) + { + return Err(OperationError::Invalid("recovery belongs to another fleet")); + } + if attempt + .completed_at_ms + .is_some_and(|at| at > self.last_observed_ms) + { + return Err(OperationError::Invalid( + "attempt result is ahead of its head", + )); + } + if attempt.spec.target.application() != self.scope.application + || attempt.spec.id.sequence >= self.next_sequence + || attempt.spec.id.sequence <= previous_sequence + || self.attempts[..i] + .iter() + .any(|a| a.spec.target.cell_id() == attempt.spec.target.cell_id()) + { + return Err(OperationError::Invalid( + "duplicate or unordered active attempt", + )); + } + previous_sequence = attempt.spec.id.sequence; + bytes = bytes + .checked_add(attempt.spec.cost.disk_bytes) + .ok_or(OperationError::Budget)?; + } + if bytes > MAX_RESTORE_BYTES { + return Err(OperationError::Budget); + } + Ok(()) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/mod.rs b/crates/cellule-runtime/src/fleet/operations/mod.rs new file mode 100644 index 00000000..9a048766 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/mod.rs @@ -0,0 +1,119 @@ +//! Deterministic fleet operation records and transitions. +//! +//! A journal adapter publishes transitions by compare-and-swap. These records +//! allocate advisory work and account its resources; only the Cell actor and +//! existing authority path can acquire, release, or serve a Cell. All times +//! and observations are supplied by adapters, so replay needs no I/O or clock. + +mod accepted; +mod acquisition; +mod actions; +mod attempt; +mod codec; +mod enrollment; +mod follower_evacuation; +pub use follower_evacuation::{ + FollowerEvacuationRecord, FollowerReplacementPolicy, FollowerReplacementWitness, +}; +mod history; +mod inspection; +mod intent; +mod journal; +mod reader_evacuation; +mod records; +mod recovery; +pub use reader_evacuation::{ + ReaderEvacuationPage, ReaderEvacuationRecord, ReaderReplacementWitness, +}; +mod registry; +mod writer_inventory; +pub use recovery::{RecoveredActivation, RecoveryBasis, RecoveryEvidence}; +pub use writer_inventory::{ + MAX_ORIGINAL_CATALOGS, MAX_ORIGINAL_WRITERS, OriginalCatalogWitness, + OriginalWriterInventoryBasis, OriginalWriterInventoryPage, OriginalWriterInventoryRecord, + OriginalWriterObservation, +}; + +pub use accepted::AcceptedFleetAction; +pub use acquisition::AcquisitionBasis; +pub use actions::{ + FleetAction, FleetActionKind, FleetActionOutcome, FleetOutcome, MaintenanceAction, +}; +pub use attempt::{ + ActivationEvidence, AttemptEvent, AttemptPhase, MoveAttempt, MoveAttemptSpec, MovementAction, + PublishedPosition, ReceiverReservation, TransferCost, +}; +pub use enrollment::{ + EnrollmentEndpoint, EnrollmentEvent, EnrollmentRecord, EnrollmentRole, EnrollmentSpec, + EnrollmentStatus, +}; +pub use history::{ProgressHead, ProgressPage}; +pub use inspection::{FleetInspectionObservation, FleetInspectionRequest}; +pub use intent::NodeIntent; +pub use journal::{ControllerLease, FleetHead, JournalTransition}; +pub use records::{ + AttemptId, DrainBlocker, DrainEvidence, FleetProfile, FleetScope, MaintenanceEvent, + MaintenanceOperation, MaintenancePhase, OperationId, +}; +pub use registry::{EnrollmentPage, IntentPage, RegistryVersion}; + +/// Fleet journal and action format version. +pub const FORMAT_VERSION: u8 = 1; +/// Maximum encoded journal head or action envelope. +pub const MAX_RECORD_BYTES: u32 = 64 * 1024; +/// Maximum encoded observation or historical progress page. +pub const MAX_PAGE_BYTES: u32 = 1024 * 1024; +/// Maximum entries in an observation or historical progress page. +pub const MAX_PAGE_ENTRIES: usize = 128; +/// Hard initial bound on unresolved planned movement attempts. +pub const MAX_ACTIVE_ATTEMPTS: usize = 2; +/// Hard initial bound on disk demand reserved by unresolved attempts. +pub const MAX_RESTORE_BYTES: u64 = 8 * 1024 * 1024 * 1024; + +/// Failure of a pure fleet operation transition or its bounded codec. +#[derive(Debug, thiserror::Error)] +pub enum OperationError { + /// A supplied record, observation, or event violates the contract. + #[error("invalid fleet operation: {0}")] + Invalid(&'static str), + /// The expected journal revision is no longer current. + #[error("fleet journal revision changed")] + Conflict, + /// The controller lease is expired or belongs to a different epoch. + #[error("fleet controller is fenced")] + Fenced, + /// The operation cannot admit new work beyond its deadline. + #[error("fleet operation deadline exceeded")] + Deadline, + /// Unresolved attempts consume the fleet count or byte budget. + #[error("fleet movement budget is exhausted")] + Budget, + /// Operator policy disables allocation of new planned attempts. + #[error("fleet movement scheduling is stopped")] + Stopped, + /// The exact attempt is not present in the current head. + #[error("fleet movement attempt is absent")] + NotFound, + /// A different maintenance request is still active. + #[error("fleet maintenance operation is already active")] + Busy, + /// Bounded wire encoding failed, preserving the original codec error. + #[error("fleet operation encoding failed")] + Codec(#[from] crate::codec::CodecError), + /// Cell target validation failed, preserving the original identity error. + #[error("fleet Cell identity is invalid")] + Identity(#[source] Box), + /// Canonical Cell control validation failed with its original source. + #[error("fleet acquisition control is invalid")] + Control(#[source] Box), +} + +/// Result of a pure fleet operation transition. +pub type Result = std::result::Result; + +fn nonzero(bytes: &[u8]) -> bool { + bytes.iter().any(|byte| *byte != 0) +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-runtime/src/fleet/operations/reader_evacuation/mod.rs b/crates/cellule-runtime/src/fleet/operations/reader_evacuation/mod.rs new file mode 100644 index 00000000..d90b1177 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/reader_evacuation/mod.rs @@ -0,0 +1,251 @@ +//! Durable historical reader replacement evidence, never current authority. +use super::*; +use crate::{ + control::{Control, ControlState}, + identity::{Digest, NodeId, SessionId}, + read_policy::MAX_READERS, +}; +use std::collections::HashSet; + +mod validation; +pub(super) const MAX_READER_PAGES: usize = (MAX_READERS as usize).div_ceil(MAX_PAGE_ENTRIES); + +/// Original enrolled reader boot and observed prefix outside the donor. +/// Digests bind authenticated adapter observations; decoding does not prove them. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct ReaderReplacementWitness { + /// Physical replacement node, distinct from every other replacement. + pub node: NodeId, + /// Exact original receiver boot. + pub session: SessionId, + /// Immutable signed boot identity, excluding renewable sample fields. + pub boot_identity: Digest, + /// Exact original reader enrollment request key. + pub enrollment_key: Digest, + /// Digest of that complete Established row, including original history. + pub enrollment_digest: Digest, + /// Ready native prefix in the manifest's Cell incarnation. + pub commit_sequence: u64, +} + +/// Immutable replacement page with an exact manifest basis and ordinal. +/// At most 128 entries; the complete manifest validates all pages together. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct ReaderEvacuationPage { + pub(super) basis: Digest, + pub(super) ordinal: u32, + pub(super) entries: Vec, +} +impl ReaderEvacuationPage { + /// Exact immutable capture basis shared by every page. + #[must_use] + pub const fn basis(&self) -> Digest { + self.basis + } + /// Zero-based ordinal; skipped, repeated and reordered pages refuse. + #[must_use] + pub const fn ordinal(&self) -> u32 { + self.ordinal + } + /// Original replacement identities and observed prefixes. + #[must_use] + pub fn entries(&self) -> &[ReaderReplacementWitness] { + &self.entries + } + /// Canonical content identity checked against the manifest. + pub fn digest(&self) -> Result { + Ok(Digest::from_bytes( + *blake3::hash(&self.to_bytes()?).as_bytes(), + )) + } +} + +/// Bounded immutable reader evacuation manifest plus paged replacement evidence. +/// Historical capture is persisted once; fresh readiness, authority, policy, +/// roster and operation checks are required before consuming it as settlement. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct ReaderEvacuationRecord { + pub(super) operation: MaintenanceOperation, + pub(super) head_digest: Digest, + pub(super) registry: RegistryVersion, + pub(super) retired: EnrollmentRecord, + pub(super) original_digest: Digest, + pub(super) authority: Control, + pub(super) policy_revision: Option, + pub(super) desired_readers: u16, + pub(super) minimum_sequence: u64, + pub(super) started_at_ms: i64, + pub(super) finished_at_ms: i64, + pub(super) pages: Vec, +} +impl ReaderEvacuationRecord { + /// Validates shape and builds canonical pages for the entire policy bound. + /// This constructor supplies no native closure or adapter authentication. + #[allow(clippy::too_many_arguments)] + pub fn new( + operation: MaintenanceOperation, + barrier: (Digest, RegistryVersion), + retired: EnrollmentRecord, + original_digest: Digest, + authority: Control, + policy_revision: Option, + desired_readers: u16, + minimum_sequence: u64, + interval: (i64, i64), + mut replacements: Vec, + ) -> Result<(Self, Vec)> { + let mut record = Self { + operation, + head_digest: barrier.0, + registry: barrier.1, + retired, + original_digest, + authority, + policy_revision, + desired_readers, + minimum_sequence, + started_at_ms: interval.0, + finished_at_ms: interval.1, + pages: Vec::new(), + }; + record.validate_basis()?; + if replacements.len() != usize::from(desired_readers) { + return Err(OperationError::Invalid("reader replacement count differs")); + } + replacements.sort_by_key(|replacement| *replacement.node.as_bytes()); + let basis = record.basis_digest()?; + let pages: Vec<_> = replacements + .chunks(MAX_PAGE_ENTRIES) + .enumerate() + .map(|(ordinal, entries)| ReaderEvacuationPage { + basis, + ordinal: ordinal as u32, + entries: entries.to_vec(), + }) + .collect(); + record.pages = pages + .iter() + .map(ReaderEvacuationPage::digest) + .collect::>()?; + record.validate_pages(&pages)?; + Ok((record, pages)) + } + /// Exact original operation; deadline/session extensions require fresh capture. + #[must_use] + pub fn operation(&self) -> &MaintenanceOperation { + &self.operation + } + /// Digest of the full original head checked before capture publication. + #[must_use] + pub const fn head_digest(&self) -> Digest { + self.head_digest + } + /// Exact original registry version, retained independently of fresh rechecks. + #[must_use] + pub const fn registry(&self) -> RegistryVersion { + self.registry + } + /// Original Retired request retaining acceptance and establishment history. + #[must_use] + pub fn retired(&self) -> &EnrollmentRecord { + &self.retired + } + /// Digest of the original Established request supplied to native closure. + #[must_use] + pub const fn original_digest(&self) -> Digest { + self.original_digest + } + /// Original serving authority capture, without ownership permission. + #[must_use] + pub fn authority(&self) -> &Control { + &self.authority + } + /// Exact policy revision, or absence meaning zero desired readers. + #[must_use] + pub const fn policy_revision(&self) -> Option { + self.policy_revision + } + /// Original required replacement count. + #[must_use] + pub const fn desired_readers(&self) -> u16 { + self.desired_readers + } + /// Prefix of the exact retired native view that every replacement must cover. + #[must_use] + pub const fn minimum_sequence(&self) -> u64 { + self.minimum_sequence + } + /// Original capture interval; replay cannot refresh it. + #[must_use] + pub const fn interval(&self) -> (i64, i64) { + (self.started_at_ms, self.finished_at_ms) + } + /// Complete ordered page identities; empty only under zero-reader policy. + #[must_use] + pub fn pages(&self) -> &[Digest] { + &self.pages + } + /// Canonical immutable capture identity, including every page digest. + pub fn digest(&self) -> Result { + Ok(Digest::from_bytes( + *blake3::hash(&self.to_bytes()?).as_bytes(), + )) + } + + /// Requires every exact page and unique physical boot, never a partial list. + pub fn validate_pages(&self, pages: &[ReaderEvacuationPage]) -> Result<()> { + self.validate()?; + if pages.len() != self.pages.len() { + return Err(OperationError::Invalid( + "reader replacement pages incomplete", + )); + } + let basis = self.basis_digest()?; + let mut nodes = HashSet::new(); + let mut sessions = HashSet::new(); + let mut requests = HashSet::new(); + let mut after = None; + let mut count = 0; + for (ordinal, (page, digest)) in pages.iter().zip(&self.pages).enumerate() { + page.validate()?; + let remaining = usize::from(self.desired_readers) - count; + if page.basis != basis + || page.ordinal as usize != ordinal + || page.digest()? != *digest + || page.entries.len() != remaining.min(MAX_PAGE_ENTRIES) + { + return Err(OperationError::Conflict); + } + for entry in &page.entries { + let node = *entry.node.as_bytes(); + if after.is_some_and(|previous| node <= previous) + || !nodes.insert(entry.node) + || !sessions.insert(entry.session) + || !requests.insert(entry.enrollment_key) + || entry.node == self.operation.node() + || entry.session == self.operation.session() + || entry.session == self.retired.spec().target.session + || self + .authority + .owner + .as_ref() + .is_some_and(|owner| owner.session == entry.session) + || entry.commit_sequence < self.minimum_sequence + { + return Err(OperationError::Invalid( + "reader replacement identity or prefix differs", + )); + } + after = Some(node); + } + count += page.entries.len(); + } + if count != usize::from(self.desired_readers) { + return Err(OperationError::Conflict); + } + Ok(()) + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-runtime/src/fleet/operations/reader_evacuation/tests.rs b/crates/cellule-runtime/src/fleet/operations/reader_evacuation/tests.rs new file mode 100644 index 00000000..6b46fb02 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/reader_evacuation/tests.rs @@ -0,0 +1,320 @@ +use super::*; +use crate::{ + control::{Owner, RootRef}, + identity::{ApplicationId, CellTarget, IncarnationId, NamespaceId, TenantId}, +}; + +fn record(desired: u16) -> (ReaderEvacuationRecord, Vec) { + let scope = FleetScope { + fleet: Digest::from_bytes([100; 32]), + application: ApplicationId::from_bytes([3; 16]), + }; + let source = NodeIntent::initial( + scope, + NodeId::from_bytes([1; 16]), + SessionId::from_bytes([11; 16]), + ) + .unwrap(); + let donor = NodeIntent::initial( + scope, + NodeId::from_bytes([2; 16]), + SessionId::from_bytes([12; 16]), + ) + .unwrap(); + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + scope.application, + NamespaceId::from_bytes([9; 16]), + &[1], + ) + .unwrap(); + let incarnation = IncarnationId::from_bytes([6; 16]); + let root = RootRef { + digest: Digest::from_bytes([9; 32]), + txid: 1, + checksum: cellule_ltx::types::CHECKSUM_FLAG | 1, + commit_sequence: 1, + }; + let original = EnrollmentRecord::pending( + EnrollmentSpec { + scope, + request: Digest::from_bytes([70; 32]), + source: Some(EnrollmentEndpoint { + node: source.node(), + session: source.session(), + intent_revision: 1, + }), + target: EnrollmentEndpoint { + node: donor.node(), + session: donor.session(), + intent_revision: 1, + }, + role: EnrollmentRole::Reader { + target: target.clone(), + position: PublishedPosition { + incarnation, + epoch: 1, + root: root.clone(), + }, + }, + }, + Some(&source), + &donor, + 10, + ) + .unwrap() + .establish(Digest::from_bytes([71; 32]), 20) + .unwrap(); + let retired = original.retire(Digest::from_bytes([72; 32]), 30).unwrap(); + let mut authority = Control::initial( + target.cell_id(), + incarnation, + Owner { + session: source.session(), + endpoint: "https://source.example".into(), + }, + Digest::from_bytes([7; 32]), + 1, + ) + .unwrap(); + authority.state = ControlState::Serving; + authority.root = Some(root); + let mut operation = MaintenanceOperation::new( + OperationId::from_bytes([80; 16]).unwrap(), + Digest::from_bytes([81; 32]), + donor.node(), + donor.session(), + 2, + 0, + 60_000, + ) + .unwrap(); + operation.apply(MaintenanceEvent::Cordoned, 0).unwrap(); + operation + .apply(MaintenanceEvent::BeginEvacuation, 0) + .unwrap(); + let replacements = (0..u32::from(desired)) + .map(|number| { + let mut node = [0; 16]; + node[..4].copy_from_slice(&(number + 100).to_be_bytes()); + let mut session = node; + session[15] = 1; + ReaderReplacementWitness { + node: NodeId::from_bytes(node), + session: SessionId::from_bytes(session), + boot_identity: Digest::from_bytes([90; 32]), + enrollment_key: Digest::from_bytes(*blake3::hash(&node).as_bytes()), + enrollment_digest: Digest::from_bytes([92; 32]), + commit_sequence: 1, + } + }) + .collect(); + ReaderEvacuationRecord::new( + operation, + ( + Digest::from_bytes([89; 32]), + RegistryVersion::new(scope).unwrap().bootstrap(0).unwrap(), + ), + retired, + Digest::from_bytes(*blake3::hash(&original.to_bytes().unwrap()).as_bytes()), + authority, + Some(1), + desired, + 1, + (25, 35), + replacements, + ) + .unwrap() +} + +#[test] +fn reader_policy_manifest_roundtrips_the_complete_ten_thousand_reader_bound() { + let (record, pages) = record(MAX_READERS); + assert_eq!(pages.len(), 79); + assert_eq!(pages.last().unwrap().entries().len(), 16); + assert_eq!( + ReaderEvacuationRecord::from_bytes(&record.to_bytes().unwrap()).unwrap(), + record + ); + for page in &pages { + let bytes = page.to_bytes().unwrap(); + assert!(bytes.len() <= MAX_RECORD_BYTES as usize); + assert_eq!(ReaderEvacuationPage::from_bytes(&bytes).unwrap(), *page); + } + record.validate_pages(&pages).unwrap(); + assert_eq!(record.retired().status(), EnrollmentStatus::Retired); + assert_eq!(record.retired().accepted_at_ms(), 10); + assert_eq!( + record.retired().established_evidence(), + Some(Digest::from_bytes([71; 32])) + ); +} + +#[test] +fn reader_policy_manifest_requires_every_exact_page_and_original_capture_basis() { + let (record, pages) = record(129); + assert!(record.validate_pages(&pages[..1]).is_err()); + let mut changed = pages.clone(); + changed.swap(0, 1); + assert!(record.validate_pages(&changed).is_err()); + let mut changed = pages.clone(); + changed[1].basis = Digest::from_bytes([93; 32]); + assert!(record.validate_pages(&changed).is_err()); + let mut changed = pages; + changed[1].ordinal = 2; + assert!(record.validate_pages(&changed).is_err()); +} + +#[test] +fn reader_policy_manifest_rejects_duplicate_physical_nodes_sessions_and_regressed_prefixes() { + let (record, pages) = record(129); + for kind in 0..5 { + let mut changed = pages.clone(); + let mut record = record.clone(); + match kind { + 0 => changed[1].entries[0].node = changed[0].entries[0].node, + 1 => changed[1].entries[0].session = changed[0].entries[0].session, + 2 => changed[1].entries[0].node = record.operation.node(), + _ => changed[1].entries[0].commit_sequence = 0, + } + record.pages[1] = changed[1].digest().unwrap(); + assert!(record.validate_pages(&changed).is_err()); + } +} + +#[test] +fn reader_policy_manifest_preserves_absent_and_explicit_zero_policy_separately() { + let (mut record, pages) = record(0); + assert!(pages.is_empty()); + let explicit = record.digest().unwrap(); + record.policy_revision = None; + let absent = record.digest().unwrap(); + assert_ne!(absent, explicit); + record.validate_pages(&[]).unwrap(); + record.desired_readers = 1; + assert!(record.to_bytes().is_err()); +} + +#[test] +fn reader_policy_manifest_rejects_scope_lifetime_history_and_interval_mismatches() { + let (record, _) = record(1); + for kind in 0..12 { + let mut changed = record.clone(); + match kind { + 0 => changed.operation.phase = MaintenancePhase::Completed, + 1 => changed.authority.state = ControlState::Idle, + 2 => changed.authority.incarnation = IncarnationId::from_bytes([99; 16]), + 3 => changed.policy_revision = Some(0), + 4 => changed.finished_at_ms = changed.started_at_ms + 30_001, + 5 => changed.finished_at_ms = changed.started_at_ms - 1, + 6 => changed.operation.node = NodeId::from_bytes([99; 16]), + 7 => changed.original_digest = Digest::from_bytes([0; 32]), + 8 => changed.minimum_sequence = 2, + 9 => changed.head_digest = Digest::from_bytes([0; 32]), + 10 => changed.registry = RegistryVersion::new(changed.retired.spec().scope).unwrap(), + _ => { + let mut scope = changed.retired.spec().scope; + scope.fleet = Digest::from_bytes([99; 32]); + changed.registry = RegistryVersion::new(scope).unwrap().bootstrap(0).unwrap(); + } + } + assert!(changed.to_bytes().is_err()); + } +} + +#[test] +fn reader_policy_codecs_reject_truncation_trailing_wrong_kind_and_oversized_data() { + let (record, pages) = record(1); + let manifest = record.to_bytes().unwrap(); + let page = pages[0].to_bytes().unwrap(); + for end in 0..manifest.len() { + assert!(ReaderEvacuationRecord::from_bytes(&manifest[..end]).is_err()); + } + for end in 0..page.len() { + assert!(ReaderEvacuationPage::from_bytes(&page[..end]).is_err()); + } + let mut extra = manifest.clone(); + extra.push(0); + assert!(ReaderEvacuationRecord::from_bytes(&extra).is_err()); + let mut extra = page.clone(); + extra.push(0); + assert!(ReaderEvacuationPage::from_bytes(&extra).is_err()); + assert!(ReaderEvacuationRecord::from_bytes(&page).is_err()); + assert!(ReaderEvacuationPage::from_bytes(&manifest).is_err()); + assert!(ReaderEvacuationRecord::from_bytes(&vec![0; MAX_PAGE_BYTES as usize + 1]).is_err()); + assert!(ReaderEvacuationPage::from_bytes(&vec![0; MAX_RECORD_BYTES as usize + 1]).is_err()); +} + +#[test] +fn reader_policy_manifest_retains_original_retirement_after_operation_boot_adoption_and_closing() { + let (original, pages) = record(1); + let mut operation = original.operation().clone(); + operation + .apply( + MaintenanceEvent::SessionReplaced(SessionId::from_bytes([99; 16])), + 40, + ) + .unwrap(); + operation.apply(MaintenanceEvent::Cordoned, 40).unwrap(); + operation + .apply(MaintenanceEvent::BeginEvacuation, 40) + .unwrap(); + let replacements = pages + .iter() + .flat_map(|page| page.entries().iter().cloned()) + .collect::>(); + let (adopted, _) = ReaderEvacuationRecord::new( + operation.clone(), + (original.head_digest(), original.registry()), + original.retired().clone(), + original.original_digest(), + original.authority().clone(), + original.policy_revision(), + 1, + 1, + (40, 50), + replacements.clone(), + ) + .unwrap(); + assert_eq!(adopted.retired(), original.retired()); + assert_ne!( + adopted.operation().session(), + adopted.retired().spec().target.session + ); + operation + .apply( + MaintenanceEvent::ReadyToClose(DrainEvidence { + node: operation.node(), + session: operation.session(), + remaining_cells: 0, + unresolved_attempts: 0, + relocated: true, + readers_settled: true, + followers_settled: true, + facilities_closed: false, + stopped: false, + withdrawn: false, + }), + 51, + ) + .unwrap(); + let (closing, _) = ReaderEvacuationRecord::new( + operation, + (original.head_digest(), original.registry()), + original.retired().clone(), + original.original_digest(), + original.authority().clone(), + original.policy_revision(), + 1, + 1, + (55, 60), + replacements, + ) + .unwrap(); + assert_eq!(closing.operation().phase(), MaintenancePhase::Closing); + assert_eq!(closing.retired(), original.retired()); + assert_eq!( + ReaderEvacuationRecord::from_bytes(&closing.to_bytes().unwrap()).unwrap(), + closing + ); +} diff --git a/crates/cellule-runtime/src/fleet/operations/reader_evacuation/validation.rs b/crates/cellule-runtime/src/fleet/operations/reader_evacuation/validation.rs new file mode 100644 index 00000000..d56b326a --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/reader_evacuation/validation.rs @@ -0,0 +1,88 @@ +use super::*; +impl ReaderEvacuationRecord { + pub(in crate::fleet::operations) fn validate_basis(&self) -> Result<()> { + self.operation.validate()?; + self.registry.confirm(self.registry)?; + self.retired.validate_replay(self.retired.spec())?; + self.authority + .encode() + .map_err(|error| OperationError::Control(Box::new(error)))?; + let EnrollmentRole::Reader { target, position } = &self.retired.spec().role else { + return Err(OperationError::Invalid("reader evacuation role differs")); + }; + if !matches!( + self.operation.phase(), + MaintenancePhase::Evacuating | MaintenancePhase::Closing + ) || self.retired.status() != EnrollmentStatus::Retired + || self.retired.established_evidence().is_none() + || self.retired.spec().target.node != self.operation.node() + || self.registry.scope() != self.retired.spec().scope + || !nonzero(self.head_digest.as_bytes()) + || !nonzero(self.original_digest.as_bytes()) + || self.started_at_ms < 0 + || self.finished_at_ms < self.started_at_ms + || self.finished_at_ms - self.started_at_ms > 30_000 + || self.started_at_ms < self.operation.created_at_ms + || self.finished_at_ms >= self.operation.deadline_ms() + || self.retired.updated_at_ms() > self.finished_at_ms + || self.authority.cell != target.cell_id() + || self.authority.incarnation != position.incarnation + || self.authority.state != ControlState::Serving + || self.authority.recovery.is_some() + || self.authority.owner.is_none() + || self + .authority + .root + .as_ref() + .is_none_or(|root| root.commit_sequence < self.minimum_sequence) + || self.minimum_sequence < position.root.commit_sequence + || self.minimum_sequence > i64::MAX as u64 + || self.desired_readers > MAX_READERS + || self.policy_revision == Some(0) + || (self.policy_revision.is_none() && self.desired_readers != 0) + { + return Err(OperationError::Invalid("invalid reader evacuation capture")); + } + Ok(()) + } + pub(in crate::fleet::operations) fn validate(&self) -> Result<()> { + self.validate_basis()?; + let expected = usize::from(self.desired_readers).div_ceil(MAX_PAGE_ENTRIES); + if self.pages.len() != expected + || self.pages.len() > MAX_READER_PAGES + || self.pages.iter().any(|digest| !nonzero(digest.as_bytes())) + { + return Err(OperationError::Invalid("invalid reader evacuation pages")); + } + Ok(()) + } +} +impl ReaderEvacuationPage { + pub(in crate::fleet::operations) fn validate(&self) -> Result<()> { + if !nonzero(self.basis.as_bytes()) + || self.ordinal as usize >= MAX_READER_PAGES + || self.entries.is_empty() + || self.entries.len() > MAX_PAGE_ENTRIES + { + return Err(OperationError::Invalid("invalid reader evacuation page")); + } + let mut after = None; + for entry in &self.entries { + let node = *entry.node.as_bytes(); + if !nonzero(&node) + || !nonzero(entry.session.as_bytes()) + || !nonzero(entry.boot_identity.as_bytes()) + || !nonzero(entry.enrollment_key.as_bytes()) + || !nonzero(entry.enrollment_digest.as_bytes()) + || entry.commit_sequence > i64::MAX as u64 + || after.is_some_and(|previous| node <= previous) + { + return Err(OperationError::Invalid( + "invalid reader replacement witness", + )); + } + after = Some(node); + } + Ok(()) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/records.rs b/crates/cellule-runtime/src/fleet/operations/records.rs new file mode 100644 index 00000000..1968fa3a --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/records.rs @@ -0,0 +1,402 @@ +use crate::identity::{ApplicationId, Digest, NodeId, SessionId}; + +use super::{MAX_ACTIVE_ATTEMPTS, MAX_RESTORE_BYTES, OperationError, Result, nonzero}; + +/// Nonzero application-assigned operation identity. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct OperationId(pub(super) [u8; 16]); + +impl OperationId { + /// Validates a stable identity supplied by the application. + pub fn from_bytes(bytes: [u8; 16]) -> Result { + if !nonzero(&bytes) { + return Err(OperationError::Invalid("zero operation identity")); + } + Ok(Self(bytes)) + } + + /// Returns the canonical identity bytes. + #[must_use] + pub const fn as_bytes(&self) -> &[u8; 16] { + &self.0 + } +} + +/// One never-reused journal allocation within an operation. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct AttemptId { + /// Operation that requested the movement. + pub operation: OperationId, + /// Monotonic sequence allocated by the fleet head. + pub sequence: u64, +} + +impl AttemptId { + pub(super) fn validate(self) -> Result<()> { + if self.sequence == 0 || !nonzero(self.operation.as_bytes()) { + return Err(OperationError::Invalid("invalid attempt identity")); + } + Ok(()) + } +} + +/// Scope of one independent controller, journal, and movement budget. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct FleetScope { + /// Authenticated deployment fleet identity. + pub fleet: Digest, + /// Application whose Cells and sessions may be managed. + pub application: ApplicationId, +} + +impl FleetScope { + pub(super) fn validate(self) -> Result<()> { + if !nonzero(self.fleet.as_bytes()) || !nonzero(self.application.as_bytes()) { + return Err(OperationError::Invalid("zero fleet operation scope")); + } + Ok(()) + } +} + +/// Validated starting bounds for caller-driven fleet reconciliation. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct FleetProfile { + /// Maximum unresolved planned attempts across all donors. + pub max_inflight: usize, + /// Maximum disk demand charged by unresolved planned attempts. + pub max_restore_bytes: u64, + /// Controller lease duration in logical milliseconds. + pub controller_lease_ms: i64, + /// Periodic observation interval; results may wake the caller earlier. + pub reconcile_interval_ms: i64, +} + +impl Default for FleetProfile { + fn default() -> Self { + Self { + max_inflight: MAX_ACTIVE_ATTEMPTS, + max_restore_bytes: MAX_RESTORE_BYTES, + controller_lease_ms: 30_000, + reconcile_interval_ms: 15_000, + } + } +} + +impl FleetProfile { + /// Rejects unbounded profiles before a driver or journal starts work. + pub fn validate(self) -> Result { + if self.max_inflight == 0 + || self.max_inflight > MAX_ACTIVE_ATTEMPTS + || self.max_restore_bytes == 0 + || self.max_restore_bytes > MAX_RESTORE_BYTES + || self.controller_lease_ms <= 0 + || self.controller_lease_ms > 30_000 + || self.reconcile_interval_ms <= 0 + || self.reconcile_interval_ms >= self.controller_lease_ms + { + return Err(OperationError::Invalid("invalid fleet operation profile")); + } + Ok(self) + } +} + +/// Bounded operational classification; detailed source errors remain in adapters. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[repr(u8)] +pub enum DrainBlocker { + /// Required membership, inventory, or readiness evidence is incomplete. + IncompleteObservation = 1, + /// A sample is stale or its time/sequence regressed. + StaleObservation = 2, + /// Release, schema, or operational codec versions disagree. + IncompatibleRelease = 3, + /// No receiver can admit the measured Cell cost. + ReceiverCapacity = 4, + /// Foreground work or a transition is still executing. + BusyExecution = 5, + /// An issued external-work lease has not settled. + ExternalLease = 6, + /// An acknowledged state still needs object coverage. + PendingPublication = 7, + /// Foreign follower tails or epochs still require this node. + FollowerObligation = 8, + /// Primitive or role inventory has not been proven. + UnknownInventory = 9, + /// Existing planned attempts consume the movement budget. + MovementBudget = 10, + /// An accepted remote action has no confirmed result yet. + OutcomeUnknown = 11, + /// The request's deadline stops further admission. + Deadline = 12, + /// An owned facility has not completed its drain. + FacilityFailure = 13, + /// Required reader redundancy has not been restored. + ReaderObligation = 14, +} + +/// Durable phase of an explicit physical-node maintenance request. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[repr(u8)] +pub enum MaintenancePhase { + /// Intent is persisted but local admission is not yet confirmed closed. + Requested = 1, + /// The current node session rejects new role acquisition. + Cordoned = 2, + /// Writers, readers, and follower obligations are being evacuated. + Evacuating = 3, + /// Relocation is proven and normal terminal host shutdown may run. + Closing = 4, + /// Host shutdown and exact session withdrawal are confirmed. + Completed = 5, +} + +/// Fresh node and relocation observations needed to close a maintenance operation. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct DrainEvidence { + /// Physical node whose durable intent is being reconciled. + pub node: NodeId, + /// Exact boot session observed by this evidence. + pub session: SessionId, + /// Source writers, including in-progress activation/release transitions. + pub remaining_cells: u64, + /// Planned attempts without terminal outcomes or reservation cleanup. + pub unresolved_attempts: u32, + /// Whether every affected Cell has verified serving evidence elsewhere. + pub relocated: bool, + /// Required reader replacements and local views are settled. + pub readers_settled: bool, + /// Complete foreign-tail inventory is proven no longer needed locally. + pub followers_settled: bool, + /// Required host facilities have joined and closed. + pub facilities_closed: bool, + /// The host has reached its successful terminal state. + pub stopped: bool, + /// The exact node session was authoritatively withdrawn. + pub withdrawn: bool, +} + +impl DrainEvidence { + pub(super) fn ready_to_close(self) -> bool { + self.remaining_cells == 0 + && self.unresolved_attempts == 0 + && self.relocated + && self.readers_settled + && self.followers_settled + } +} + +/// Replayable event of the maintenance lifecycle. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum MaintenanceEvent { + /// Local admission closure and advertised mode are confirmed. + Cordoned, + /// Begin bounded evacuation under the persisted cordon. + BeginEvacuation, + /// Prove all writer and role obligations before entering terminal shutdown. + ReadyToClose(DrainEvidence), + /// Prove the successful host shutdown and exact withdrawal. + Stopped(DrainEvidence), + /// Report a temporary blocker without losing the durable phase. + Blocked(DrainBlocker), + /// Adopt a rebooted session while retaining the physical-node cordon. + SessionReplaced(SessionId), + /// Extend a deadline through an authorized newer request revision. + ExtendDeadline(i64), +} + +/// Bounded durable maintenance intent; the operation never reopens acquisition. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct MaintenanceOperation { + pub(super) id: OperationId, + pub(super) request_digest: Digest, + pub(super) node: NodeId, + pub(super) session: SessionId, + pub(super) intent_revision: u64, + pub(super) created_at_ms: i64, + pub(super) deadline_ms: i64, + pub(super) phase: MaintenancePhase, + pub(super) blocker: Option, + pub(super) drain_evidence: Option, +} + +impl MaintenanceOperation { + /// Constructs the persisted request before any node side effect. + pub fn new( + id: OperationId, + request_digest: Digest, + node: NodeId, + session: SessionId, + intent_revision: u64, + created_at_ms: i64, + deadline_ms: i64, + ) -> Result { + let operation = Self { + id, + request_digest, + node, + session, + intent_revision, + created_at_ms, + deadline_ms, + phase: MaintenancePhase::Requested, + blocker: None, + drain_evidence: None, + }; + operation.validate()?; + Ok(operation) + } + + /// Returns the idempotent operation identity. + #[must_use] + pub const fn id(&self) -> OperationId { + self.id + } + /// Returns the physical node to keep cordoned across sessions. + #[must_use] + pub const fn node(&self) -> NodeId { + self.node + } + /// Returns the current targeted boot session. + #[must_use] + pub const fn session(&self) -> SessionId { + self.session + } + /// Returns the durable node-intent revision. + #[must_use] + pub const fn intent_revision(&self) -> u64 { + self.intent_revision + } + /// Returns the operation's admission deadline. + #[must_use] + pub const fn deadline_ms(&self) -> i64 { + self.deadline_ms + } + /// Returns the durable progress phase. + #[must_use] + pub const fn phase(&self) -> MaintenancePhase { + self.phase + } + /// Returns the last temporary blocker. + #[must_use] + pub const fn blocker(&self) -> Option { + self.blocker + } + + /// Returns the exact evidence retained by a closing or completed operation. + #[must_use] + pub const fn drain_evidence(&self) -> Option { + self.drain_evidence + } + + pub(super) fn validate(&self) -> Result<()> { + if !nonzero(self.id.as_bytes()) + || !nonzero(self.request_digest.as_bytes()) + || !nonzero(self.node.as_bytes()) + || !nonzero(self.session.as_bytes()) + || self.intent_revision == 0 + || self.created_at_ms < 0 + || self.deadline_ms <= self.created_at_ms + || (self.phase == MaintenancePhase::Completed && self.blocker.is_some()) + { + return Err(OperationError::Invalid("invalid maintenance record")); + } + if matches!( + self.phase, + MaintenancePhase::Closing | MaintenancePhase::Completed + ) != self.drain_evidence.is_some() + { + return Err(OperationError::Invalid( + "maintenance phase lacks drain evidence", + )); + } + if let Some(evidence) = self.drain_evidence + && (evidence.node != self.node + || evidence.session != self.session + || !evidence.ready_to_close() + || (self.phase == MaintenancePhase::Completed + && !(evidence.facilities_closed && evidence.stopped && evidence.withdrawn))) + { + return Err(OperationError::Invalid("invalid stored drain evidence")); + } + Ok(()) + } + + pub(super) fn apply(&mut self, event: MaintenanceEvent, now_ms: i64) -> Result<()> { + if now_ms < self.created_at_ms { + return Err(OperationError::Invalid("maintenance time regressed")); + } + let matches = |e: DrainEvidence| e.node == self.node && e.session == self.session; + match event { + MaintenanceEvent::Cordoned if self.phase == MaintenancePhase::Requested => { + self.phase = MaintenancePhase::Cordoned; + self.blocker = None; + } + MaintenanceEvent::Cordoned if self.phase != MaintenancePhase::Completed => {} + MaintenanceEvent::BeginEvacuation if self.phase == MaintenancePhase::Cordoned => { + self.phase = MaintenancePhase::Evacuating; + self.blocker = None; + } + MaintenanceEvent::BeginEvacuation if self.phase == MaintenancePhase::Evacuating => {} + MaintenanceEvent::ReadyToClose(e) + if matches(e) + && e.ready_to_close() + && matches!( + self.phase, + MaintenancePhase::Evacuating | MaintenancePhase::Closing + ) => + { + self.phase = MaintenancePhase::Closing; + self.drain_evidence = Some(e); + self.blocker = None; + } + MaintenanceEvent::Stopped(e) + if matches(e) + && e.ready_to_close() + && e.facilities_closed + && e.stopped + && e.withdrawn + && matches!( + self.phase, + MaintenancePhase::Closing | MaintenancePhase::Completed + ) => + { + self.phase = MaintenancePhase::Completed; + self.drain_evidence = Some(e); + self.blocker = None; + } + MaintenanceEvent::Blocked(blocker) if self.phase != MaintenancePhase::Completed => { + self.blocker = Some(blocker); + } + MaintenanceEvent::SessionReplaced(session) + if nonzero(session.as_bytes()) && self.phase != MaintenancePhase::Completed => + { + if session != self.session { + // A new boot invalidates every enrollment check made with + // the previous session, even when desired mode is unchanged. + self.intent_revision = self + .intent_revision + .checked_add(1) + .ok_or(OperationError::Invalid("maintenance revision overflow"))?; + self.session = session; + self.phase = MaintenancePhase::Requested; + self.drain_evidence = None; + self.blocker = Some(DrainBlocker::IncompleteObservation); + } + } + MaintenanceEvent::ExtendDeadline(deadline) + if deadline > self.deadline_ms + && deadline > now_ms + && self.phase != MaintenancePhase::Completed => + { + self.deadline_ms = deadline; + self.intent_revision = self + .intent_revision + .checked_add(1) + .ok_or(OperationError::Invalid("maintenance revision overflow"))?; + self.blocker = None; + } + _ => return Err(OperationError::Invalid("unproven maintenance transition")), + } + self.validate() + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/recovery.rs b/crates/cellule-runtime/src/fleet/operations/recovery.rs new file mode 100644 index 00000000..93a445b3 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/recovery.rs @@ -0,0 +1,271 @@ +use crate::control::{Control, ControlState, Transition}; +use crate::identity::{Digest, NodeId, SessionId}; +use crate::node::NodeTakeoverProof; + +use super::{ + AcceptedFleetAction, ActivationEvidence, FleetActionKind, FleetScope, MoveAttemptSpec, + MovementAction, OperationError, PublishedPosition, Result, +}; + +/// Original pinned input of recovery of an unresolved source release. +/// +/// Construction requires canonical failed-session takeover proof. Decoding +/// validates only shape; the trusted journal supplies provenance. The accepted +/// action remains separately retained under `action_key`; this record grants +/// no takeover authority and cannot replace fresh actor/authority checks. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct RecoveryBasis { + pub(super) spec: MoveAttemptSpec, + pub(super) scope: FleetScope, + pub(super) action_key: Digest, + pub(super) node: NodeId, + pub(super) session: SessionId, + pub(super) accepted_at_ms: i64, + pub(super) control: Control, + pub(super) observed_at_ms: i64, +} + +impl RecoveryBasis { + /// Captures the exact source-epoch input before canonical acquisition CAS. + /// An Idle input covers source loss after an unobserved release CAS; it is + /// still recovery evidence and never manufactures a clean release result. + pub fn new( + accepted: &AcceptedFleetAction, + control: Control, + takeover: NodeTakeoverProof, + observed_at_ms: i64, + ) -> Result { + let FleetActionKind::Movement { + action: MovementAction::Recover, + attempt, + } = accepted.action().kind() + else { + return Err(OperationError::Invalid( + "recovery requires accepted recovery action", + )); + }; + if takeover.session() != attempt.spec().source || takeover.claimant() != accepted.session() + { + return Err(OperationError::Fenced); + } + let basis = Self { + spec: attempt.spec().clone(), + scope: accepted.action().scope(), + action_key: accepted.action().key()?, + node: accepted.node(), + session: accepted.session(), + accepted_at_ms: accepted.accepted_at_ms(), + control, + observed_at_ms, + }; + basis.validate()?; + basis.validate_acceptance(accepted)?; + Ok(basis) + } + + /// Checks the original acceptance without accepting another effect. + pub fn validate_acceptance(&self, accepted: &AcceptedFleetAction) -> Result<()> { + self.validate()?; + let FleetActionKind::Movement { + action: MovementAction::Recover, + attempt, + } = accepted.action().kind() + else { + return Err(OperationError::Invalid("recovery acceptance kind mismatch")); + }; + if self.spec != *attempt.spec() + || self.scope != accepted.action().scope() + || self.action_key != accepted.action().key()? + || self.node != accepted.node() + || self.session != accepted.session() + || self.accepted_at_ms != accepted.accepted_at_ms() + { + return Err(OperationError::Conflict); + } + Ok(()) + } + + /// Returns all immutable movement inputs, including the failed source. + #[must_use] + pub const fn spec(&self) -> &MoveAttemptSpec { + &self.spec + } + /// Returns the exact canonical input retained before CAS. + #[must_use] + pub const fn control(&self) -> &Control { + &self.control + } + /// Returns the original action index, never a substitute for full identity. + #[must_use] + pub const fn action_key(&self) -> Digest { + self.action_key + } + /// Returns the original observation time, never refreshed by republication. + #[must_use] + pub const fn observed_at_ms(&self) -> i64 { + self.observed_at_ms + } + + pub(super) fn validate(&self) -> Result<()> { + self.spec.validate()?; + self.scope.validate()?; + self.control + .encode() + .map_err(|e| OperationError::Control(Box::new(e)))?; + if self.scope.application != self.spec.target.application() + || self.action_key + != super::actions::movement_key(self.scope, MovementAction::Recover, &self.spec) + || self.node != self.spec.destination_node + || self.session != self.spec.destination + || self.accepted_at_ms < 0 + || self.observed_at_ms < self.accepted_at_ms + || self.control.cell != self.spec.target.cell_id() + || self.control.incarnation != self.spec.incarnation + || self.control.epoch != self.spec.source_epoch + || self.control.root.is_none() + { + return Err(OperationError::Invalid("recovery basis scope mismatch")); + } + match self.control.state { + ControlState::Serving | ControlState::Recovering + if self + .control + .owner + .as_ref() + .is_some_and(|o| o.session == self.spec.source) => {} + ControlState::Idle + if self.control.owner.is_none() && self.control.recovery.is_none() => {} + _ => { + return Err(OperationError::Invalid( + "recovery basis is not the failed source", + )); + } + } + PublishedPosition { + incarnation: self.control.incarnation, + epoch: self.control.epoch, + root: self + .control + .root + .clone() + .ok_or(OperationError::Invalid("recovery basis lacks root"))?, + } + .validate() + } +} + +/// Immutable canonical recovery position confirmed before successor admission. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct RecoveryEvidence { + pub(super) basis: RecoveryBasis, + pub(super) restored: Control, + pub(super) recorded_at_ms: i64, +} + +impl RecoveryEvidence { + /// Checks the actual takeover and optional pinned-overlay publication. + /// The recorder must durably confirm this value before actor activation. + pub fn new(basis: RecoveryBasis, restored: Control, recorded_at_ms: i64) -> Result { + let evidence = Self { + basis, + restored, + recorded_at_ms, + }; + evidence.validate()?; + Ok(evidence) + } + /// Returns the retained pre-CAS input and exact accepted action binding. + #[must_use] + pub const fn basis(&self) -> &RecoveryBasis { + &self.basis + } + /// Returns the canonical control after recovery publication, before activation. + #[must_use] + pub const fn restored(&self) -> &Control { + &self.restored + } + /// Returns the original recording time, not a serving-freshness timestamp. + #[must_use] + pub const fn recorded_at_ms(&self) -> i64 { + self.recorded_at_ms + } + /// Returns the position a current successor must have restored or advanced. + pub fn position(&self) -> Result { + Ok(PublishedPosition { + incarnation: self.restored.incarnation, + epoch: self.restored.epoch, + root: self + .restored + .root + .clone() + .ok_or(OperationError::Invalid("recovery result lacks root"))?, + }) + } + pub(super) fn validate(&self) -> Result<()> { + self.basis.validate()?; + self.restored + .encode() + .map_err(|e| OperationError::Control(Box::new(e)))?; + let owner = self + .restored + .owner + .clone() + .ok_or(OperationError::Invalid("recovery lacks claimant"))?; + if owner.session != self.basis.session || self.recorded_at_ms < self.basis.observed_at_ms { + return Err(OperationError::Invalid( + "recovery claimant or time mismatch", + )); + } + let claimed = self + .basis + .control + .takeover(owner) + .map_err(|e| OperationError::Control(Box::new(e)))?; + if claimed.recovery.is_some() { + claimed + .validate_transition(&self.restored, Transition::PublishRecovery) + .map_err(|e| OperationError::Control(Box::new(e)))?; + } else if claimed != self.restored { + return Err(OperationError::Invalid( + "recovery changed the pinned root without an overlay", + )); + } + self.position()?.validate() + } +} + +/// Canonical failed-source recovery plus independently checked current serving. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct RecoveredActivation { + /// Immutable basis and exact materialized recovery position. + pub recovery: RecoveryEvidence, + /// Current actor-backed successor, potentially advanced beyond recovery. + pub serving: ActivationEvidence, +} + +impl RecoveredActivation { + pub(super) fn validate(&self) -> Result<()> { + self.recovery.validate()?; + self.serving.position.validate()?; + let spec = self.recovery.basis.spec(); + let required = self.recovery.position()?; + if !super::nonzero(self.serving.node.as_bytes()) + || !super::nonzero(self.serving.session.as_bytes()) + || self.serving.node == spec.source_node + || self.serving.session == spec.source + || self.serving.position.incarnation != spec.incarnation + || self.serving.position.epoch < required.epoch + || (self.serving.position.epoch == required.epoch + && (self.serving.node != self.recovery.basis.node + || self.serving.session != self.recovery.basis.session)) + || !super::attempt::successor_position(&self.serving.position, &required) + || (self.serving.session == spec.destination + && self.serving.node != spec.destination_node) + { + return Err(OperationError::Invalid( + "recovered successor lacks required position", + )); + } + Ok(()) + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/registry.rs b/crates/cellule-runtime/src/fleet/operations/registry.rs new file mode 100644 index 00000000..32587dfb --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/registry.rs @@ -0,0 +1,346 @@ +use crate::identity::{Digest, NodeId}; + +use super::{ + EnrollmentRecord, FleetHead, FleetScope, MAX_PAGE_ENTRIES, MaintenancePhase, MoveAttemptSpec, + NodeIntent, OperationError, Result, nonzero, +}; +use crate::node::NodeMode; + +/// Revision shared by retained physical-node intents, enrollment rows and +/// committed role-evidence pointers in the same journal transaction domain. +/// The adapter advances it in the same transaction as every registry mutation. +/// Readers must recheck it after collecting every required observation. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct RegistryVersion { + pub(super) scope: FleetScope, + pub(super) revision: u64, + pub(super) bootstrap_revision: Option, + pub(super) scheduling_enabled: bool, +} + +impl RegistryVersion { + /// Creates an unbootstrapped empty registry. It cannot prove complete coverage. + pub fn new(scope: FleetScope) -> Result { + scope.validate()?; + Ok(Self { + scope, + revision: 0, + bootstrap_revision: None, + scheduling_enabled: false, + }) + } + /// Calculates the revision committed with a row or intent mutation. + pub fn advance(self, expected_revision: u64) -> Result { + self.validate()?; + if expected_revision != self.revision { + return Err(OperationError::Conflict); + } + let revision = self + .revision + .checked_add(1) + .ok_or(OperationError::Invalid("registry revision overflow"))?; + Ok(Self { revision, ..self }) + } + /// Records the controlled initial coverage barrier. The application pauses + /// enrollment and imports every existing obligation before committing this. + /// Merely invoking this method supplies no evidence that coverage is complete. + pub fn bootstrap(self, expected_revision: u64) -> Result { + self.validate()?; + if expected_revision != self.revision { + return Err(OperationError::Conflict); + } + if self.bootstrap_revision.is_some() { + return Ok(self); + } + let mut next = self.advance(expected_revision)?; + next.bootstrap_revision = Some(next.revision); + Ok(next) + } + + /// Stops or resumes allocation of new planned movement in the same registry + /// transaction. Accepted actions, permits, and retained cordons are unchanged. + /// Enabling requires initial coverage; it is not deployment qualification. + pub fn set_scheduling(self, expected_revision: u64, enabled: bool) -> Result { + self.validate()?; + if expected_revision != self.revision { + return Err(OperationError::Conflict); + } + if enabled && self.bootstrap_revision.is_none() { + return Err(OperationError::Invalid( + "cannot schedule before registry bootstrap", + )); + } + if enabled == self.scheduling_enabled { + return Ok(self); + } + Ok(Self { + scheduling_enabled: enabled, + ..self.advance(expected_revision)? + }) + } + + /// Returns whether the shared transaction may allocate a new move permit. + /// Existing accepted attempts still reconcile while this is false. + #[must_use] + pub const fn scheduling_enabled(self) -> bool { + self.scheduling_enabled + } + + /// Checks allocation against the retained rows inside the head transaction. + /// Invoke before the head's `Allocate` transition; that transition still + /// checks controller fencing, sequence, deadline, and count/byte permits. + /// A draining source may move only for its current evacuation operation. + pub fn authorize_allocation( + self, + head: &FleetHead, + spec: &MoveAttemptSpec, + source: &NodeIntent, + destination: &NodeIntent, + ) -> Result<()> { + self.validate()?; + spec.validate()?; + source.validate()?; + destination.validate()?; + if !self.scheduling_enabled { + return Err(OperationError::Stopped); + } + if head.scope() != self.scope + || spec.target.application() != self.scope.application + || source.scope() != self.scope + || destination.scope() != self.scope + || source.node() != spec.source_node + || source.session() != spec.source + || destination.node() != spec.destination_node + || destination.session() != spec.destination + || destination.mode() != NodeMode::Active + { + return Err(OperationError::Conflict); + } + if source.mode() != NodeMode::Active + && head.maintenance().is_none_or(|operation| { + source.operation() != Some(operation.id()) + || spec.id.operation != operation.id() + || operation.node() != source.node() + || operation.session() != source.session() + || operation.intent_revision() != source.revision() + || operation.phase() != MaintenancePhase::Evacuating + }) + { + return Err(OperationError::Conflict); + } + Ok(()) + } + /// Returns the journal/application namespace. + #[must_use] + pub const fn scope(self) -> FleetScope { + self.scope + } + /// Returns the stable-scan revision, including failed-session obligations. + #[must_use] + pub const fn revision(self) -> u64 { + self.revision + } + /// Returns the first committed coverage barrier, retained after mutations. + #[must_use] + pub const fn bootstrap_revision(self) -> Option { + self.bootstrap_revision + } + /// Requires an identical bootstrapped revision after collecting observations. + /// A filtered scan or omitted row cannot satisfy this check on its own. + pub fn confirm(self, current: Self) -> Result<()> { + self.validate()?; + current.validate()?; + if self.bootstrap_revision.is_none() { + return Err(OperationError::Invalid( + "registry coverage is unbootstrapped", + )); + } + if self != current { + return Err(OperationError::Conflict); + } + Ok(()) + } + pub(super) fn validate(self) -> Result<()> { + self.scope.validate()?; + if (self.scheduling_enabled && self.bootstrap_revision.is_none()) + || self + .bootstrap_revision + .is_some_and(|revision| revision == 0 || revision > self.revision) + { + return Err(OperationError::Invalid( + "invalid registry bootstrap revision", + )); + } + Ok(()) + } +} + +/// Bounded retained physical-node intents. Cursors bind to the shared version. +/// The adapter includes older cordons, not only the head's current operation. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct IntentPage { + pub(super) version: RegistryVersion, + pub(super) after: Option, + pub(super) entries: Vec, + pub(super) next: Option, +} + +impl IntentPage { + /// Builds a sorted registry page from one consistent transaction. + pub fn new( + version: RegistryVersion, + after: Option, + entries: Vec, + next: Option, + ) -> Result { + let page = Self { + version, + after, + entries, + next, + }; + page.validate()?; + Ok(page) + } + /// Returns the revision to repeat in the next request and final barrier. + #[must_use] + pub const fn version(&self) -> RegistryVersion { + self.version + } + /// Returns the exclusive starting physical identity requested by this page. + #[must_use] + pub const fn after(&self) -> Option { + self.after + } + /// Returns at most 128 intents in ascending physical-node order. + #[must_use] + pub fn entries(&self) -> &[NodeIntent] { + &self.entries + } + /// Returns the last emitted key only when another page remains. + #[must_use] + pub const fn next(&self) -> Option { + self.next + } + pub(super) fn validate(&self) -> Result<()> { + self.version.validate()?; + if self.version.revision == 0 && !self.entries.is_empty() { + return Err(OperationError::Invalid( + "rows exist before the first registry commit", + )); + } + let mut keys = Vec::with_capacity(self.entries.len().min(MAX_PAGE_ENTRIES)); + if self.entries.len() > MAX_PAGE_ENTRIES { + return Err(OperationError::Invalid("intent page exceeds entry bound")); + } + for intent in &self.entries { + intent.validate()?; + if intent.scope() != self.version.scope { + return Err(OperationError::Invalid("intent page scope differs")); + } + keys.push(*intent.node().as_bytes()); + } + validate_page( + self.after.map(|v| *v.as_bytes()), + &keys, + self.next.map(|v| *v.as_bytes()), + ) + } +} + +/// Bounded complete-registry scan, including pending, retired, and failed boots. +/// Tombstones remain visible to preserve request identity and original evidence. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct EnrollmentPage { + pub(super) version: RegistryVersion, + pub(super) after: Option, + pub(super) entries: Vec, + pub(super) next: Option, +} + +impl EnrollmentPage { + /// Builds a sorted page without treating lease expiry as retirement. + pub fn new( + version: RegistryVersion, + after: Option, + entries: Vec, + next: Option, + ) -> Result { + let page = Self { + version, + after, + entries, + next, + }; + page.validate()?; + Ok(page) + } + /// Returns the shared revision to recheck after the scan. + #[must_use] + pub const fn version(&self) -> RegistryVersion { + self.version + } + /// Returns the exclusive request-index starting key. + #[must_use] + pub const fn after(&self) -> Option { + self.after + } + /// Returns at most 128 obligations in ascending stable-index order. + #[must_use] + pub fn entries(&self) -> &[EnrollmentRecord] { + &self.entries + } + /// Returns the last emitted key only when another page remains. + #[must_use] + pub const fn next(&self) -> Option { + self.next + } + pub(super) fn validate(&self) -> Result<()> { + self.version.validate()?; + if self.version.revision == 0 && !self.entries.is_empty() { + return Err(OperationError::Invalid( + "rows exist before the first registry commit", + )); + } + let mut keys = Vec::with_capacity(self.entries.len().min(MAX_PAGE_ENTRIES)); + if self.entries.len() > MAX_PAGE_ENTRIES { + return Err(OperationError::Invalid( + "enrollment page exceeds entry bound", + )); + } + for record in &self.entries { + record.validate()?; + if record.spec().scope != self.version.scope { + return Err(OperationError::Invalid("enrollment page scope differs")); + } + keys.push(*record.spec().key()?.as_bytes()); + } + validate_page( + self.after.map(|v| *v.as_bytes()), + &keys, + self.next.map(|v| *v.as_bytes()), + ) + } +} + +fn validate_page( + after: Option<[u8; N]>, + keys: &[[u8; N]], + next: Option<[u8; N]>, +) -> Result<()> { + if after.is_some_and(|v| !nonzero(&v)) + || next.is_some_and(|v| !nonzero(&v)) + || keys.len() > MAX_PAGE_ENTRIES + || keys.windows(2).any(|w| w[0] >= w[1]) + || keys + .first() + .is_some_and(|key| after.is_some_and(|a| *key <= a)) + || next.is_some_and(|n| keys.last().copied() != Some(n)) + || (keys.is_empty() && after.is_some()) + { + return Err(OperationError::Invalid( + "invalid registry page cursor or ordering", + )); + } + Ok(()) +} diff --git a/crates/cellule-runtime/src/fleet/operations/tests/absence.rs b/crates/cellule-runtime/src/fleet/operations/tests/absence.rs new file mode 100644 index 00000000..28169e56 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/tests/absence.rs @@ -0,0 +1,128 @@ +use super::*; + +fn resolve(head: &FleetHead, id: AttemptId, effect: MovementAction, now: i64) -> FleetHead { + head.transition( + FleetProfile::default(), + head.revision(), + head.controller().unwrap().epoch, + now, + JournalTransition::ResolveUnaccepted { id, effect }, + ) + .unwrap() +} + +#[test] +fn atomic_absence_resolution_keeps_resources_and_fences_old_authorization() { + let (head, id) = reserved(); + let head = attempt(&head, id, AttemptEvent::BeginRelease); + let delayed = head + .movement_action(id, MovementAction::Release, 0) + .unwrap(); + let unmarked = resolve(&head, id, MovementAction::Release, 0); + assert_eq!(unmarked.revision(), head.revision() + 1); + assert_eq!(unmarked.attempts(), head.attempts()); + assert!( + AcceptedFleetAction::new( + delayed.clone(), + &unmarked, + spec(1).source_node, + spec(1).source, + 0 + ) + .is_err() + ); + let head = attempt(&head, id, AttemptEvent::OutcomeUnknown); + let resolved = resolve(&head, id, MovementAction::Release, 1); + assert_eq!(resolved.revision(), head.revision() + 1); + assert_eq!(resolved.attempts()[0].phase(), AttemptPhase::Releasing); + assert_eq!(resolved.attempts()[0].blocker(), None); + assert_eq!( + resolved.reserved_restore_bytes(), + head.reserved_restore_bytes() + ); + assert_eq!( + resolved.attempts()[0].reservation(), + head.attempts()[0].reservation() + ); + assert!(resolved.attempts()[0].released().is_none()); + assert!( + AcceptedFleetAction::new(delayed, &resolved, spec(1).source_node, spec(1).source, 1) + .is_err() + ); + let fresh = resolved + .movement_action(id, MovementAction::Release, 1) + .unwrap(); + assert!( + AcceptedFleetAction::new(fresh, &resolved, spec(1).source_node, spec(1).source, 1).is_ok() + ); +} + +#[test] +fn expired_unaccepted_release_returns_to_cleanup_without_fabricating_release() { + let (head, id) = reserved(); + let head = attempt(&head, id, AttemptEvent::BeginRelease); + let head = attempt(&head, id, AttemptEvent::OutcomeUnknown); + let resolved = resolve(&head, id, MovementAction::Release, spec(1).deadline_ms); + let row = &resolved.attempts()[0]; + assert_eq!(row.phase(), AttemptPhase::Reserved); + assert_eq!(row.blocker(), Some(DrainBlocker::Deadline)); + assert!(row.released().is_none() && row.activated().is_none()); + assert!(!row.can_retire()); + assert_eq!(resolved.reserved_restore_bytes(), 4096); + let cancelled = attempt(&resolved, id, AttemptEvent::BeginCancel); + assert_eq!(cancelled.attempts()[0].phase(), AttemptPhase::Cancelling); + assert_eq!(cancelled.reserved_restore_bytes(), 4096); +} + +#[test] +fn absence_resolution_matches_dispatch_phase_and_does_not_extend_admission() { + let spec = spec(1); + let id = spec.id; + let planned = transition(&head(), JournalTransition::Allocate(spec.clone())); + assert!( + planned + .transition( + FleetProfile::default(), + planned.revision(), + 1, + 0, + JournalTransition::ResolveUnaccepted { + id, + effect: MovementAction::Prepare + } + ) + .is_err() + ); + let preparing = attempt(&planned, id, AttemptEvent::BeginPrepare); + assert!( + preparing + .transition( + FleetProfile::default(), + preparing.revision(), + 1, + 0, + JournalTransition::ResolveUnaccepted { + id, + effect: MovementAction::Release + } + ) + .is_err() + ); + let expired = resolve(&preparing, id, MovementAction::Prepare, spec.deadline_ms); + assert_eq!(expired.attempts()[0].phase(), AttemptPhase::Cancelling); + assert_eq!(expired.attempts()[0].spec().deadline_ms, spec.deadline_ms); + assert_eq!(expired.reserved_restore_bytes(), 4096); + let (head, id) = reserved(); + let head = attempt(&head, id, AttemptEvent::BeginRelease); + let head = attempt(&head, id, AttemptEvent::Released(release())); + let head = attempt(&head, id, AttemptEvent::BeginActivate); + let head = attempt(&head, id, AttemptEvent::OutcomeUnknown); + let activated = resolve(&head, id, MovementAction::Activate, spec.deadline_ms); + assert_eq!(activated.attempts()[0].phase(), AttemptPhase::Activating); + assert_eq!( + activated.attempts()[0].released(), + head.attempts()[0].released() + ); + assert!(activated.attempts()[0].activated().is_none()); + assert_eq!(activated.reserved_restore_bytes(), 4096); +} diff --git a/crates/cellule-runtime/src/fleet/operations/tests/accepted.rs b/crates/cellule-runtime/src/fleet/operations/tests/accepted.rs new file mode 100644 index 00000000..88abc09b --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/tests/accepted.rs @@ -0,0 +1,157 @@ +use super::*; + +pub(super) fn accepted_release() -> (FleetHead, AcceptedFleetAction) { + let (head, id) = reserved(); + let head = attempt(&head, id, AttemptEvent::BeginRelease); + let action = head + .movement_action(id, MovementAction::Release, 0) + .unwrap(); + let accepted = + AcceptedFleetAction::new(action, &head, spec(1).source_node, spec(1).source, 1).unwrap(); + (head, accepted) +} + +#[test] +fn acceptance_checks_current_head_endpoint_deadline_and_scope() { + let (head, accepted) = accepted_release(); + let action = accepted.action().clone(); + for (node, session) in [ + (spec(1).destination_node, spec(1).destination), + (spec(1).source_node, SessionId::from_bytes([99; 16])), + (NodeId::from_bytes([99; 16]), spec(1).source), + ] { + assert!(AcceptedFleetAction::new(action.clone(), &head, node, session, 1).is_err()); + } + assert!( + AcceptedFleetAction::new( + action.clone(), + &head, + accepted.node(), + accepted.session(), + 10_000 + ) + .is_err() + ); + let mut altered = action.clone(); + altered.scope.application = ApplicationId::from_bytes([99; 16]); + assert!( + AcceptedFleetAction::new(altered, &head, accepted.node(), accepted.session(), 1).is_err() + ); + let newer = head + .claim( + FleetProfile::default(), + head.revision(), + action.controller(), + 1, + ) + .unwrap(); + assert!( + AcceptedFleetAction::new(action, &newer, accepted.node(), accepted.session(), 1).is_err() + ); +} + +#[test] +fn stable_key_cannot_substitute_cost_epoch_nodes_or_snapshot() { + let (_, accepted) = accepted_release(); + for field in 0..5 { + let mut altered = accepted.action().clone(); + let FleetActionKind::Movement { attempt, .. } = &mut altered.kind else { + panic!() + }; + match field { + 0 => attempt.spec.cost.disk_bytes += 1, + 1 => attempt.spec.source_epoch += 1, + 2 => attempt.spec.source_node = NodeId::from_bytes([99; 16]), + 3 => attempt.spec.snapshot_digest = Digest::from_bytes([99; 32]), + _ => attempt.spec.deadline_ms += 1, + } + assert_eq!(altered.key().unwrap(), accepted.action().key().unwrap()); + assert!(matches!( + accepted.validate_replay(&altered, accepted.node(), accepted.session()), + Err(OperationError::Conflict) + )); + } +} + +#[test] +fn existing_acceptance_survives_changed_controller_but_never_creates_new_work() { + let (head, accepted) = accepted_release(); + let mut replay = accepted.action().clone(); + replay.controller = SessionId::from_bytes([88; 16]); + replay.controller_epoch += 1; + replay.journal_revision += 1; + replay.issued_at_ms = 2; + accepted + .validate_replay(&replay, accepted.node(), accepted.session()) + .unwrap(); + assert!(replay.authorize_against(&head, 2).is_err()); + let expired = head + .claim( + FleetProfile::default(), + head.revision(), + replay.controller, + 40_000, + ) + .unwrap(); + assert!( + AcceptedFleetAction::new( + accepted.action().clone(), + &expired, + accepted.node(), + accepted.session(), + 40_000 + ) + .is_err() + ); +} + +#[test] +fn result_must_bind_original_endpoint_and_acceptance_time() { + let (_, accepted) = accepted_release(); + let result = FleetActionOutcome { + scope: accepted.action().scope(), + action_key: accepted.action().key().unwrap(), + node: accepted.node(), + session: accepted.session(), + observed_at_ms: 2, + outcome: FleetOutcome::Released(release()), + }; + accepted.validate_result(&result).unwrap(); + let mut changed = result.clone(); + changed.observed_at_ms = 0; + assert!(accepted.validate_result(&changed).is_err()); + changed = result.clone(); + changed.session = spec(1).destination; + assert!(accepted.validate_result(&changed).is_err()); + changed = result; + let FleetOutcome::Released(position) = &mut changed.outcome else { + panic!() + }; + position.epoch += 1; + assert!(accepted.validate_result(&changed).is_err()); +} + +#[test] +fn bounded_acceptance_codec_rejects_partial_trailing_version_and_oversized_records() { + let (_, accepted) = accepted_release(); + let bytes = accepted.to_bytes().unwrap(); + assert_eq!(AcceptedFleetAction::from_bytes(&bytes).unwrap(), accepted); + for end in 0..bytes.len() { + assert!(AcceptedFleetAction::from_bytes(&bytes[..end]).is_err()); + } + let mut trailing = bytes.clone(); + trailing.push(0); + assert!(AcceptedFleetAction::from_bytes(&trailing).is_err()); + let mut wrong_version = bytes.clone(); + wrong_version[4 + b"cellule.fleet-operation\0".len()] = 255; + assert!(AcceptedFleetAction::from_bytes(&wrong_version).is_err()); + assert!(AcceptedFleetAction::from_bytes(&vec![0; MAX_RECORD_BYTES as usize + 1]).is_err()); + let mut wrong_time = accepted.clone(); + wrong_time.accepted_at_ms = -1; + assert!(wrong_time.to_bytes().is_err()); + wrong_time.accepted_at_ms = spec(1).deadline_ms; + assert!(wrong_time.to_bytes().is_err()); + let mut wrong_endpoint = accepted; + wrong_endpoint.session = spec(1).destination; + assert!(wrong_endpoint.to_bytes().is_err()); +} diff --git a/crates/cellule-runtime/src/fleet/operations/tests/acquisition.rs b/crates/cellule-runtime/src/fleet/operations/tests/acquisition.rs new file mode 100644 index 00000000..f609100f --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/tests/acquisition.rs @@ -0,0 +1,116 @@ +use super::*; +use crate::control::{Control, ControlState, Owner}; + +fn basis() -> AcquisitionBasis { + let (head, id) = reserved(); + let head = attempt(&head, id, AttemptEvent::BeginRelease); + let head = attempt(&head, id, AttemptEvent::Released(release())); + let head = attempt(&head, id, AttemptEvent::BeginActivate); + let action = head + .movement_action(id, MovementAction::Activate, 0) + .unwrap(); + let accepted = AcceptedFleetAction::new( + action, + &head, + spec(1).destination_node, + spec(1).destination, + 1, + ) + .unwrap(); + let mut control = Control::initial( + spec(1).target.cell_id(), + spec(1).incarnation, + Owner { + session: spec(1).source, + endpoint: "https://source.internal:8789".into(), + }, + Digest::from_bytes([1; 32]), + 1, + ) + .unwrap(); + control.epoch = release().epoch; + control.revision = 9; + control.progress = 9; + control.state = ControlState::Idle; + control.owner = None; + control.root = Some(release().root); + AcquisitionBasis::new(accepted, control, 2).unwrap() +} + +#[test] +fn acquisition_basis_requires_exact_idle_scope_and_release_position() { + let original = basis(); + for change in 0..6 { + let mut control = original.control().clone(); + match change { + 0 => control.state = ControlState::Serving, + 1 => control.incarnation = IncarnationId::from_bytes([99; 16]), + 2 => control.epoch -= 1, + 3 => control.root.as_mut().unwrap().digest = Digest::from_bytes([99; 32]), + 4 => control.root = None, + _ => control.cell = spec(2).target.cell_id(), + } + assert!(AcquisitionBasis::new(original.accepted().clone(), control, 2).is_err()); + } + assert!( + AcquisitionBasis::new(original.accepted().clone(), original.control().clone(), 0).is_err() + ); + let (_, accepted_release) = super::accepted::accepted_release(); + assert!(AcquisitionBasis::new(accepted_release, original.control().clone(), 2).is_err()); +} + +#[test] +fn intervening_idle_epoch_retains_its_exact_input_without_rewriting_source_release() { + let original = basis(); + let mut control = original.control().clone(); + control.epoch += 1; + let root = control.root.as_mut().unwrap(); + root.commit_sequence += 1; + root.txid += 1; + root.digest = Digest::from_bytes([99; 32]); + let newer = AcquisitionBasis::new(original.accepted().clone(), control, 3).unwrap(); + assert_ne!(newer.position().unwrap(), release()); + assert_eq!(original.position().unwrap(), release()); + assert_eq!( + AcquisitionBasis::from_bytes(&newer.to_bytes().unwrap()).unwrap(), + newer + ); +} + +#[test] +fn acquisition_basis_requires_an_activation_result_after_the_retained_input() { + let original = basis(); + let result = FleetActionOutcome { + scope: original.accepted().action().scope(), + action_key: original.accepted().action().key().unwrap(), + node: spec(1).destination_node, + session: spec(1).destination, + observed_at_ms: 3, + outcome: FleetOutcome::Activated(activation(2, 12)), + }; + original.validate_result(&result).unwrap(); + let mut changed = result.clone(); + changed.observed_at_ms = 1; + assert!(original.validate_result(&changed).is_err()); + changed = result; + let FleetOutcome::Activated(evidence) = &mut changed.outcome else { + panic!() + }; + evidence.position.epoch = original.control().epoch; + assert!(original.validate_result(&changed).is_err()); +} + +#[test] +fn acquisition_basis_codec_rejects_partial_extra_wrong_kind_and_oversized_records() { + let original = basis(); + let bytes = original.to_bytes().unwrap(); + assert_eq!(AcquisitionBasis::from_bytes(&bytes).unwrap(), original); + for end in 0..bytes.len() { + assert!(AcquisitionBasis::from_bytes(&bytes[..end]).is_err()); + } + let mut changed = bytes; + changed.push(0); + assert!(AcquisitionBasis::from_bytes(&changed).is_err()); + assert!(AcquisitionBasis::from_bytes(&original.accepted().to_bytes().unwrap()).is_err()); + assert!(AcquisitionBasis::from_bytes(&vec![0; MAX_RECORD_BYTES as usize + 1]).is_err()); +} diff --git a/crates/cellule-runtime/src/fleet/operations/tests/cleanup.rs b/crates/cellule-runtime/src/fleet/operations/tests/cleanup.rs new file mode 100644 index 00000000..819d9171 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/tests/cleanup.rs @@ -0,0 +1,95 @@ +use super::*; + +fn released() -> (FleetHead, AttemptId) { + let (head, id) = reserved(); + let head = attempt(&head, id, AttemptEvent::BeginRelease); + (attempt(&head, id, AttemptEvent::Released(release())), id) +} + +#[test] +fn receiver_cleanup_after_release_retains_position_and_both_fleet_permits() { + let (head, id) = released(); + let bytes = head.reserved_restore_bytes(); + let cleaning = attempt(&head, id, AttemptEvent::BeginCancel); + assert_eq!( + cleaning.attempts()[0].phase(), + AttemptPhase::CleaningReceiver + ); + assert_eq!(cleaning.attempts()[0].next_action(), MovementAction::Cancel); + assert_eq!(cleaning.attempts()[0].released(), Some(&release())); + assert_eq!(cleaning.reserved_restore_bytes(), bytes); + assert_eq!( + FleetHead::from_bytes(&cleaning.to_bytes().unwrap()).unwrap(), + cleaning + ); + let action = cleaning + .movement_action(id, MovementAction::Cancel, 0) + .unwrap(); + assert_eq!( + FleetAction::from_bytes(&action.to_bytes().unwrap()).unwrap(), + action + ); + rejected(&cleaning, id, AttemptEvent::Cancelled); + let cleaned = attempt(&cleaning, id, AttemptEvent::ReceiverCleaned); + assert_eq!(cleaned.attempts()[0].phase(), AttemptPhase::Released); + assert_eq!( + cleaned.attempts()[0].next_action(), + MovementAction::Activate + ); + assert_eq!(cleaned.attempts()[0].released(), Some(&release())); + assert_eq!(cleaned.reserved_restore_bytes(), bytes); + assert!(cleaned.retirement_page(&[id]).is_err()); + let activated = attempt(&cleaned, id, AttemptEvent::BeginActivate); + let activated = attempt(&activated, id, AttemptEvent::Activated(activation(2, 12))); + assert_eq!( + activated.attempts()[0].next_action(), + MovementAction::Retire + ); + assert!( + transition(&activated, retirement(&activated, id)) + .attempts() + .is_empty() + ); +} + +#[test] +fn preferred_session_serving_is_not_evidence_that_prepared_credit_was_consumed() { + let (head, id) = released(); + let head = attempt(&head, id, AttemptEvent::BeginActivate); + let head = attempt(&head, id, AttemptEvent::Activated(activation(2, 12))); + assert_eq!(head.attempts()[0].next_action(), MovementAction::Cancel); + assert!(!head.attempts()[0].can_retire()); + assert!(head.retirement_page(&[id]).is_err()); + let head = attempt(&head, id, AttemptEvent::ReceiverCleaned); + assert!(head.attempts()[0].can_retire()); + assert_eq!( + FleetHead::from_bytes(&head.to_bytes().unwrap()).unwrap(), + head + ); +} + +#[test] +fn ambiguous_cleanup_preserves_relocation_until_positive_resource_and_serving_evidence() { + let (head, id) = released(); + let head = attempt(&head, id, AttemptEvent::BeginActivate); + let head = attempt(&head, id, AttemptEvent::BeginCancel); + let head = attempt(&head, id, AttemptEvent::OutcomeUnknown); + assert_eq!(head.attempts()[0].next_action(), MovementAction::Inspect); + assert!(head.movement_action(id, MovementAction::Cancel, 0).is_err()); + rejected(&head, id, AttemptEvent::Cancelled); + let head = attempt(&head, id, AttemptEvent::ReceiverCleaned); + assert_eq!(head.attempts()[0].phase(), AttemptPhase::Released); + assert_eq!(head.attempts()[0].blocker(), None); + assert!(!head.attempts()[0].can_retire()); +} + +#[test] +fn activation_finishing_during_cleanup_still_needs_independent_resource_settlement() { + let (head, id) = released(); + let head = attempt(&head, id, AttemptEvent::BeginActivate); + let head = attempt(&head, id, AttemptEvent::BeginCancel); + let head = attempt(&head, id, AttemptEvent::Activated(activation(2, 12))); + assert!(!head.attempts()[0].can_retire()); + let head = attempt(&head, id, AttemptEvent::ReceiverCleaned); + assert!(head.attempts()[0].can_retire()); +} diff --git a/crates/cellule-runtime/src/fleet/operations/tests/contracts.rs b/crates/cellule-runtime/src/fleet/operations/tests/contracts.rs new file mode 100644 index 00000000..a9b1443d --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/tests/contracts.rs @@ -0,0 +1,475 @@ +use super::*; +use crate::node::NodeMode; + +fn cancelled(head: &FleetHead, sequence: u64) -> FleetHead { + let spec = spec(sequence); + let id = spec.id; + let head = transition(head, JournalTransition::Allocate(spec)); + let head = attempt(&head, id, AttemptEvent::BeginCancel); + attempt(&head, id, AttemptEvent::Cancelled) +} + +#[test] +fn retirement_publishes_exact_history_atomically_and_extends_it_after_restart() { + let head = cancelled(&head(), 1); + let id = spec(1).id; + let page = head.retirement_page(&[id]).unwrap(); + assert_eq!(page.sequence(), 1); + assert_eq!(page.previous(), None); + assert_eq!(page.entries()[0].completed_at_ms(), Some(0)); + assert_eq!( + ProgressPage::from_bytes(&page.to_bytes().unwrap()).unwrap(), + page + ); + let digest = page.digest().unwrap(); + let old_revision = head.revision(); + let retired = transition( + &head, + JournalTransition::Retire { + progress: page.clone(), + }, + ); + assert!(retired.attempts().is_empty()); + assert_eq!( + retired.progress(), + Some(ProgressHead { + digest, + sequence: 1 + }) + ); + assert_eq!(retired.reserved_restore_bytes(), 0); + assert_eq!(head.attempts().len(), 1); + assert!(matches!( + retired.transition( + FleetProfile::default(), + old_revision, + 1, + 0, + JournalTransition::Retire { progress: page } + ), + Err(OperationError::Conflict) + )); + + let restored = FleetHead::from_bytes(&retired.to_bytes().unwrap()).unwrap(); + let adopted = restored + .claim( + FleetProfile::default(), + restored.revision(), + SessionId::from_bytes([33; 16]), + 30_000, + ) + .unwrap(); + let mut next_spec = spec(2); + next_spec.deadline_ms = 50_000; + let second = transition(&adopted, JournalTransition::Allocate(next_spec)); + let second = attempt(&second, spec(2).id, AttemptEvent::BeginCancel); + let second = attempt(&second, spec(2).id, AttemptEvent::Cancelled); + let page = second.retirement_page(&[spec(2).id]).unwrap(); + assert_eq!(page.previous(), Some(digest)); + assert_eq!(page.sequence(), 2); + assert_eq!(page.entries()[0].completed_at_ms(), Some(30_000)); + let retired = transition(&second, JournalTransition::Retire { progress: page }); + assert_eq!(retired.progress().unwrap().sequence, 2); +} + +#[test] +fn retirement_refuses_unknown_unfinished_changed_or_unreachable_history() { + let allocated = transition(&head(), JournalTransition::Allocate(spec(1))); + assert!(allocated.retirement_page(&[spec(1).id]).is_err()); + let head = cancelled(&head(), 1); + assert!(head.retirement_page(&[]).is_err()); + assert!(head.retirement_page(&[spec(2).id]).is_err()); + assert!(head.retirement_page(&[spec(1).id, spec(1).id]).is_err()); + let page = head.retirement_page(&[spec(1).id]).unwrap(); + for invalid in [ + ProgressPage { + sequence: 2, + previous: Some(Digest::from_bytes([88; 32])), + ..page.clone() + }, + ProgressPage { + scope: FleetScope { + fleet: Digest::from_bytes([99; 32]), + ..page.scope + }, + ..page.clone() + }, + ProgressPage { + entries: vec![MoveAttempt { + completed_at_ms: Some(1), + ..page.entries[0].clone() + }], + ..page.clone() + }, + ] { + assert!( + head.transition( + FleetProfile::default(), + head.revision(), + 1, + 1, + JournalTransition::Retire { progress: invalid } + ) + .is_err() + ); + } + assert_eq!(head.attempts().len(), 1); + assert_eq!(head.reserved_restore_bytes(), 4096); +} + +#[test] +fn progress_pages_bound_count_before_allocation_and_reject_noncanonical_envelopes() { + let head = cancelled(&head(), 1); + let page = head.retirement_page(&[spec(1).id]).unwrap(); + let bytes = page.to_bytes().unwrap(); + for end in 0..bytes.len() { + assert!(ProgressPage::from_bytes(&bytes[..end]).is_err()); + } + let mut trailing = bytes.clone(); + trailing.push(0); + assert!(ProgressPage::from_bytes(&trailing).is_err()); + assert!(ProgressPage::from_bytes(&vec![0; MAX_PAGE_BYTES as usize + 1]).is_err()); + let mut future = bytes.clone(); + future[4 + b"cellule.fleet-operation\0".len()] = FORMAT_VERSION + 1; + assert!(ProgressPage::from_bytes(&future).is_err()); + let mut count = bytes; + // First page ends its fixed header with count, then canonical attempt bytes. + let count_offset = + 4 + b"cellule.fleet-operation\0".len() + 2 + (4 + 32) + (4 + 16) + (4 + 16) + 8 + 1; + count[count_offset..count_offset + 4].copy_from_slice(&u32::MAX.to_be_bytes()); + assert!(matches!( + ProgressPage::from_bytes(&count), + Err(OperationError::Codec(crate::codec::CodecError::Limit)) + )); +} + +#[test] +fn node_intent_retains_cordon_until_exact_completed_revision_and_new_boot() { + let scope = head().scope(); + let initial = + NodeIntent::initial(scope, maintenance().node(), maintenance().session()).unwrap(); + assert_eq!(initial.mode(), NodeMode::Active); + assert_eq!( + NodeIntent::from_bytes(&initial.to_bytes().unwrap()).unwrap(), + initial + ); + let mut operation = maintenance(); + let intent = NodeIntent::maintenance(scope, &operation).unwrap(); + assert_eq!(intent.mode(), NodeMode::Draining); + assert_eq!( + NodeIntent::from_bytes(&intent.to_bytes().unwrap()).unwrap(), + intent + ); + let new_boot = SessionId::from_bytes([77; 16]); + assert!(intent.return_to_service(&operation, new_boot, 2).is_err()); + operation.apply(MaintenanceEvent::Cordoned, 0).unwrap(); + operation + .apply(MaintenanceEvent::BeginEvacuation, 0) + .unwrap(); + operation + .apply(MaintenanceEvent::ReadyToClose(drain_evidence()), 0) + .unwrap(); + operation + .apply(MaintenanceEvent::Stopped(drain_evidence()), 0) + .unwrap(); + assert!(intent.return_to_service(&operation, new_boot, 1).is_err()); + assert!( + intent + .return_to_service(&operation, intent.session(), 2) + .is_err() + ); + let active = intent.return_to_service(&operation, new_boot, 2).unwrap(); + assert_eq!(active.mode(), NodeMode::Active); + assert_eq!(active.operation(), None); + let stale = NodeIntent { + revision: 2, + ..intent + }; + assert!(stale.return_to_service(&operation, new_boot, 3).is_err()); +} + +#[test] +fn action_requires_persisted_dispatch_current_revision_live_controller_and_deadline() { + let head = transition(&head(), JournalTransition::Allocate(spec(1))); + let id = spec(1).id; + assert!( + head.movement_action(id, MovementAction::Prepare, 0) + .is_err() + ); + let head = attempt(&head, id, AttemptEvent::BeginPrepare); + let action = head + .movement_action(id, MovementAction::Prepare, 0) + .unwrap(); + action.authorize_against(&head, 1).unwrap(); + assert_eq!( + FleetAction::from_bytes(&action.to_bytes().unwrap()).unwrap(), + action + ); + let advanced = attempt(&head, id, AttemptEvent::OutcomeUnknown); + assert!(matches!( + action.authorize_against(&advanced, 1), + Err(OperationError::Conflict) + )); + assert!( + advanced + .movement_action(id, MovementAction::Prepare, 1) + .is_err() + ); + assert!( + advanced + .movement_action(id, MovementAction::Inspect, 1) + .is_ok() + ); + assert!(matches!( + action.authorize_against(&head, 10_000), + Err(OperationError::Deadline) + )); + assert!(matches!( + action.authorize_against(&head, 30_000), + Err(OperationError::Fenced) + )); + let mut forged = action.clone(); + forged.controller_epoch += 1; + assert!(matches!( + forged.authorize_against(&head, 1), + Err(OperationError::Fenced) + )); + forged = action; + forged.scope.fleet = Digest::from_bytes([55; 32]); + assert!(matches!( + forged.authorize_against(&head, 1), + Err(OperationError::Conflict) + )); +} + +#[test] +fn action_key_survives_controller_adoption_but_not_another_generation() { + let mut spec = spec(1); + spec.deadline_ms = 80_000; + let id = spec.id; + let head = transition(&head(), JournalTransition::Allocate(spec)); + let head = attempt(&head, id, AttemptEvent::BeginPrepare); + let first = head + .movement_action(id, MovementAction::Prepare, 0) + .unwrap(); + let adopted = head + .claim( + FleetProfile::default(), + head.revision(), + SessionId::from_bytes([33; 16]), + 30_000, + ) + .unwrap(); + let second = adopted + .movement_action(id, MovementAction::Prepare, 30_001) + .unwrap(); + assert_eq!(first.key().unwrap(), second.key().unwrap()); + assert!(first.authorize_against(&adopted, 30_001).is_err()); + second.authorize_against(&adopted, 30_001).unwrap(); + let mut other = second.clone(); + if let FleetActionKind::Movement { attempt, .. } = &mut other.kind { + attempt.spec.generation += 1; + } + assert_ne!(other.key().unwrap(), second.key().unwrap()); + assert!(other.authorize_against(&adopted, 30_001).is_err()); +} + +fn reply(action: &FleetAction, source: bool, outcome: FleetOutcome) -> FleetActionOutcome { + FleetActionOutcome { + scope: action.scope(), + action_key: action.key().unwrap(), + node: NodeId::from_bytes([if source { 1 } else { 2 }; 16]), + session: SessionId::from_bytes([if source { 11 } else { 12 }; 16]), + observed_at_ms: 1, + outcome, + } +} + +#[test] +fn action_results_bind_effect_and_authenticated_endpoint_including_failures() { + let (head, id) = reserved(); + let head = attempt(&head, id, AttemptEvent::BeginRelease); + let action = head + .movement_action(id, MovementAction::Release, 0) + .unwrap(); + for outcome in [ + FleetOutcome::Unknown, + FleetOutcome::Blocked(DrainBlocker::BusyExecution), + FleetOutcome::Rejected(DrainBlocker::ReceiverCapacity), + ] { + reply(&action, true, outcome.clone()) + .validate_for(&action) + .unwrap(); + assert!( + reply(&action, false, outcome) + .validate_for(&action) + .is_err() + ); + } + let released = reply(&action, true, FleetOutcome::Released(release())); + released.validate_for(&action).unwrap(); + assert_eq!( + FleetActionOutcome::from_bytes(&released.to_bytes().unwrap()).unwrap(), + released + ); + assert!( + reply(&action, false, FleetOutcome::Released(release())) + .validate_for(&action) + .is_err() + ); + assert!( + reply(&action, true, FleetOutcome::ReceiverCleaned) + .validate_for(&action) + .is_err() + ); + assert!( + reply( + &action, + true, + FleetOutcome::Blocked(DrainBlocker::OutcomeUnknown) + ) + .to_bytes() + .is_err() + ); + assert!( + reply( + &action, + true, + FleetOutcome::Rejected(DrainBlocker::OutcomeUnknown) + ) + .to_bytes() + .is_err() + ); +} + +#[test] +fn activation_reply_cannot_impersonate_a_rebooted_receiver_or_the_source() { + let (head, id) = reserved(); + let head = attempt(&head, id, AttemptEvent::BeginRelease); + let head = attempt(&head, id, AttemptEvent::Released(release())); + let head = attempt(&head, id, AttemptEvent::BeginActivate); + let action = head + .movement_action(id, MovementAction::Activate, 0) + .unwrap(); + let actual = reply(&action, false, FleetOutcome::Activated(activation(2, 12))); + actual.validate_for(&action).unwrap(); + for (node, session) in [(2, 99), (1, 11)] { + let evidence = activation(node, session); + let forged = FleetActionOutcome { + node: evidence.node, + session: evidence.session, + outcome: FleetOutcome::Activated(evidence), + ..actual.clone() + }; + assert!(forged.validate_for(&action).is_err()); + } + let evidence = activation(3, 13); + FleetActionOutcome { + node: evidence.node, + session: evidence.session, + outcome: FleetOutcome::Activated(evidence), + ..actual + } + .validate_for(&action) + .unwrap(); +} + +#[test] +fn maintenance_actions_and_results_require_exact_phase_and_terminal_proof() { + let head = transition(&head(), JournalTransition::BeginMaintenance(maintenance())); + assert!( + head.maintenance_action(MaintenanceAction::Finalize, 0) + .is_err() + ); + let cordon = head + .maintenance_action(MaintenanceAction::Cordon, 0) + .unwrap(); + reply(&cordon, true, FleetOutcome::Cordoned) + .validate_for(&cordon) + .unwrap(); + assert!( + reply(&cordon, true, FleetOutcome::Stopped(drain_evidence())) + .validate_for(&cordon) + .is_err() + ); + let head = transition( + &head, + JournalTransition::Maintenance(MaintenanceEvent::Cordoned), + ); + let head = transition( + &head, + JournalTransition::Maintenance(MaintenanceEvent::BeginEvacuation), + ); + let settle = head + .maintenance_action(MaintenanceAction::SettleRoles, 0) + .unwrap(); + assert!( + reply( + &settle, + true, + FleetOutcome::RolesSettled { + inventory: Digest::from_bytes([0; 32]) + } + ) + .to_bytes() + .is_err() + ); + let head = transition( + &head, + JournalTransition::Maintenance(MaintenanceEvent::ReadyToClose(drain_evidence())), + ); + let finalize = head + .maintenance_action(MaintenanceAction::Finalize, 0) + .unwrap(); + let stopped = reply(&finalize, true, FleetOutcome::Stopped(drain_evidence())); + stopped.validate_for(&finalize).unwrap(); + assert_eq!( + FleetActionOutcome::from_bytes(&stopped.to_bytes().unwrap()).unwrap(), + stopped + ); + assert!( + reply( + &finalize, + true, + FleetOutcome::Stopped(DrainEvidence { + withdrawn: false, + ..drain_evidence() + }) + ) + .to_bytes() + .is_err() + ); +} + +#[test] +fn action_and_intent_codecs_reject_truncation_extra_bytes_and_other_record_families() { + let head = transition(&head(), JournalTransition::BeginMaintenance(maintenance())); + let action = head + .maintenance_action(MaintenanceAction::Cordon, 0) + .unwrap(); + let intent = head.node_intent().unwrap().unwrap(); + let outcome = reply(&action, true, FleetOutcome::Cordoned); + for bytes in [ + action.to_bytes().unwrap(), + intent.to_bytes().unwrap(), + outcome.to_bytes().unwrap(), + ] { + let kind = bytes[5 + b"cellule.fleet-operation\0".len()]; + for end in 0..bytes.len() { + let slice = &bytes[..end]; + assert!(match kind { + 4 => NodeIntent::from_bytes(slice).is_err(), + 6 => FleetAction::from_bytes(slice).is_err(), + 7 => FleetActionOutcome::from_bytes(slice).is_err(), + _ => unreachable!(), + }); + } + let mut trailing = bytes; + trailing.push(0); + assert!(FleetAction::from_bytes(&trailing).is_err()); + assert!(NodeIntent::from_bytes(&trailing).is_err()); + assert!(FleetActionOutcome::from_bytes(&trailing).is_err()); + } + assert!(FleetAction::from_bytes(&intent.to_bytes().unwrap()).is_err()); + assert!(NodeIntent::from_bytes(&action.to_bytes().unwrap()).is_err()); +} diff --git a/crates/cellule-runtime/src/fleet/operations/tests/inspection.rs b/crates/cellule-runtime/src/fleet/operations/tests/inspection.rs new file mode 100644 index 00000000..7566228f --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/tests/inspection.rs @@ -0,0 +1,426 @@ +use super::*; + +fn successor_request(node: u8, session: u8) -> (FleetHead, FleetInspectionRequest) { + let (head, id) = reserved(); + let head = attempt(&head, id, AttemptEvent::BeginRelease); + let head = attempt(&head, id, AttemptEvent::Released(release())); + let request = FleetInspectionRequest::new( + head.movement_action(id, MovementAction::Inspect, 0) + .unwrap(), + RegistryVersion::new(head.scope()).unwrap(), + Digest::from_bytes([71; 32]), + NodeId::from_bytes([node; 16]), + SessionId::from_bytes([session; 16]), + 5_000, + ) + .unwrap(); + (head, request) +} + +#[test] +fn released_successor_inspection_is_read_only_and_preserves_receiver_credit() { + for (node, session) in [(3, 13), (2, 13)] { + let (head, request) = successor_request(node, session); + request + .authorize_against(&head, request.registry(), 1) + .unwrap(); + request + .validate_endpoint(request.node(), request.session()) + .unwrap(); + assert!( + request + .validate_endpoint(NodeId::from_bytes([99; 16]), request.session()) + .is_err() + ); + assert!( + request + .validate_endpoint(request.node(), SessionId::from_bytes([99; 16])) + .is_err() + ); + assert_eq!( + FleetInspectionRequest::from_bytes(&request.to_bytes().unwrap()).unwrap(), + request + ); + let observed = FleetInspectionObservation::new( + request.clone(), + 1, + FleetActionOutcome { + scope: head.scope(), + action_key: request.action().key().unwrap(), + node: request.node(), + session: request.session(), + observed_at_ms: 3, + outcome: FleetOutcome::Activated(activation(node, session)), + }, + ) + .unwrap(); + observed.validate_for(&request, 4, 10).unwrap(); + assert_eq!( + FleetInspectionObservation::from_bytes(&observed.to_bytes().unwrap()).unwrap(), + observed + ); + let id = head.attempts()[0].spec().id; + let activating = attempt(&head, id, AttemptEvent::BeginActivate); + let effect = activating + .movement_action(id, MovementAction::Activate, 1) + .unwrap(); + assert!( + effect + .validate_endpoint(request.node(), request.session()) + .is_err() + ); + assert!( + AcceptedFleetAction::new(effect, &activating, request.node(), request.session(), 1) + .is_err() + ); + assert!( + AcceptedFleetAction::new( + request.action().clone(), + &head, + request.node(), + request.session(), + 1 + ) + .is_err() + ); + let activated = attempt( + &activating, + id, + AttemptEvent::Activated(activation(node, session)), + ); + assert!(!activated.attempts()[0].receiver_resources_settled()); + assert_eq!( + activated.reserved_restore_bytes(), + head.reserved_restore_bytes() + ); + assert!(activated.retirement_page(&[id]).is_err()); + } +} + +#[test] +fn successor_inspection_rejects_source_aliases_and_unreleased_attempts() { + let (head, request) = successor_request(3, 13); + for (node, session) in [(1, 13), (3, 11), (3, 12), (0, 13), (3, 0)] { + assert!( + FleetInspectionRequest::new( + request.action().clone(), + request.registry(), + request.nonce(), + NodeId::from_bytes([node; 16]), + SessionId::from_bytes([session; 16]), + 5_000 + ) + .is_err() + ); + } + let (reserved, id) = reserved(); + for state in [ + reserved.clone(), + attempt(&reserved, id, AttemptEvent::BeginRelease), + ] { + assert!( + FleetInspectionRequest::new( + state + .movement_action(id, MovementAction::Inspect, 0) + .unwrap(), + request.registry(), + request.nonce(), + request.node(), + request.session(), + 5_000 + ) + .is_err() + ); + } + let renewed = head + .claim( + FleetProfile::default(), + head.revision(), + SessionId::from_bytes([99; 16]), + 30_000, + ) + .unwrap(); + assert!( + request + .authorize_against(&renewed, request.registry(), 30_000) + .is_err() + ); +} + +#[test] +fn successor_inspection_cannot_supply_cleanup_or_a_position_before_release() { + let (_, request) = successor_request(3, 13); + let mut behind = activation(3, 13); + behind.position.root.commit_sequence = release().root.commit_sequence - 1; + let mut different_root = activation(3, 13); + different_root.position.root.commit_sequence = release().root.commit_sequence; + different_root.position.root.digest = Digest::from_bytes([99; 32]); + for outcome in [ + FleetOutcome::Activated(behind), + FleetOutcome::Activated(different_root), + FleetOutcome::ReceiverCleaned, + FleetOutcome::Reserved(ReceiverReservation { + session: request.session(), + expires_at_ms: 10_000, + }), + ] { + assert!( + FleetInspectionObservation::new( + request.clone(), + 1, + FleetActionOutcome { + scope: request.action().scope(), + action_key: request.action().key().unwrap(), + node: request.node(), + session: request.session(), + observed_at_ms: 3, + outcome, + } + ) + .is_err() + ); + } + for outcome in [ + FleetOutcome::Unknown, + FleetOutcome::Blocked(DrainBlocker::OutcomeUnknown), + ] { + let observed = FleetInspectionObservation::new( + request.clone(), + 1, + FleetActionOutcome { + scope: request.action().scope(), + action_key: request.action().key().unwrap(), + node: request.node(), + session: request.session(), + observed_at_ms: 3, + outcome: outcome.clone(), + }, + ); + assert_eq!(observed.is_ok(), matches!(outcome, FleetOutcome::Unknown)); + } +} + +fn request() -> (FleetHead, FleetInspectionRequest) { + let (head, id) = reserved(); + let request = FleetInspectionRequest::new( + head.movement_action(id, MovementAction::Inspect, 0) + .unwrap(), + RegistryVersion::new(head.scope()).unwrap(), + Digest::from_bytes([70; 32]), + spec(1).destination_node, + spec(1).destination, + 5_000, + ) + .unwrap(); + (head, request) +} + +fn observation(request: &FleetInspectionRequest) -> FleetInspectionObservation { + FleetInspectionObservation::new( + request.clone(), + 1, + FleetActionOutcome { + scope: request.action().scope(), + action_key: request.action().key().unwrap(), + node: request.node(), + session: request.session(), + observed_at_ms: 3, + outcome: FleetOutcome::Activated(activation(2, 12)), + }, + ) + .unwrap() +} + +#[test] +fn inspection_requires_current_head_and_never_authorizes_an_effect() { + let (head, request) = request(); + request + .authorize_against(&head, request.registry(), 1) + .unwrap(); + assert!(matches!( + request.authorize_against(&head, request.registry(), 5_000), + Err(OperationError::Deadline) + )); + let renewed = head + .claim( + FleetProfile::default(), + head.revision(), + head.controller().unwrap().claimant, + 2, + ) + .unwrap(); + assert!( + request + .authorize_against(&renewed, request.registry(), 2) + .is_err() + ); + assert!( + request + .authorize_against(&head, request.registry(), 30_000) + .is_err() + ); + let changed_registry = request + .registry() + .advance(request.registry().revision()) + .unwrap(); + assert!( + request + .authorize_against(&head, changed_registry, 2) + .is_err() + ); + let release = attempt(&head, spec(1).id, AttemptEvent::BeginRelease) + .movement_action(spec(1).id, MovementAction::Release, 0) + .unwrap(); + assert!( + FleetInspectionRequest::new( + release, + request.registry(), + request.nonce(), + spec(1).source_node, + spec(1).source, + 5_000 + ) + .is_err() + ); + for (node, session) in [ + (request.node(), SessionId::from_bytes([99; 16])), + (NodeId::from_bytes([99; 16]), request.session()), + ] { + assert!( + FleetInspectionRequest::new( + request.action().clone(), + request.registry(), + request.nonce(), + node, + session, + 5_000 + ) + .is_err() + ); + } + assert!( + FleetInspectionRequest::new( + request.action().clone(), + request.registry(), + Digest::from_bytes([0; 32]), + request.node(), + request.session(), + 5_000 + ) + .is_err() + ); + assert!( + FleetInspectionRequest::new( + request.action().clone(), + request.registry(), + request.nonce(), + request.node(), + request.session(), + 0 + ) + .is_err() + ); +} + +#[test] +fn inspection_binds_nonce_full_payload_endpoint_and_original_capture_interval() { + let (_, request) = request(); + let observed = observation(&request); + observed.validate_for(&request, 5, 4).unwrap(); + assert!(observed.validate_for(&request, 5, 3).is_err()); + assert!(observed.validate_for(&request, 2, 10).is_err()); + assert!(observed.validate_for(&request, 5, 0).is_err()); + assert!(observed.validate_for(&request, 5_000, 10_000).is_err()); + for field in 0..6 { + let mut changed = request.clone(); + match field { + 0 => changed.nonce = Digest::from_bytes([71; 32]), + 1 => changed.deadline_ms += 1, + 2 => changed.action.journal_revision += 1, + 3 => changed.action.issued_at_ms += 1, + 4 => { + changed.registry = changed + .registry + .advance(changed.registry.revision()) + .unwrap() + } + _ => { + let FleetActionKind::Movement { attempt, .. } = &mut changed.action.kind else { + panic!() + }; + attempt.spec.cost.disk_bytes += 1; + } + } + assert_eq!( + request.action().key().unwrap(), + changed.action().key().unwrap() + ); + assert_ne!(request.key().unwrap(), changed.key().unwrap()); + assert!(matches!( + observed.validate_for(&changed, 5, 10), + Err(OperationError::Conflict) + )); + } + let source = FleetInspectionRequest::new( + request.action().clone(), + request.registry(), + request.nonce(), + spec(1).source_node, + spec(1).source, + 5_000, + ) + .unwrap(); + assert_ne!(request.key().unwrap(), source.key().unwrap()); + assert!(observed.validate_for(&source, 5, 10).is_err()); + assert_eq!(observed.capture_started_at_ms(), 1); + assert_eq!(observed.capture_finished_at_ms(), 3); +} + +#[test] +fn inspection_codec_rejects_malformed_nested_records_and_does_not_restamp_evidence() { + let (_, request) = request(); + let observed = observation(&request); + let bytes = request.to_bytes().unwrap(); + let encoded = observed.to_bytes().unwrap(); + assert_eq!(FleetInspectionRequest::from_bytes(&bytes).unwrap(), request); + let restored = FleetInspectionObservation::from_bytes(&encoded).unwrap(); + assert_eq!(restored, observed); + assert!(restored.validate_for(&request, 100, 50).is_err()); + for end in 0..bytes.len() { + assert!(FleetInspectionRequest::from_bytes(&bytes[..end]).is_err()); + } + for end in 0..encoded.len() { + assert!(FleetInspectionObservation::from_bytes(&encoded[..end]).is_err()); + } + let mut trailing = encoded.clone(); + trailing.push(0); + assert!(FleetInspectionObservation::from_bytes(&trailing).is_err()); + let mut wrong = encoded; + wrong[4 + b"cellule.fleet-operation\0".len()] = 255; + assert!(FleetInspectionObservation::from_bytes(&wrong).is_err()); + let oversized = vec![0; MAX_RECORD_BYTES as usize + 1]; + assert!(FleetInspectionRequest::from_bytes(&oversized).is_err()); + assert!(FleetInspectionObservation::from_bytes(&oversized).is_err()); + assert!(matches!( + FleetInspectionObservation::from_bytes(&[0]), + Err(OperationError::Codec(_)) + )); +} + +#[test] +fn invalid_observation_origin_and_capture_times_fail_before_encoding() { + let (_, request) = request(); + let observed = observation(&request); + for field in 0..6 { + let mut changed = observed.clone(); + match field { + 0 => changed.capture_started_at_ms = -1, + 1 => changed.capture_started_at_ms = 4, + 2 => changed.outcome.observed_at_ms = request.deadline_ms(), + 3 => changed.outcome.session = spec(1).source, + 4 => changed.outcome.node = spec(1).source_node, + _ => changed.outcome.action_key = Digest::from_bytes([99; 32]), + } + assert!(changed.to_bytes().is_err()); + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/tests/maintenance_release.rs b/crates/cellule-runtime/src/fleet/operations/tests/maintenance_release.rs new file mode 100644 index 00000000..41638e81 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/tests/maintenance_release.rs @@ -0,0 +1,234 @@ +use super::*; + +fn maintenance_reserved() -> (FleetHead, AttemptId) { + let (head, id) = reserved(); + let head = transition(&head, JournalTransition::BeginMaintenance(maintenance())); + let head = transition( + &head, + JournalTransition::Maintenance(MaintenanceEvent::Cordoned), + ); + let head = transition( + &head, + JournalTransition::Maintenance(MaintenanceEvent::BeginEvacuation), + ); + (head, id) +} + +#[test] +fn busy_release_requires_exact_evacuating_maintenance_intent() { + let (head, id) = reserved(); + rejected(&head, id, AttemptEvent::BeginMaintenanceRelease); + let requested = transition(&head, JournalTransition::BeginMaintenance(maintenance())); + rejected(&requested, id, AttemptEvent::BeginMaintenanceRelease); + let (head, id) = maintenance_reserved(); + for field in 0..3 { + let mut changed = head.clone(); + let operation = changed.maintenance.as_mut().unwrap(); + match field { + 0 => operation.id = operation_id(99), + 1 => operation.node = NodeId::from_bytes([99; 16]), + _ => operation.session = SessionId::from_bytes([99; 16]), + } + rejected(&changed, id, AttemptEvent::BeginMaintenanceRelease); + } + assert!( + head.transition( + FleetProfile::default(), + head.revision(), + 1, + 10_000, + JournalTransition::Attempt { + id, + event: AttemptEvent::BeginMaintenanceRelease + } + ) + .is_err() + ); + assert_eq!(head.attempts()[0].phase(), AttemptPhase::Reserved); + assert_eq!(head.reserved_restore_bytes(), 4096); +} + +#[test] +fn busy_release_codec_and_results_retain_distinct_dispatch_identity() { + let (head, id) = maintenance_reserved(); + let busy = attempt(&head, id, AttemptEvent::BeginMaintenanceRelease); + assert_eq!( + busy.attempts()[0].phase(), + AttemptPhase::MaintenanceReleasing + ); + assert_eq!(AttemptPhase::MaintenanceReleasing as u8, 13); + assert_eq!(MovementAction::ReleaseMaintenance as u8, 8); + assert_eq!( + FleetHead::from_bytes(&busy.to_bytes().unwrap()).unwrap(), + busy + ); + let action = busy + .movement_action(id, MovementAction::ReleaseMaintenance, 0) + .unwrap(); + assert!( + busy.movement_action(id, MovementAction::Release, 0) + .is_err() + ); + let accepted = + AcceptedFleetAction::new(action, &busy, spec(1).source_node, spec(1).source, 0).unwrap(); + assert_eq!( + AcceptedFleetAction::from_bytes(&accepted.to_bytes().unwrap()).unwrap(), + accepted + ); + assert!( + AcceptedFleetAction::new( + accepted.action().clone(), + &busy, + spec(1).destination_node, + spec(1).destination, + 0 + ) + .is_err() + ); + let result = FleetActionOutcome { + scope: busy.scope(), + action_key: accepted.action().key().unwrap(), + node: spec(1).source_node, + session: spec(1).source, + observed_at_ms: 1, + outcome: FleetOutcome::Released(release()), + }; + accepted.validate_result(&result).unwrap(); + assert_eq!( + FleetActionOutcome::from_bytes(&result.to_bytes().unwrap()).unwrap(), + result + ); + let ordinary = attempt(&head, id, AttemptEvent::BeginRelease); + let ordinary_action = ordinary + .movement_action(id, MovementAction::Release, 0) + .unwrap(); + assert_ne!( + ordinary_action.key().unwrap(), + accepted.action().key().unwrap() + ); + let ordinary_accepted = AcceptedFleetAction::new( + ordinary_action, + &ordinary, + spec(1).source_node, + spec(1).source, + 0, + ) + .unwrap(); + assert!(ordinary_accepted.validate_result(&result).is_err()); + assert!( + accepted + .validate_replay( + ordinary_accepted.action(), + spec(1).source_node, + spec(1).source + ) + .is_err() + ); +} + +#[test] +fn busy_release_unknown_and_absence_keep_permits_and_exact_effect() { + let (head, id) = maintenance_reserved(); + let head = attempt(&head, id, AttemptEvent::BeginMaintenanceRelease); + let head = attempt(&head, id, AttemptEvent::OutcomeUnknown); + rejected(&head, id, AttemptEvent::BeginCancel); + assert_eq!(head.attempts()[0].next_action(), MovementAction::Inspect); + assert!( + head.transition( + FleetProfile::default(), + head.revision(), + 1, + 1, + JournalTransition::ResolveUnaccepted { + id, + effect: MovementAction::Release + } + ) + .is_err() + ); + let resolved = head + .transition( + FleetProfile::default(), + head.revision(), + 1, + 1, + JournalTransition::ResolveUnaccepted { + id, + effect: MovementAction::ReleaseMaintenance, + }, + ) + .unwrap(); + assert_eq!( + resolved.attempts()[0].phase(), + AttemptPhase::MaintenanceReleasing + ); + assert_eq!(resolved.attempts()[0].blocker(), None); + assert_eq!(resolved.reserved_restore_bytes(), 4096); + let expired = resolved + .transition( + FleetProfile::default(), + resolved.revision(), + 1, + 10_000, + JournalTransition::ResolveUnaccepted { + id, + effect: MovementAction::ReleaseMaintenance, + }, + ) + .unwrap(); + assert_eq!(expired.attempts()[0].phase(), AttemptPhase::Reserved); + assert_eq!( + expired.attempts()[0].blocker(), + Some(DrainBlocker::Deadline) + ); + assert!(expired.attempts()[0].released().is_none()); + assert_eq!(expired.reserved_restore_bytes(), 4096); +} + +#[test] +fn controller_replacement_adopts_busy_release_without_changing_policy() { + let (head, id) = maintenance_reserved(); + let head = attempt(&head, id, AttemptEvent::BeginMaintenanceRelease); + let head = attempt(&head, id, AttemptEvent::OutcomeUnknown); + let restored = FleetHead::from_bytes(&head.to_bytes().unwrap()).unwrap(); + let successor = restored + .claim( + FleetProfile::default(), + restored.revision(), + SessionId::from_bytes([99; 16]), + 30_000, + ) + .unwrap(); + assert_eq!(successor.attempts(), head.attempts()); + assert_eq!(successor.reserved_restore_bytes(), 4096); + let released = attempt(&successor, id, AttemptEvent::Released(release())); + assert_eq!(released.attempts()[0].phase(), AttemptPhase::Released); + assert_eq!(released.attempts()[0].released(), Some(&release())); + let recovering = attempt(&successor, id, AttemptEvent::BeginRecover); + assert_eq!(recovering.attempts()[0].phase(), AttemptPhase::Recovering); + assert_eq!(recovering.reserved_restore_bytes(), 4096); +} + +#[test] +fn retained_ordinary_release_is_not_reinterpreted_by_new_maintenance() { + let (head, id) = reserved(); + let ordinary = attempt(&head, id, AttemptEvent::BeginRelease); + let action = ordinary + .movement_action(id, MovementAction::Release, 0) + .unwrap(); + let newer = transition( + &ordinary, + JournalTransition::BeginMaintenance(maintenance()), + ); + assert_eq!(newer.attempts()[0].phase(), AttemptPhase::Releasing); + assert!( + newer + .movement_action(id, MovementAction::ReleaseMaintenance, 0) + .is_err() + ); + let replay = newer + .movement_action(id, MovementAction::Release, 0) + .unwrap(); + assert_eq!(action.key().unwrap(), replay.key().unwrap()); + assert_eq!(action.kind(), replay.kind()); +} diff --git a/crates/cellule-runtime/src/fleet/operations/tests/mod.rs b/crates/cellule-runtime/src/fleet/operations/tests/mod.rs new file mode 100644 index 00000000..48146c71 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/tests/mod.rs @@ -0,0 +1,726 @@ +use crate::identity::{ + ApplicationId, CellTarget, Digest, IncarnationId, NamespaceId, NodeId, SessionId, TenantId, +}; + +use super::*; + +mod absence; +mod accepted; +mod acquisition; +mod cleanup; +mod contracts; +mod inspection; +mod maintenance_release; +mod recovery; +mod registry; + +fn operation_id(byte: u8) -> OperationId { + OperationId::from_bytes([byte; 16]).unwrap() +} + +fn head() -> FleetHead { + FleetHead::new( + FleetScope { + fleet: Digest::from_bytes([1; 32]), + application: ApplicationId::from_bytes([2; 16]), + }, + 0, + ) + .unwrap() + .claim( + FleetProfile::default(), + 0, + SessionId::from_bytes([9; 16]), + 0, + ) + .unwrap() +} + +fn spec(sequence: u64) -> MoveAttemptSpec { + MoveAttemptSpec { + id: AttemptId { + operation: operation_id(1), + sequence, + }, + target: CellTarget::new( + TenantId::from_bytes([3; 16]), + ApplicationId::from_bytes([2; 16]), + NamespaceId::from_bytes([4; 16]), + &sequence.to_be_bytes(), + ) + .unwrap(), + incarnation: IncarnationId::from_bytes([5; 16]), + source_node: NodeId::from_bytes([1; 16]), + source: SessionId::from_bytes([11; 16]), + generation: 7, + source_epoch: 4, + destination_node: NodeId::from_bytes([2; 16]), + destination: SessionId::from_bytes([12; 16]), + cost: TransferCost { + memory_bytes: 64 * 1024, + disk_bytes: 4096, + file_descriptors: 8, + job_credits: 1, + }, + snapshot_digest: Digest::from_bytes([6; 32]), + deadline_ms: 10_000, + } +} + +fn transition(head: &FleetHead, event: JournalTransition) -> FleetHead { + head.transition( + FleetProfile::default(), + head.revision(), + head.controller().unwrap().epoch, + head.last_observed_ms, + event, + ) + .unwrap() +} + +fn retirement(head: &FleetHead, id: AttemptId) -> JournalTransition { + let entry = head + .attempts + .iter() + .find(|attempt| attempt.spec.id == id) + .unwrap() + .clone(); + JournalTransition::Retire { + progress: ProgressPage { + scope: head.scope(), + operation: id.operation, + sequence: head.progress().map_or(1, |progress| progress.sequence + 1), + previous: head.progress().map(|progress| progress.digest), + entries: vec![entry], + }, + } +} + +fn attempt(head: &FleetHead, id: AttemptId, event: AttemptEvent) -> FleetHead { + transition(head, JournalTransition::Attempt { id, event }) +} + +fn release() -> PublishedPosition { + PublishedPosition { + incarnation: IncarnationId::from_bytes([5; 16]), + epoch: 4, + root: crate::control::RootRef { + digest: Digest::from_bytes([8; 32]), + txid: 9, + checksum: cellule_ltx::types::CHECKSUM_FLAG | 10, + commit_sequence: 9, + }, + } +} + +fn activation(node: u8, session: u8) -> ActivationEvidence { + let mut position = release(); + position.epoch = 5; + position.root.commit_sequence = 10; + ActivationEvidence { + node: NodeId::from_bytes([node; 16]), + session: SessionId::from_bytes([session; 16]), + position, + } +} + +fn reserved() -> (FleetHead, AttemptId) { + let spec = spec(1); + let id = spec.id; + let head = transition(&head(), JournalTransition::Allocate(spec)); + let head = attempt(&head, id, AttemptEvent::BeginPrepare); + let head = attempt( + &head, + id, + AttemptEvent::Reserved(ReceiverReservation { + session: SessionId::from_bytes([12; 16]), + expires_at_ms: 20_000, + }), + ); + (head, id) +} + +fn rejected(head: &FleetHead, id: AttemptId, event: AttemptEvent) { + assert!( + head.transition( + FleetProfile::default(), + head.revision(), + head.controller().unwrap().epoch, + head.last_observed_ms, + JournalTransition::Attempt { id, event } + ) + .is_err() + ); +} + +#[test] +fn successor_session_node_root_and_cleanup_proofs_cannot_be_substituted() { + let (head, id) = reserved(); + let head = attempt(&head, id, AttemptEvent::BeginRelease); + let head = attempt(&head, id, AttemptEvent::Released(release())); + let head = attempt(&head, id, AttemptEvent::BeginActivate); + rejected(&head, id, AttemptEvent::Activated(activation(3, 12))); + let mut changed_root = activation(2, 12); + changed_root.position.root.commit_sequence = release().root.commit_sequence; + changed_root.position.root.digest = Digest::from_bytes([99; 32]); + rejected(&head, id, AttemptEvent::Activated(changed_root)); + let alternate = activation(3, 13); + let head = attempt(&head, id, AttemptEvent::Activated(alternate.clone())); + let head = attempt(&head, id, AttemptEvent::OutcomeUnknown); + assert_eq!(head.attempts()[0].next_action(), MovementAction::Inspect); + let head = attempt(&head, id, AttemptEvent::ReceiverCleaned); + let duplicate = attempt(&head, id, AttemptEvent::Activated(alternate)); + assert_eq!(duplicate, head); + assert!(duplicate.attempts()[0].can_retire()); + let mut malformed = head.attempts()[0].clone(); + malformed.phase = AttemptPhase::Cancelled; + malformed.released = None; + malformed.activated = None; + malformed.receiver_cleaned = false; + assert!(malformed.to_bytes().is_err()); +} + +#[test] +fn terminal_maintenance_codec_retains_proof_and_rejects_fabricated_completion() { + let mut operation = MaintenanceOperation::new( + operation_id(1), + Digest::from_bytes([2; 32]), + NodeId::from_bytes([1; 16]), + SessionId::from_bytes([11; 16]), + 1, + 0, + 10_000, + ) + .unwrap(); + operation.apply(MaintenanceEvent::Cordoned, 0).unwrap(); + operation + .apply(MaintenanceEvent::BeginEvacuation, 0) + .unwrap(); + let mut evidence = DrainEvidence { + node: operation.node(), + session: operation.session(), + remaining_cells: 0, + unresolved_attempts: 0, + relocated: true, + readers_settled: true, + followers_settled: true, + facilities_closed: false, + stopped: false, + withdrawn: false, + }; + operation + .apply(MaintenanceEvent::ReadyToClose(evidence), 0) + .unwrap(); + assert_eq!( + MaintenanceOperation::from_bytes(&operation.to_bytes().unwrap()).unwrap(), + operation + ); + let mut corrupted = operation.clone(); + corrupted.phase = MaintenancePhase::Completed; + assert!(corrupted.to_bytes().is_err()); + corrupted = operation.clone(); + corrupted.drain_evidence = None; + assert!(corrupted.to_bytes().is_err()); + evidence.facilities_closed = true; + evidence.stopped = true; + evidence.withdrawn = true; + operation + .apply(MaintenanceEvent::Stopped(evidence), 1) + .unwrap(); + assert_eq!(operation.drain_evidence(), Some(evidence)); + assert_eq!( + MaintenanceOperation::from_bytes(&operation.to_bytes().unwrap()).unwrap(), + operation + ); + corrupted = operation; + corrupted.drain_evidence.as_mut().unwrap().session = SessionId::from_bytes([22; 16]); + assert!(corrupted.to_bytes().is_err()); +} + +#[test] +fn expiry_and_failover_retain_unknown_release_and_fence_the_old_controller() { + let (head, id) = reserved(); + let head = attempt(&head, id, AttemptEvent::BeginRelease); + let head = attempt(&head, id, AttemptEvent::OutcomeUnknown); + let restored = FleetHead::from_bytes(&head.to_bytes().unwrap()).unwrap(); + let successor = restored + .claim( + FleetProfile::default(), + restored.revision(), + SessionId::from_bytes([10; 16]), + 30_000, + ) + .unwrap(); + assert_eq!(successor.controller().unwrap().epoch, 2); + assert_eq!(successor.attempts(), head.attempts()); + assert_eq!(successor.reserved_restore_bytes(), 4096); + assert_eq!( + successor.attempts()[0].next_action(), + MovementAction::Inspect + ); + assert!(matches!( + successor.transition( + FleetProfile::default(), + successor.revision(), + 1, + 30_000, + retirement(&successor, id) + ), + Err(OperationError::Fenced) + )); + assert!(matches!( + head.transition( + FleetProfile::default(), + head.revision(), + 1, + 30_000, + retirement(&head, id) + ), + Err(OperationError::Fenced) + )); + rejected(&successor, id, AttemptEvent::BeginCancel); + let observed = attempt(&successor, id, AttemptEvent::Released(release())); + assert_eq!( + observed.attempts()[0].next_action(), + MovementAction::Activate + ); + // Reconciliation finishes accepted work even though its original admission deadline expired. + let activating = attempt(&observed, id, AttemptEvent::BeginActivate); + let done = attempt(&activating, id, AttemptEvent::Activated(activation(2, 12))); + let done = attempt(&done, id, AttemptEvent::ReceiverCleaned); + let done = transition(&done, retirement(&done, id)); + assert!(done.attempts().is_empty()); +} + +#[test] +fn same_claimant_after_expiry_gets_a_new_fencing_epoch() { + let head = head(); + let renewed = head + .claim( + FleetProfile::default(), + head.revision(), + SessionId::from_bytes([9; 16]), + 10_000, + ) + .unwrap(); + assert_eq!(renewed.controller().unwrap().epoch, 1); + let reacquired = renewed + .claim( + FleetProfile::default(), + renewed.revision(), + SessionId::from_bytes([9; 16]), + 40_000, + ) + .unwrap(); + assert_eq!(reacquired.controller().unwrap().epoch, 2); +} + +#[test] +fn competing_claim_and_cas_revision_do_not_allocate_work() { + let head = head(); + assert!(matches!( + head.claim( + FleetProfile::default(), + head.revision(), + SessionId::from_bytes([8; 16]), + 1 + ), + Err(OperationError::Fenced) + )); + assert!(matches!( + head.transition( + FleetProfile::default(), + 0, + 1, + 1, + JournalTransition::Allocate(spec(1)) + ), + Err(OperationError::Conflict) + )); + assert!( + head.claim( + FleetProfile::default(), + head.revision(), + SessionId::from_bytes([9; 16]), + -1 + ) + .is_err() + ); +} + +#[test] +fn out_of_order_and_wrong_scope_evidence_cannot_establish_readiness() { + let (head, id) = reserved(); + rejected(&head, id, AttemptEvent::Activated(activation(2, 12))); + rejected(&head, id, AttemptEvent::Released(release())); + let head = attempt(&head, id, AttemptEvent::BeginRelease); + let mut wrong = release(); + wrong.epoch += 1; + rejected(&head, id, AttemptEvent::Released(wrong)); + let head = attempt(&head, id, AttemptEvent::Released(release())); + let duplicate = attempt(&head, id, AttemptEvent::Released(release())); + assert_eq!(duplicate, head); + let head = attempt(&head, id, AttemptEvent::BeginActivate); + let mut behind = activation(2, 12); + behind.position.root.commit_sequence = 8; + rejected(&head, id, AttemptEvent::Activated(behind)); + rejected(&head, id, AttemptEvent::Activated(activation(1, 11))); + let mut stale_epoch = activation(2, 12); + stale_epoch.position.epoch = 4; + rejected(&head, id, AttemptEvent::Activated(stale_epoch)); +} + +#[test] +fn alternate_successor_requires_positive_cleanup_before_permit_retirement() { + let (head, id) = reserved(); + let head = attempt(&head, id, AttemptEvent::BeginRelease); + let head = attempt(&head, id, AttemptEvent::Released(release())); + let head = attempt(&head, id, AttemptEvent::BeginActivate); + let head = attempt(&head, id, AttemptEvent::Activated(activation(3, 13))); + assert_eq!(head.attempts()[0].next_action(), MovementAction::Cancel); + assert!( + head.transition( + FleetProfile::default(), + head.revision(), + 1, + 0, + retirement(&head, id) + ) + .is_err() + ); + let head = attempt(&head, id, AttemptEvent::ReceiverCleaned); + assert!( + transition(&head, retirement(&head, id)) + .attempts() + .is_empty() + ); +} + +#[test] +fn unresolved_preparation_and_cancellation_still_consume_the_fleet_budget() { + let head = transition(&head(), JournalTransition::Allocate(spec(1))); + let head = attempt(&head, spec(1).id, AttemptEvent::BeginPrepare); + let head = attempt(&head, spec(1).id, AttemptEvent::OutcomeUnknown); + let head = transition(&head, JournalTransition::Allocate(spec(2))); + assert!(matches!( + head.transition( + FleetProfile::default(), + head.revision(), + 1, + 0, + JournalTransition::Allocate(spec(3)) + ), + Err(OperationError::Budget) + )); + let head = attempt(&head, spec(1).id, AttemptEvent::BeginCancel); + assert_eq!(head.reserved_restore_bytes(), 8192); + assert!( + head.transition( + FleetProfile::default(), + head.revision(), + 1, + 0, + retirement(&head, spec(1).id) + ) + .is_err() + ); + let head = attempt(&head, spec(1).id, AttemptEvent::Cancelled); + let head = transition(&head, retirement(&head, spec(1).id)); + let head = transition(&head, JournalTransition::Allocate(spec(3))); + assert_eq!(head.attempts().len(), 2); + assert_eq!(head.next_sequence(), 4); +} + +#[test] +fn cost_and_duplicate_cell_are_checked_before_allocation() { + let mut large = spec(1); + large.cost.disk_bytes = MAX_RESTORE_BYTES; + let head = transition(&head(), JournalTransition::Allocate(large)); + assert!(matches!( + head.transition( + FleetProfile::default(), + head.revision(), + 1, + 0, + JournalTransition::Allocate(spec(2)) + ), + Err(OperationError::Budget) + )); + let mut duplicate = spec(2); + duplicate.target = spec(1).target; + assert!(matches!( + head.transition( + FleetProfile::default(), + head.revision(), + 1, + 0, + JournalTransition::Allocate(duplicate) + ), + Err(OperationError::Busy) + )); + let mut unknown = spec(2); + unknown.cost.disk_bytes = 0; + assert!( + head.transition( + FleetProfile::default(), + head.revision(), + 1, + 0, + JournalTransition::Allocate(unknown) + ) + .is_err() + ); +} + +fn maintenance() -> MaintenanceOperation { + MaintenanceOperation::new( + operation_id(1), + Digest::from_bytes([10; 32]), + NodeId::from_bytes([1; 16]), + SessionId::from_bytes([11; 16]), + 1, + 0, + 10_000, + ) + .unwrap() +} + +fn drain_evidence() -> DrainEvidence { + DrainEvidence { + node: NodeId::from_bytes([1; 16]), + session: SessionId::from_bytes([11; 16]), + remaining_cells: 0, + unresolved_attempts: 0, + relocated: true, + readers_settled: true, + followers_settled: true, + facilities_closed: true, + stopped: true, + withdrawn: true, + } +} + +#[test] +fn maintenance_relocation_foreign_tails_and_shutdown_all_gate_completion() { + let head = transition(&head(), JournalTransition::BeginMaintenance(maintenance())); + let head = transition( + &head, + JournalTransition::Maintenance(MaintenanceEvent::Cordoned), + ); + let head = transition( + &head, + JournalTransition::Maintenance(MaintenanceEvent::BeginEvacuation), + ); + for evidence in [ + DrainEvidence { + followers_settled: false, + ..drain_evidence() + }, + DrainEvidence { + relocated: false, + ..drain_evidence() + }, + DrainEvidence { + remaining_cells: 1, + ..drain_evidence() + }, + DrainEvidence { + unresolved_attempts: 1, + ..drain_evidence() + }, + DrainEvidence { + readers_settled: false, + ..drain_evidence() + }, + ] { + assert!( + head.transition( + FleetProfile::default(), + head.revision(), + 1, + 0, + JournalTransition::Maintenance(MaintenanceEvent::ReadyToClose(evidence)) + ) + .is_err() + ); + } + let head = transition( + &head, + JournalTransition::Maintenance(MaintenanceEvent::ReadyToClose(drain_evidence())), + ); + let missing_withdrawal = DrainEvidence { + withdrawn: false, + ..drain_evidence() + }; + assert!( + head.transition( + FleetProfile::default(), + head.revision(), + 1, + 0, + JournalTransition::Maintenance(MaintenanceEvent::Stopped(missing_withdrawal)) + ) + .is_err() + ); + let done = transition( + &head, + JournalTransition::Maintenance(MaintenanceEvent::Stopped(drain_evidence())), + ); + assert_eq!( + done.maintenance().unwrap().phase(), + MaintenancePhase::Completed + ); + let duplicate = transition( + &done, + JournalTransition::Maintenance(MaintenanceEvent::Stopped(drain_evidence())), + ); + assert_eq!(duplicate, done); +} + +#[test] +fn deadline_and_reboot_preserve_physical_node_intent_without_false_success() { + let head = transition(&head(), JournalTransition::BeginMaintenance(maintenance())); + let head = transition( + &head, + JournalTransition::Maintenance(MaintenanceEvent::Cordoned), + ); + let head = transition( + &head, + JournalTransition::Maintenance(MaintenanceEvent::BeginEvacuation), + ); + assert!(matches!( + head.transition( + FleetProfile::default(), + head.revision(), + 1, + 10_000, + JournalTransition::Allocate(spec(1)) + ), + Err(OperationError::Deadline) + )); + let head = transition( + &head, + JournalTransition::Maintenance(MaintenanceEvent::Blocked(DrainBlocker::Deadline)), + ); + assert_eq!( + head.maintenance().unwrap().phase(), + MaintenancePhase::Evacuating + ); + let head = transition( + &head, + JournalTransition::Maintenance(MaintenanceEvent::ExtendDeadline(20_000)), + ); + let rebooted = transition( + &head, + JournalTransition::Maintenance(MaintenanceEvent::SessionReplaced(SessionId::from_bytes( + [99; 16], + ))), + ); + let operation = rebooted.maintenance().unwrap(); + assert_eq!(operation.node(), NodeId::from_bytes([1; 16])); + assert_eq!(operation.phase(), MaintenancePhase::Requested); + assert_eq!(operation.intent_revision(), 3); + assert!( + rebooted + .transition( + FleetProfile::default(), + rebooted.revision(), + 1, + 0, + JournalTransition::Maintenance(MaintenanceEvent::Stopped(drain_evidence())) + ) + .is_err() + ); +} + +#[test] +fn maintenance_idempotency_and_unresolved_attempts_prevent_false_close() { + let request = maintenance(); + let head = transition( + &head(), + JournalTransition::BeginMaintenance(request.clone()), + ); + assert_eq!( + transition(&head, JournalTransition::BeginMaintenance(request.clone())), + head + ); + let mut conflicting = request; + conflicting.request_digest = Digest::from_bytes([99; 32]); + assert!(matches!( + head.transition( + FleetProfile::default(), + head.revision(), + 1, + 0, + JournalTransition::BeginMaintenance(conflicting) + ), + Err(OperationError::Conflict) + )); + let head = transition( + &head, + JournalTransition::Maintenance(MaintenanceEvent::Cordoned), + ); + let head = transition( + &head, + JournalTransition::Maintenance(MaintenanceEvent::BeginEvacuation), + ); + let head = transition(&head, JournalTransition::Allocate(spec(1))); + assert!( + head.transition( + FleetProfile::default(), + head.revision(), + 1, + 0, + JournalTransition::Maintenance(MaintenanceEvent::ReadyToClose(drain_evidence())) + ) + .is_err() + ); +} + +#[test] +fn codec_rejects_truncation_trailing_bytes_wrong_record_type_and_oversize() { + let (head, _) = reserved(); + let bytes = head.to_bytes().unwrap(); + assert_eq!(FleetHead::from_bytes(&bytes).unwrap(), head); + for end in 0..bytes.len() { + assert!(FleetHead::from_bytes(&bytes[..end]).is_err()); + } + let mut trailing = bytes.clone(); + trailing.push(0); + assert!(FleetHead::from_bytes(&trailing).is_err()); + assert!(FleetHead::from_bytes(&vec![0; MAX_RECORD_BYTES as usize + 1]).is_err()); + let attempt = &head.attempts()[0]; + let encoded_attempt = attempt.to_bytes().unwrap(); + assert_eq!(MoveAttempt::from_bytes(&encoded_attempt).unwrap(), *attempt); + assert!(FleetHead::from_bytes(&encoded_attempt).is_err()); + let maintenance = maintenance(); + assert_eq!( + MaintenanceOperation::from_bytes(&maintenance.to_bytes().unwrap()).unwrap(), + maintenance + ); +} + +#[test] +fn decoder_and_encoder_reject_unproven_or_overbudget_stored_states() { + let (head, _) = reserved(); + let mut corrupted = head.clone(); + corrupted.attempts[0].phase = AttemptPhase::Activated; + assert!(corrupted.to_bytes().is_err()); + let mut corrupted = head.clone(); + corrupted.attempts[0].spec.cost.disk_bytes = MAX_RESTORE_BYTES + 1; + assert!(corrupted.to_bytes().is_err()); + let mut corrupted = head; + corrupted.next_sequence = 1; + assert!(corrupted.to_bytes().is_err()); +} + +proptest::proptest! { + #[test] + fn arbitrary_small_records_never_invent_a_valid_head(bytes in proptest::collection::vec(proptest::prelude::any::(), 0..2048)) { + if let Ok(head) = FleetHead::from_bytes(&bytes) { + proptest::prop_assert_eq!(head.to_bytes().unwrap(), bytes); + proptest::prop_assert!(head.attempts().len() <= MAX_ACTIVE_ATTEMPTS); + proptest::prop_assert!(head.reserved_restore_bytes() <= MAX_RESTORE_BYTES); + } + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/tests/recovery.rs b/crates/cellule-runtime/src/fleet/operations/tests/recovery.rs new file mode 100644 index 00000000..cb528b70 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/tests/recovery.rs @@ -0,0 +1,281 @@ +use super::*; +use crate::control::{Control, ControlState, Owner, RecoveryOverlayRef}; + +fn recovered(overlay: bool, idle: bool) -> (FleetHead, AcceptedFleetAction, RecoveredActivation) { + let (head, id) = reserved(); + let head = attempt(&head, id, AttemptEvent::BeginRelease); + let head = attempt(&head, id, AttemptEvent::BeginRecover); + let spec = spec(1); + let action = head + .movement_action(id, MovementAction::Recover, 0) + .unwrap(); + let accepted = + AcceptedFleetAction::new(action, &head, spec.destination_node, spec.destination, 0) + .unwrap(); + let mut control = Control::initial( + spec.target.cell_id(), + spec.incarnation, + Owner { + session: spec.source, + endpoint: "https://source.internal:8789".into(), + }, + Digest::from_bytes([35; 32]), + 1, + ) + .unwrap(); + control.epoch = spec.source_epoch; + control.revision = 9; + control.progress = 9; + control.state = if idle { + ControlState::Idle + } else { + ControlState::Serving + }; + control.root = Some(release().root); + if idle { + control.owner = None; + } + if overlay { + control = control + .attach_recovery(RecoveryOverlayRef { + leader_session: spec.source, + log_epoch: 1, + manifest_digest: Digest::from_bytes([36; 32]), + first_node_sequence: 1, + last_node_sequence: 2, + predecessor: control.root.clone().unwrap(), + final_txid: 10, + final_checksum: cellule_ltx::types::CHECKSUM_FLAG | 11, + final_commit_sequence: 10, + }) + .unwrap(); + } + // Pure tests model a retained record's shape. Canonical capability creation + // and actual takeover are exercised through public host/runtime scenarios. + let basis = RecoveryBasis { + spec: spec.clone(), + scope: head.scope(), + action_key: accepted.action().key().unwrap(), + node: spec.destination_node, + session: spec.destination, + accepted_at_ms: 0, + control, + observed_at_ms: 0, + }; + basis.validate_acceptance(&accepted).unwrap(); + let mut restored = basis + .control() + .takeover(Owner { + session: spec.destination, + endpoint: "https://successor.internal:8789".into(), + }) + .unwrap(); + if let Some(overlay) = restored.recovery.take() { + restored.revision += 1; + restored.progress += 1; + restored.root = Some(crate::control::RootRef { + digest: Digest::from_bytes([37; 32]), + txid: overlay.final_txid, + checksum: overlay.final_checksum, + commit_sequence: overlay.final_commit_sequence, + }); + } + let recovery = RecoveryEvidence::new(basis, restored, 0).unwrap(); + let serving = ActivationEvidence { + node: spec.destination_node, + session: spec.destination, + position: recovery.position().unwrap(), + }; + (head, accepted, RecoveredActivation { recovery, serving }) +} + +#[test] +fn recovered_history_is_distinct_and_charged_until_independent_credit_settlement() { + let (head, accepted, evidence) = recovered(true, false); + let id = evidence.recovery.basis().spec().id; + let pending = attempt(&head, id, AttemptEvent::OutcomeUnknown); + assert_eq!(pending.reserved_restore_bytes(), spec(1).cost.disk_bytes); + rejected(&pending, id, AttemptEvent::Cancelled); + rejected(&pending, id, AttemptEvent::Released(release())); + let done = attempt( + &pending, + id, + AttemptEvent::Recovered(Box::new(evidence.clone())), + ); + assert_eq!(done.attempts()[0].phase(), AttemptPhase::Recovered); + assert!(done.attempts()[0].released().is_none() && done.attempts()[0].activated().is_none()); + assert_eq!(done.attempts()[0].next_action(), MovementAction::Cancel); + assert!(done.retirement_page(&[id]).is_err()); + let duplicate = attempt( + &done, + id, + AttemptEvent::Recovered(Box::new(evidence.clone())), + ); + assert_eq!(duplicate, done); + let done = attempt(&done, id, AttemptEvent::ReceiverCleaned); + let page = done.retirement_page(&[id]).unwrap(); + assert_eq!( + ProgressPage::from_bytes(&page.to_bytes().unwrap()).unwrap(), + page + ); + let retired = transition( + &done, + JournalTransition::Retire { + progress: page.clone(), + }, + ); + assert!(retired.attempts().is_empty()); + assert!(page.entries()[0].recovered().is_some()); + let result = FleetActionOutcome { + scope: head.scope(), + action_key: accepted.action().key().unwrap(), + node: spec(1).destination_node, + session: spec(1).destination, + observed_at_ms: 0, + outcome: FleetOutcome::Recovered(Box::new(evidence)), + }; + accepted.validate_result(&result).unwrap(); + assert_eq!( + FleetActionOutcome::from_bytes(&result.to_bytes().unwrap()).unwrap(), + result + ); +} + +#[test] +fn recovery_records_require_the_exact_canonical_takeover_and_overlay_position() { + for overlay in [false, true] { + let (_, _, evidence) = recovered(overlay, false); + let basis = evidence.recovery.basis(); + for mutation in 0..8 { + let mut restored = evidence.recovery.restored().clone(); + match mutation { + 0 => restored.epoch += 1, + 1 => restored.revision += 1, + 2 => restored.state = ControlState::Serving, + 3 => restored.owner.as_mut().unwrap().session = spec(1).source, + 4 => restored.root.as_mut().unwrap().txid += 1, + 5 => restored.root.as_mut().unwrap().commit_sequence += 1, + 6 => restored.root.as_mut().unwrap().checksum ^= 1, + _ => restored.incarnation = IncarnationId::from_bytes([99; 16]), + } + assert!(RecoveryEvidence::new(basis.clone(), restored, 0).is_err()); + } + let mut before = basis.clone(); + before.control.epoch += 1; + assert!(before.to_bytes().is_err()); + let mut before = basis.clone(); + before.action_key = Digest::from_bytes([99; 32]); + assert!(before.to_bytes().is_err()); + let mut before = basis.clone(); + before.control.owner.as_mut().unwrap().session = spec(1).destination; + assert!(before.to_bytes().is_err()); + } +} + +#[test] +fn idle_source_input_is_recovery_without_a_clean_release_claim() { + let (head, _, evidence) = recovered(false, true); + let id = spec(1).id; + let done = attempt(&head, id, AttemptEvent::Recovered(Box::new(evidence))); + assert!(done.attempts()[0].released().is_none()); + assert_eq!( + done.attempts()[0] + .recovered() + .unwrap() + .recovery + .basis() + .control() + .state, + ControlState::Idle + ); + assert_eq!( + FleetHead::from_bytes(&done.to_bytes().unwrap()).unwrap(), + done + ); +} + +#[test] +fn recovery_cannot_substitute_a_newer_root_or_foreign_fleet_for_retained_evidence() { + let (head, accepted, original) = recovered(true, false); + let mut changed = original.clone(); + changed.serving.position.root.digest = Digest::from_bytes([99; 32]); + rejected( + &head, + spec(1).id, + AttemptEvent::Recovered(Box::new(changed)), + ); + let mut advanced = original.clone(); + advanced.serving.position.root.digest = Digest::from_bytes([99; 32]); + advanced.serving.position.root.txid += 1; + advanced.serving.position.root.commit_sequence += 1; + let done = attempt( + &head, + spec(1).id, + AttemptEvent::Recovered(Box::new(advanced)), + ); + assert_eq!( + done.attempts()[0].recovered().unwrap().recovery, + original.recovery + ); + let mut foreign = original; + foreign.recovery.basis.scope.fleet = Digest::from_bytes([99; 32]); + foreign.recovery.basis.action_key = super::super::actions::movement_key( + foreign.recovery.basis.scope, + MovementAction::Recover, + &spec(1), + ); + rejected( + &head, + spec(1).id, + AttemptEvent::Recovered(Box::new(foreign.clone())), + ); + assert!( + foreign + .recovery + .basis + .validate_acceptance(&accepted) + .is_err() + ); + let mut early = done.attempts()[0].clone(); + early.recovered.as_mut().unwrap().recovery.recorded_at_ms = 1; + assert!(early.to_bytes().is_err()); +} + +#[test] +fn recovery_codecs_reject_truncation_extra_wrong_kind_version_and_oversize() { + let (head, _, evidence) = recovered(true, false); + let basis = evidence.recovery.basis(); + let basis_bytes = basis.to_bytes().unwrap(); + let evidence_bytes = evidence.recovery.to_bytes().unwrap(); + assert_eq!(RecoveryBasis::from_bytes(&basis_bytes).unwrap(), *basis); + assert_eq!( + RecoveryEvidence::from_bytes(&evidence_bytes).unwrap(), + evidence.recovery + ); + for end in 0..basis_bytes.len() { + assert!(RecoveryBasis::from_bytes(&basis_bytes[..end]).is_err()); + } + for end in 0..evidence_bytes.len() { + assert!(RecoveryEvidence::from_bytes(&evidence_bytes[..end]).is_err()); + } + let mut trailing = basis_bytes.clone(); + trailing.push(0); + assert!(RecoveryBasis::from_bytes(&trailing).is_err()); + let mut trailing = evidence_bytes.clone(); + trailing.push(0); + assert!(RecoveryEvidence::from_bytes(&trailing).is_err()); + let mut version = basis_bytes; + version[4 + b"cellule.fleet-operation\0".len()] = 255; + assert!(RecoveryBasis::from_bytes(&version).is_err()); + assert!(RecoveryBasis::from_bytes(&evidence_bytes).is_err()); + assert!(RecoveryEvidence::from_bytes(&head.to_bytes().unwrap()).is_err()); + assert!(RecoveryBasis::from_bytes(&vec![0; MAX_RECORD_BYTES as usize + 1]).is_err()); + assert!(RecoveryEvidence::from_bytes(&vec![0; MAX_RECORD_BYTES as usize + 1]).is_err()); + let action = head + .movement_action(spec(1).id, MovementAction::Recover, 0) + .unwrap(); + assert_eq!( + FleetAction::from_bytes(&action.to_bytes().unwrap()).unwrap(), + action + ); +} diff --git a/crates/cellule-runtime/src/fleet/operations/tests/registry.rs b/crates/cellule-runtime/src/fleet/operations/tests/registry.rs new file mode 100644 index 00000000..2fea99c3 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/tests/registry.rs @@ -0,0 +1,655 @@ +use super::*; +use crate::codec::CodecError; +use crate::node::NodeMode; + +fn endpoint(byte: u8) -> EnrollmentEndpoint { + EnrollmentEndpoint { + node: NodeId::from_bytes([byte; 16]), + session: SessionId::from_bytes([byte + 10; 16]), + intent_revision: 1, + } +} + +#[test] +fn unexecuted_refusal_is_a_terminal_exclusion_row_without_role_admission() { + let spec = enrollment(EnrollmentRole::Follower { log_epoch: 7 }); + let proof = Digest::from_bytes([42; 32]); + let record = EnrollmentRecord::unexecuted_refusal(spec.clone(), proof, 10).unwrap(); + assert_eq!(record.status(), EnrollmentStatus::Refused); + assert!(!record.unresolved()); + record.validate_replay(&spec).unwrap(); + assert_eq!( + EnrollmentRecord::from_bytes(&record.to_bytes().unwrap()).unwrap(), + record + ); + assert_eq!(record.refuse(proof, 20).unwrap(), record); + assert!(record.establish(proof, 20).is_err()); + assert!(record.retire(proof, 20).is_err()); + assert!( + EnrollmentRecord::unexecuted_refusal(spec.clone(), Digest::from_bytes([0; 32]), 10) + .is_err() + ); + assert!(EnrollmentRecord::unexecuted_refusal(spec, proof, -1).is_err()); +} + +fn intent(byte: u8) -> NodeIntent { + NodeIntent::initial(head().scope(), endpoint(byte).node, endpoint(byte).session).unwrap() +} + +fn enrollment(role: EnrollmentRole) -> EnrollmentSpec { + EnrollmentSpec { + scope: head().scope(), + request: Digest::from_bytes([20; 32]), + source: (!matches!(role, EnrollmentRole::Node { .. })).then_some(endpoint(1)), + target: endpoint(2), + role, + } +} + +fn pending(role: EnrollmentRole) -> EnrollmentRecord { + let spec = enrollment(role); + EnrollmentRecord::pending( + spec.clone(), + spec.source.map(|_| intent(1)).as_ref(), + &intent(2), + 10, + ) + .unwrap() +} + +fn bootstrapped() -> RegistryVersion { + RegistryVersion::new(head().scope()) + .unwrap() + .advance(0) + .unwrap() + .bootstrap(1) + .unwrap() +} + +fn maintenance_for(node: u8, revision: u64) -> MaintenanceOperation { + let mut operation = maintenance(); + operation.node = endpoint(node).node; + operation.session = endpoint(node).session; + operation.intent_revision = revision; + operation +} + +#[test] +fn retained_intent_revision_blocks_reboot_reopening_and_old_enrollment_checks() { + let active = intent(2); + let mut operation = maintenance_for(2, 2); + let draining = active.advance_maintenance(&operation).unwrap(); + assert_eq!(draining.mode(), NodeMode::Draining); + assert_eq!(draining.advance_maintenance(&operation).unwrap(), draining); + assert!( + draining + .rebind_active(SessionId::from_bytes([99; 16]), 3) + .is_err() + ); + assert!(active.advance_maintenance(&maintenance_for(2, 1)).is_err()); + assert!(active.advance_maintenance(&maintenance_for(1, 2)).is_err()); + let mut foreign = operation.clone(); + foreign.id = operation_id(9); + foreign.intent_revision = 3; + assert!(draining.advance_maintenance(&foreign).is_err()); + + operation + .apply( + MaintenanceEvent::SessionReplaced(SessionId::from_bytes([99; 16])), + 1, + ) + .unwrap(); + assert_eq!(operation.intent_revision(), 3); + let rebound = draining.advance_maintenance(&operation).unwrap(); + assert_eq!(rebound.node(), draining.node()); + assert_eq!(rebound.mode(), NodeMode::Draining); + assert_eq!(rebound.session(), operation.session()); + assert_eq!(rebound.revision(), 3); + assert_eq!( + NodeIntent::from_bytes(&rebound.to_bytes().unwrap()).unwrap(), + rebound + ); + let mut old_revision = operation.clone(); + old_revision.intent_revision = 2; + assert!(rebound.advance_maintenance(&old_revision).is_err()); + let mut overflow = operation; + overflow.intent_revision = u64::MAX; + assert!( + overflow + .apply( + MaintenanceEvent::SessionReplaced(SessionId::from_bytes([98; 16])), + 2 + ) + .is_err() + ); + let active_boot = active + .rebind_active(SessionId::from_bytes([97; 16]), 2) + .unwrap(); + assert_eq!(active_boot.mode(), NodeMode::Active); + assert!(active.rebind_active(active.session(), 2).is_err()); + assert!(active.rebind_active(active_boot.session(), 1).is_err()); +} + +#[test] +fn pending_enrollment_checks_exact_committed_intents_and_allows_draining_donor() { + let source = intent(1) + .advance_maintenance(&maintenance_for(1, 2)) + .unwrap(); + let mut spec = enrollment(EnrollmentRole::Follower { log_epoch: 7 }); + assert!(EnrollmentRecord::pending(spec.clone(), Some(&source), &intent(2), 10).is_err()); + spec.source.as_mut().unwrap().intent_revision = source.revision(); + let admitted = EnrollmentRecord::pending(spec.clone(), Some(&source), &intent(2), 10).unwrap(); + assert!(admitted.unresolved()); + assert_eq!(admitted.status(), EnrollmentStatus::Pending); + let draining_target = intent(2) + .advance_maintenance(&maintenance_for(2, 2)) + .unwrap(); + spec.target.intent_revision = draining_target.revision(); + assert!(EnrollmentRecord::pending(spec.clone(), Some(&source), &draining_target, 10).is_err()); + let cordoned_target = NodeIntent { + mode: NodeMode::Cordoned, + ..draining_target + }; + assert!(EnrollmentRecord::pending(spec.clone(), Some(&source), &cordoned_target, 10).is_err()); + let mut boot = enrollment(EnrollmentRole::Node { + mode: NodeMode::Cordoned, + }); + boot.target.intent_revision = cordoned_target.revision(); + let enrolled_boot = + EnrollmentRecord::pending(boot.clone(), None, &cordoned_target, 10).unwrap(); + assert!(enrolled_boot.unresolved()); + boot.role = EnrollmentRole::Node { + mode: NodeMode::Active, + }; + assert!(EnrollmentRecord::pending(boot, None, &cordoned_target, 10).is_err()); + assert!(EnrollmentRecord::pending(spec.clone(), None, &intent(2), 10).is_err()); + let foreign = NodeIntent { + scope: FleetScope { + fleet: Digest::from_bytes([99; 32]), + ..source.scope() + }, + ..source.clone() + }; + assert!(EnrollmentRecord::pending(spec, Some(&foreign), &intent(2), 10).is_err()); + assert!( + EnrollmentRecord::pending( + enrollment(EnrollmentRole::Node { + mode: NodeMode::Active + }), + Some(&intent(1)), + &intent(2), + 10 + ) + .is_err() + ); + assert!( + EnrollmentRecord::pending( + enrollment(EnrollmentRole::Node { + mode: NodeMode::Active + }), + None, + &intent(2), + -1 + ) + .is_err() + ); +} + +#[test] +fn duplicate_enrollment_compares_full_inputs_and_preserves_original_progress_time() { + let original = pending(EnrollmentRole::Follower { log_epoch: 7 }); + let complete = original + .establish(Digest::from_bytes([31; 32]), 20) + .unwrap(); + complete.validate_replay(original.spec()).unwrap(); + assert_eq!( + complete + .establish(Digest::from_bytes([31; 32]), 100) + .unwrap(), + complete + ); + assert_eq!(complete.updated_at_ms(), 20); + assert_eq!(complete.accepted_at_ms(), 10); + assert!( + complete + .establish(Digest::from_bytes([32; 32]), 100) + .is_err() + ); + assert!(complete.refuse(Digest::from_bytes([33; 32]), 100).is_err()); + for change in 0..5 { + let mut spec = original.spec().clone(); + match change { + 0 => spec.role = EnrollmentRole::Follower { log_epoch: 8 }, + 1 => spec.source.as_mut().unwrap().intent_revision += 1, + 2 => spec.source.as_mut().unwrap().session = SessionId::from_bytes([90; 16]), + 3 => spec.target.node = NodeId::from_bytes([91; 16]), + _ => spec.target.intent_revision += 1, + } + assert_eq!(spec.key().unwrap(), original.spec().key().unwrap()); + assert!(original.validate_replay(&spec).is_err()); + } + let retired = complete.retire(Digest::from_bytes([34; 32]), 30).unwrap(); + assert_eq!( + retired.established_evidence(), + complete.established_evidence() + ); + assert!(!retired.unresolved()); + assert_eq!( + retired.retire(Digest::from_bytes([34; 32]), 500).unwrap(), + retired + ); + assert!( + retired + .establish(Digest::from_bytes([31; 32]), 500) + .is_err() + ); + assert!(retired.retire(Digest::from_bytes([35; 32]), 500).is_err()); +} + +#[test] +fn unknown_enrollment_requires_canonical_settlement_and_refusal_cannot_reopen() { + let unknown = pending(EnrollmentRole::Node { + mode: NodeMode::Active, + }); + assert_eq!( + unknown + .apply( + EnrollmentEvent::Established(Digest::from_bytes([43; 32])), + 20 + ) + .unwrap(), + unknown.establish(Digest::from_bytes([43; 32]), 20).unwrap() + ); + assert!(unknown.unresolved()); + assert!( + unknown + .retire(Digest::from_bytes([0; 32]), 500_000) + .is_err() + ); + assert_eq!(unknown.status(), EnrollmentStatus::Pending); + let closed = unknown + .retire(Digest::from_bytes([40; 32]), 500_000) + .unwrap(); + assert_eq!(closed.status(), EnrollmentStatus::Retired); + assert_eq!(closed.established_evidence(), None); + assert_eq!(closed.accepted_at_ms(), 10); + let refused = unknown.refuse(Digest::from_bytes([41; 32]), 15).unwrap(); + assert!(!refused.unresolved()); + assert_eq!( + refused.refuse(Digest::from_bytes([41; 32]), 30).unwrap(), + refused + ); + assert!(refused.retire(Digest::from_bytes([42; 32]), 30).is_err()); + assert!(unknown.establish(Digest::from_bytes([43; 32]), 9).is_err()); + let mut fabricated = unknown.clone(); + fabricated.status = EnrollmentStatus::Established; + assert!(fabricated.to_bytes().is_err()); + fabricated = closed; + fabricated.settlement = None; + assert!(fabricated.to_bytes().is_err()); +} + +#[test] +fn registry_barrier_is_explicit_stable_and_retained_after_restart() { + let initial = RegistryVersion::new(head().scope()).unwrap(); + assert!(initial.confirm(initial).is_err()); + assert!(initial.advance(1).is_err()); + let complete = initial.bootstrap(0).unwrap(); + assert_eq!(complete.bootstrap_revision(), Some(1)); + complete.confirm(complete).unwrap(); + assert_eq!(complete.bootstrap(1).unwrap(), complete); + let changed = complete.advance(1).unwrap(); + assert_eq!(changed.bootstrap_revision(), Some(1)); + assert!(complete.confirm(changed).is_err()); + assert!(changed.confirm(complete).is_err()); + let restored = RegistryVersion::from_bytes(&changed.to_bytes().unwrap()).unwrap(); + assert_eq!(restored, changed); + assert!(restored.advance(1).is_err()); + let future = RegistryVersion { + bootstrap_revision: Some(3), + ..restored + }; + assert!(future.to_bytes().is_err()); + let overflow = RegistryVersion { + revision: u64::MAX, + ..restored + }; + assert!(overflow.advance(u64::MAX).is_err()); + assert_eq!(overflow.bootstrap(u64::MAX).unwrap(), overflow); +} + +#[test] +fn scheduling_stop_resume_is_revision_checked_and_preserves_coverage() { + let empty = RegistryVersion::new(head().scope()).unwrap(); + assert!(!empty.scheduling_enabled()); + assert!(empty.set_scheduling(0, true).is_err()); + let covered = empty.bootstrap(0).unwrap(); + let running = covered.set_scheduling(covered.revision(), true).unwrap(); + assert!(running.scheduling_enabled()); + assert_eq!( + running.set_scheduling(running.revision(), true).unwrap(), + running + ); + assert!(running.set_scheduling(covered.revision(), false).is_err()); + let stopped = running.set_scheduling(running.revision(), false).unwrap(); + assert!(!stopped.scheduling_enabled()); + assert_eq!(stopped.bootstrap_revision(), running.bootstrap_revision()); + assert_eq!( + RegistryVersion::from_bytes(&stopped.to_bytes().unwrap()).unwrap(), + stopped + ); + let resumed = stopped.set_scheduling(stopped.revision(), true).unwrap(); + assert_eq!(resumed.bootstrap_revision(), covered.bootstrap_revision()); + assert_eq!(resumed.revision(), stopped.revision() + 1); +} + +#[test] +fn allocation_checks_retained_cordons_and_stop_preserves_charged_unknown_attempts() { + let running = bootstrapped().set_scheduling(2, true).unwrap(); + let current = head(); + running + .authorize_allocation(¤t, &spec(1), &intent(1), &intent(2)) + .unwrap(); + let cordon = intent(2) + .advance_maintenance(&maintenance_for(2, 2)) + .unwrap(); + assert!( + running + .authorize_allocation(¤t, &spec(1), &intent(1), &cordon) + .is_err() + ); + let changed_boot = intent(2) + .rebind_active(SessionId::from_bytes([99; 16]), 2) + .unwrap(); + assert!( + running + .authorize_allocation(¤t, &spec(1), &intent(1), &changed_boot) + .is_err() + ); + let draining_source = intent(1) + .advance_maintenance(&maintenance_for(1, 2)) + .unwrap(); + assert!( + running + .authorize_allocation(¤t, &spec(1), &draining_source, &intent(2)) + .is_err() + ); + let maintenance_head = transition( + ¤t, + JournalTransition::BeginMaintenance(maintenance_for(1, 2)), + ); + assert!( + running + .authorize_allocation(&maintenance_head, &spec(1), &draining_source, &intent(2)) + .is_err() + ); + let maintenance_head = transition( + &maintenance_head, + JournalTransition::Maintenance(MaintenanceEvent::Cordoned), + ); + let maintenance_head = transition( + &maintenance_head, + JournalTransition::Maintenance(MaintenanceEvent::BeginEvacuation), + ); + running + .authorize_allocation(&maintenance_head, &spec(1), &draining_source, &intent(2)) + .unwrap(); + let allocated = transition(&maintenance_head, JournalTransition::Allocate(spec(1))); + let preparing = attempt(&allocated, spec(1).id, AttemptEvent::BeginPrepare); + let unknown = attempt(&preparing, spec(1).id, AttemptEvent::OutcomeUnknown); + let stopped = running.set_scheduling(running.revision(), false).unwrap(); + assert!(matches!( + stopped.authorize_allocation(&unknown, &spec(2), &draining_source, &intent(2)), + Err(OperationError::Stopped) + )); + assert_eq!(unknown.attempts().len(), 1); + assert_eq!(unknown.reserved_restore_bytes(), spec(1).cost.disk_bytes); + let cancelling = attempt(&unknown, spec(1).id, AttemptEvent::BeginCancel); + assert_eq!( + cancelling.reserved_restore_bytes(), + unknown.reserved_restore_bytes() + ); + assert_eq!( + cancelling.node_intent().unwrap().unwrap().mode(), + NodeMode::Draining + ); + let mut foreign_spec = spec(1); + foreign_spec.id.operation = operation_id(99); + assert!( + running + .authorize_allocation( + &maintenance_head, + &foreign_spec, + &draining_source, + &intent(2) + ) + .is_err() + ); +} + +#[test] +fn bounded_pages_retain_older_cordons_and_failed_session_obligations() { + let version = bootstrapped(); + let old_cordon = intent(1) + .advance_maintenance(&maintenance_for(1, 2)) + .unwrap(); + let page = IntentPage::new( + version, + None, + vec![old_cordon.clone(), intent(2)], + Some(endpoint(2).node), + ) + .unwrap(); + assert_eq!( + IntentPage::from_bytes(&page.to_bytes().unwrap()).unwrap(), + page + ); + assert_eq!(page.entries()[0], old_cordon); + assert_eq!(page.next(), Some(endpoint(2).node)); + let last = IntentPage::new(version, page.next(), vec![intent(3)], None).unwrap(); + assert_eq!( + IntentPage::from_bytes(&last.to_bytes().unwrap()).unwrap(), + last + ); + assert!(IntentPage::new(version, None, vec![intent(2), intent(1)], None).is_err()); + assert!(IntentPage::new(version, None, vec![intent(1), intent(1)], None).is_err()); + assert!(IntentPage::new(version, None, vec![intent(1)], Some(endpoint(2).node)).is_err()); + assert!(IntentPage::new(version, Some(endpoint(2).node), vec![intent(1)], None).is_err()); + assert!(IntentPage::new(version, page.next(), vec![], None).is_err()); + assert!(IntentPage::new(version, None, vec![intent(1); MAX_PAGE_ENTRIES + 1], None).is_err()); + let foreign = RegistryVersion::new(FleetScope { + fleet: Digest::from_bytes([99; 32]), + ..version.scope() + }) + .unwrap(); + assert!(IntentPage::new(foreign, None, vec![intent(1)], None).is_err()); + let unknown = pending(EnrollmentRole::Follower { log_epoch: 7 }); + let retired = pending(EnrollmentRole::Node { + mode: NodeMode::Active, + }) + .retire(Digest::from_bytes([60; 32]), 100) + .unwrap(); + let mut retired = retired; + retired.spec.request = Digest::from_bytes([61; 32]); + let mut records = vec![unknown, retired]; + records.sort_by_key(|record| *record.spec().key().unwrap().as_bytes()); + let last_key = records.last().unwrap().spec().key().unwrap(); + let page = EnrollmentPage::new(version, None, records.clone(), Some(last_key)).unwrap(); + assert_eq!( + EnrollmentPage::from_bytes(&page.to_bytes().unwrap()).unwrap(), + page + ); + assert_eq!( + page.entries() + .iter() + .filter(|record| record.unresolved()) + .count(), + 1 + ); + assert!( + EnrollmentPage::new( + version, + None, + vec![records[0].clone(), records[0].clone()], + None + ) + .is_err() + ); + records.reverse(); + assert!(EnrollmentPage::new(version, None, records, None).is_err()); + assert!(EnrollmentPage::new(version, Some(last_key), vec![], None).is_err()); + assert!(EnrollmentPage::new(version, None, vec![], Some(last_key)).is_err()); + assert!(EnrollmentPage::new(foreign, None, page.entries().to_vec(), None).is_err()); + assert!(IntentPage::new(version, None, vec![], None).is_ok()); + assert!(EnrollmentPage::new(version, None, vec![], None).is_ok()); + let empty_version = RegistryVersion::new(version.scope()).unwrap(); + assert!(IntentPage::new(empty_version, None, vec![intent(1)], None).is_err()); + assert!(EnrollmentPage::new(empty_version, None, page.entries().to_vec(), None).is_err()); +} + +#[test] +fn enrollment_roles_and_all_progress_states_round_trip_with_exact_inputs() { + for role in [ + EnrollmentRole::Node { + mode: NodeMode::Active, + }, + EnrollmentRole::Reader { + target: spec(1).target, + position: release(), + }, + EnrollmentRole::Follower { log_epoch: 7 }, + ] { + let record = pending(role); + assert_eq!( + EnrollmentSpec::from_bytes(&record.spec().to_bytes().unwrap()).unwrap(), + *record.spec() + ); + for state in [ + record.clone(), + record.establish(Digest::from_bytes([70; 32]), 20).unwrap(), + record.refuse(Digest::from_bytes([71; 32]), 20).unwrap(), + record.retire(Digest::from_bytes([72; 32]), 20).unwrap(), + record + .establish(Digest::from_bytes([73; 32]), 20) + .unwrap() + .retire(Digest::from_bytes([74; 32]), 30) + .unwrap(), + ] { + assert_eq!( + EnrollmentRecord::from_bytes(&state.to_bytes().unwrap()).unwrap(), + state + ); + } + } + let mut invalid = enrollment(EnrollmentRole::Node { + mode: NodeMode::Active, + }); + invalid.source = Some(endpoint(1)); + assert!(invalid.key().is_err()); + invalid = enrollment(EnrollmentRole::Follower { log_epoch: 0 }); + assert!(invalid.key().is_err()); + invalid = enrollment(EnrollmentRole::Follower { log_epoch: 1 }); + invalid.source = None; + assert!(invalid.key().is_err()); + invalid = enrollment(EnrollmentRole::Follower { log_epoch: 1 }); + invalid.source = Some(invalid.target); + assert!(invalid.key().is_err()); + invalid = enrollment(EnrollmentRole::Reader { + target: spec(1).target, + position: release(), + }); + invalid.scope.application = ApplicationId::from_bytes([99; 16]); + assert!(invalid.key().is_err()); +} + +fn malformed(bytes: &[u8], decode: fn(&[u8]) -> Result<()>, limit: u32) { + for end in 0..bytes.len() { + assert!(decode(&bytes[..end]).is_err()); + } + let mut trailing = bytes.to_vec(); + trailing.push(0); + assert!(decode(&trailing).is_err()); + let mut future = bytes.to_vec(); + future[4 + b"cellule.fleet-operation\0".len()] = FORMAT_VERSION + 1; + assert!(decode(&future).is_err()); + let mut unknown_kind = bytes.to_vec(); + unknown_kind[5 + b"cellule.fleet-operation\0".len()] = 255; + assert!(decode(&unknown_kind).is_err()); + assert!(decode(&vec![0; limit as usize + 1]).is_err()); +} + +#[test] +fn registry_codecs_reject_partial_future_wrong_type_oversize_and_unbounded_counts() { + let version = bootstrapped(); + malformed( + &version.to_bytes().unwrap(), + |b| RegistryVersion::from_bytes(b).map(|_| ()), + MAX_RECORD_BYTES, + ); + let record = pending(EnrollmentRole::Reader { + target: spec(1).target, + position: release(), + }); + malformed( + &record.to_bytes().unwrap(), + |b| EnrollmentRecord::from_bytes(b).map(|_| ()), + MAX_RECORD_BYTES, + ); + let intents = IntentPage::new(version, None, vec![intent(1)], None) + .unwrap() + .to_bytes() + .unwrap(); + malformed( + &intents, + |b| IntentPage::from_bytes(b).map(|_| ()), + MAX_PAGE_BYTES, + ); + let enrollments = EnrollmentPage::new(version, None, vec![record], None) + .unwrap() + .to_bytes() + .unwrap(); + malformed( + &enrollments, + |b| EnrollmentPage::from_bytes(b).map(|_| ()), + MAX_PAGE_BYTES, + ); + malformed( + &pending(EnrollmentRole::Node { + mode: NodeMode::Active, + }) + .spec() + .to_bytes() + .unwrap(), + |b| EnrollmentSpec::from_bytes(b).map(|_| ()), + MAX_RECORD_BYTES, + ); + // Fixed envelope, scope, version, bootstrap marker/revision, policy, absent after. + let count_offset = + 4 + b"cellule.fleet-operation\0".len() + 2 + 4 + 32 + 4 + 16 + 8 + 1 + 8 + 1 + 1; + for mut page in [intents, enrollments] { + page[count_offset..count_offset + 4].copy_from_slice(&u32::MAX.to_be_bytes()); + assert!( + matches!( + IntentPage::from_bytes(&page), + Err(OperationError::Codec(CodecError::Limit)) + ) || matches!( + EnrollmentPage::from_bytes(&page), + Err(OperationError::Codec(CodecError::Limit)) + ) + ); + } + assert!(IntentPage::from_bytes(&version.to_bytes().unwrap()).is_err()); + assert!(EnrollmentRecord::from_bytes(&version.to_bytes().unwrap()).is_err()); + assert!(EnrollmentPage::from_bytes(&version.to_bytes().unwrap()).is_err()); + let boot = pending(EnrollmentRole::Node { + mode: NodeMode::Active, + }); + let role_offset = 4 + b"cellule.fleet-operation\0".len() + 2 + 4 + 32 + 4 + 16 + 4 + 32; + let mut unknown_role = boot.spec().to_bytes().unwrap(); + unknown_role[role_offset] = 255; + assert!(EnrollmentSpec::from_bytes(&unknown_role).is_err()); + let mut unknown_mode = boot.to_bytes().unwrap(); + unknown_mode[role_offset + 1] = 255; + assert!(EnrollmentRecord::from_bytes(&unknown_mode).is_err()); + assert!(EnrollmentSpec::from_bytes(&boot.to_bytes().unwrap()).is_err()); +} diff --git a/crates/cellule-runtime/src/fleet/operations/writer_inventory/mod.rs b/crates/cellule-runtime/src/fleet/operations/writer_inventory/mod.rs new file mode 100644 index 00000000..46890360 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/writer_inventory/mod.rs @@ -0,0 +1,248 @@ +//! Original failed-boot writer metadata, retained before dependent effects. +use super::*; +use crate::{ + control::Control, + identity::{ApplicationId, CellTarget, Digest, TenantId}, +}; + +mod validation; + +/// Hard bounds for complete original observations and authenticated catalog scopes. +pub const MAX_ORIGINAL_WRITERS: usize = 10_000; +/// Maximum canonical application/tenant sources in one complete capture. +pub const MAX_ORIGINAL_CATALOGS: usize = 128; +// A full control is bounded to 8 KiB and a target partition to 1 KiB. Sixty-four +// rows fit the ordinary one-MiB page even at those bounds, without truncation. +pub(super) const WRITERS_PER_PAGE: usize = 64; +pub(super) const MAX_WRITER_PAGES: usize = MAX_ORIGINAL_WRITERS.div_ceil(WRITERS_PER_PAGE); + +/// Complete original publication basis. Digests identify provider attestations; +/// constructing or decoding this metadata grants no authentication or authority. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct OriginalWriterInventoryBasis { + /// Operation as captured, without restamping or later session substitution. + pub operation: MaintenanceOperation, + /// Digest of the complete original controller head. + pub head_digest: Digest, + /// Original registry barrier. + pub registry: RegistryVersion, + /// Original enrolled physical boot, including first establishment history. + pub boot: EnrollmentRecord, + /// Immutable original process/fence request identity. + pub process_request: Digest, + /// Durable original process and accepted-work joining evidence. + pub process_witness: Digest, + /// Application-authenticated complete catalog source-set identity. + pub catalog_witness: Digest, + /// Original capture interval. + pub interval: (i64, i64), +} + +/// One fully traversed application/tenant scope. The history digest covers every +/// catalog target, including absence and owners outside the failed boot. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct OriginalCatalogWitness { + /// Canonical application identity. + pub application: ApplicationId, + /// Canonical tenant identity. + pub tenant: TenantId, + /// Application's durable identity of this canonical storage source. + pub source: Digest, + /// Digest of all 256 original revisions and ordered page identities. + pub heads: Digest, + /// Digest of all original per-Cell authority/history observations. + pub histories: Digest, + /// Complete catalog entry count, including unused bootstrap entries. + pub cells: u64, + /// Original boot's ownership observations retained from this scope. + pub owners: u64, +} + +/// Exact original owner observation. Rootless, recovering and object-covered +/// writers remain obligations; the complete Control includes every overlay. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct OriginalWriterObservation { + /// Complete tenant/application/namespace/partition binding. + pub target: CellTarget, + /// Original control, not reconstructed from a successor counter. + pub control: Control, +} + +/// Immutable bounded page of the complete original ownership set. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct OriginalWriterInventoryPage { + pub(super) basis: Digest, + pub(super) ordinal: u32, + pub(super) entries: Vec, +} +impl OriginalWriterInventoryPage { + /// Manifest basis shared by all original pages. + #[must_use] + pub const fn basis(&self) -> Digest { + self.basis + } + /// Zero-based complete-page ordinal. + #[must_use] + pub const fn ordinal(&self) -> u32 { + self.ordinal + } + /// Original owner observations in canonical scope/Cell/epoch order. + #[must_use] + pub fn entries(&self) -> &[OriginalWriterObservation] { + &self.entries + } + /// Canonical immutable page identity. + pub fn digest(&self) -> Result { + Ok(Digest::from_bytes( + *blake3::hash(&self.to_bytes()?).as_bytes(), + )) + } +} + +/// Durable original set; historical capture cannot prove current successor +/// serving, availability, acknowledged-prefix inclusion or maintenance completion. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct OriginalWriterInventoryRecord { + pub(super) basis: OriginalWriterInventoryBasis, + pub(super) catalogs: Vec, + pub(super) count: usize, + pub(super) pages: Vec, +} +impl OriginalWriterInventoryRecord { + /// Builds canonical bounded pages after validating complete metadata shape. + /// Providers still authenticate/collect the complete set before publication. + pub fn new( + basis: OriginalWriterInventoryBasis, + mut catalogs: Vec, + mut owners: Vec, + ) -> Result<(Self, Vec)> { + if owners.len() > MAX_ORIGINAL_WRITERS || catalogs.len() > MAX_ORIGINAL_CATALOGS { + return Err(OperationError::Budget); + } + catalogs.sort_by_key(|row| (*row.application.as_bytes(), *row.tenant.as_bytes())); + owners.sort_by_key(validation::key); + let mut record = Self { + basis, + catalogs, + count: owners.len(), + pages: Vec::new(), + }; + record.validate_basis()?; + let digest = record.basis_digest()?; + let pages: Vec<_> = owners + .chunks(WRITERS_PER_PAGE) + .enumerate() + .map(|(ordinal, entries)| OriginalWriterInventoryPage { + basis: digest, + ordinal: ordinal as u32, + entries: entries.to_vec(), + }) + .collect(); + record.pages = pages + .iter() + .map(OriginalWriterInventoryPage::digest) + .collect::>()?; + record.validate_pages(&pages)?; + Ok((record, pages)) + } + /// Original immutable basis; never a fresh settlement assertion. + #[must_use] + pub const fn basis(&self) -> &OriginalWriterInventoryBasis { + &self.basis + } + /// Complete original catalog scope observations. + #[must_use] + pub fn catalogs(&self) -> &[OriginalCatalogWitness] { + &self.catalogs + } + /// Complete original ownership count, including repeated epochs of one Cell. + #[must_use] + pub const fn owner_count(&self) -> usize { + self.count + } + /// Complete ordered original page identities. + #[must_use] + pub fn pages(&self) -> &[Digest] { + &self.pages + } + /// Immutable complete record identity. + pub fn digest(&self) -> Result { + Ok(Digest::from_bytes( + *blake3::hash(&self.to_bytes()?).as_bytes(), + )) + } + /// Checks every exact page, original boot, scope, ordinal and unique epoch. + pub fn validate_pages(&self, pages: &[OriginalWriterInventoryPage]) -> Result<()> { + self.validate()?; + if pages.len() != self.pages.len() { + return Err(OperationError::Conflict); + } + let basis = self.basis_digest()?; + let mut after = None; + let mut previous: Option<&OriginalWriterObservation> = None; + let mut counts = vec![0_u64; self.catalogs.len()]; + let mut total = 0; + for (ordinal, (page, expected)) in pages.iter().zip(&self.pages).enumerate() { + page.validate()?; + if page.basis != basis + || page.ordinal as usize != ordinal + || page.digest()? != *expected + || page.entries.len() != (self.count - total).min(WRITERS_PER_PAGE) + { + return Err(OperationError::Conflict); + } + for row in &page.entries { + let key = validation::key(row); + if after.is_some_and(|previous| previous >= key) + || row + .control + .owner + .as_ref() + .is_none_or(|owner| owner.session != self.basis.boot.spec().target.session) + { + return Err(OperationError::Conflict); + } + // One traversal observes one incarnation per Cell. Repeated + // original epochs, including across page boundaries, retain the + // same target and strictly increasing canonical history. + if previous.is_some_and(|prior| { + prior.target.application() == row.target.application() + && prior.target.tenant() == row.target.tenant() + && prior.control.cell == row.control.cell + && (prior.target != row.target + || prior.control.incarnation != row.control.incarnation + || prior.control.revision >= row.control.revision + || prior.control.progress >= row.control.progress) + }) { + return Err(OperationError::Conflict); + } + let index = self + .catalogs + .binary_search_by_key( + &( + *row.target.application().as_bytes(), + *row.target.tenant().as_bytes(), + ), + |scope| (*scope.application.as_bytes(), *scope.tenant.as_bytes()), + ) + .map_err(|_| OperationError::Conflict)?; + counts[index] += 1; + after = Some(key); + previous = Some(row); + total += 1; + } + } + if total != self.count + || counts + .iter() + .zip(&self.catalogs) + .any(|(count, scope)| *count != scope.owners) + { + return Err(OperationError::Conflict); + } + Ok(()) + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-runtime/src/fleet/operations/writer_inventory/tests.rs b/crates/cellule-runtime/src/fleet/operations/writer_inventory/tests.rs new file mode 100644 index 00000000..4d9eb856 --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/writer_inventory/tests.rs @@ -0,0 +1,278 @@ +use super::*; +use crate::{ + control::{ControlState, Owner}, + identity::{IncarnationId, NamespaceId, NodeId, SessionId}, + node::NodeMode, +}; + +fn fixture( + count: usize, +) -> ( + OriginalWriterInventoryRecord, + Vec, +) { + let scope = FleetScope { + fleet: Digest::from_bytes([9; 32]), + application: ApplicationId::from_bytes([3; 16]), + }; + let intent = NodeIntent::initial( + scope, + NodeId::from_bytes([1; 16]), + SessionId::from_bytes([2; 16]), + ) + .unwrap(); + let boot = EnrollmentRecord::pending( + EnrollmentSpec { + scope, + request: Digest::from_bytes([10; 32]), + source: None, + target: EnrollmentEndpoint { + node: intent.node(), + session: intent.session(), + intent_revision: 1, + }, + role: EnrollmentRole::Node { + mode: NodeMode::Active, + }, + }, + None, + &intent, + 10, + ) + .unwrap() + .establish(Digest::from_bytes([11; 32]), 20) + .unwrap(); + let basis = OriginalWriterInventoryBasis { + operation: MaintenanceOperation::new( + OperationId::from_bytes([12; 16]).unwrap(), + Digest::from_bytes([13; 32]), + intent.node(), + intent.session(), + 2, + 0, + 60_000, + ) + .unwrap(), + head_digest: Digest::from_bytes([14; 32]), + registry: RegistryVersion::new(scope).unwrap().bootstrap(0).unwrap(), + boot, + process_request: Digest::from_bytes([15; 32]), + process_witness: Digest::from_bytes([16; 32]), + catalog_witness: Digest::from_bytes([17; 32]), + interval: (30, 40), + }; + let tenant = TenantId::from_bytes([4; 16]); + let rows = (0..count) + .map(|n| { + let target = CellTarget::new( + tenant, + scope.application, + NamespaceId::from_bytes([5; 16]), + &(n as u64).to_be_bytes(), + ) + .unwrap(); + let control = Control::initial( + target.cell_id(), + IncarnationId::from_bytes([6; 16]), + Owner { + session: intent.session(), + endpoint: "https://original.example".into(), + }, + Digest::from_bytes([7; 32]), + 1, + ) + .unwrap(); + OriginalWriterObservation { target, control } + }) + .collect(); + OriginalWriterInventoryRecord::new( + basis, + vec![OriginalCatalogWitness { + application: scope.application, + tenant, + source: Digest::from_bytes([18; 32]), + heads: Digest::from_bytes([19; 32]), + histories: Digest::from_bytes([20; 32]), + cells: count as u64, + owners: count as u64, + }], + rows, + ) + .unwrap() +} +#[test] +fn original_writer_manifest_and_all_pages_roundtrip_full_rootless_controls() { + let (record, pages) = fixture(129); + assert_eq!( + pages.iter().map(|p| p.entries.len()).collect::>(), + [64, 64, 1] + ); + assert_eq!( + OriginalWriterInventoryRecord::from_bytes(&record.to_bytes().unwrap()).unwrap(), + record + ); + for page in &pages { + assert_eq!( + OriginalWriterInventoryPage::from_bytes(&page.to_bytes().unwrap()).unwrap(), + *page + ); + assert!( + page.entries + .iter() + .all(|row| row.control.root.is_none() + && row.control.state == ControlState::Recovering) + ); + } + record.validate_pages(&pages).unwrap(); +} +#[test] +fn empty_original_writer_set_is_explicit_complete_metadata() { + let (record, pages) = fixture(0); + assert_eq!(record.owner_count(), 0); + assert!(pages.is_empty()); + record.validate_pages(&pages).unwrap(); + let mut basis = record.basis.clone(); + basis.catalog_witness = Digest::from_bytes([0; 32]); + assert!(OriginalWriterInventoryRecord::new(basis, Vec::new(), Vec::new()).is_err()); +} +#[test] +fn incomplete_reordered_duplicate_and_foreign_writer_pages_refuse() { + let (record, pages) = fixture(65); + assert!(record.validate_pages(&pages[..1]).is_err()); + let mut changed = pages.clone(); + changed.reverse(); + assert!(record.validate_pages(&changed).is_err()); + let mut changed = pages.clone(); + changed[0].entries[1] = changed[0].entries[0].clone(); + assert!(record.validate_pages(&changed).is_err()); + let mut rows = pages + .into_iter() + .flat_map(|p| p.entries) + .collect::>(); + rows[0].control.owner.as_mut().unwrap().session = SessionId::from_bytes([99; 16]); + assert!(OriginalWriterInventoryRecord::new(record.basis, record.catalogs, rows).is_err()); +} +#[test] +fn original_inventory_rejects_noncanonical_scope_counts_intervals_and_limits() { + let (record, pages) = fixture(1); + let rows = pages[0].entries.clone(); + let mut catalogs = record.catalogs.clone(); + catalogs[0].owners = 0; + assert!( + OriginalWriterInventoryRecord::new(record.basis.clone(), catalogs, rows.clone()).is_err() + ); + let mut catalogs = record.catalogs.clone(); + catalogs.push(catalogs[0].clone()); + assert!( + OriginalWriterInventoryRecord::new(record.basis.clone(), catalogs, rows.clone()).is_err() + ); + let mut basis = record.basis.clone(); + basis.interval = (40, 30); + assert!( + OriginalWriterInventoryRecord::new(basis, record.catalogs.clone(), rows.clone()).is_err() + ); + let mut basis = record.basis.clone(); + basis.registry = RegistryVersion::new(basis.registry.scope()).unwrap(); + assert!( + OriginalWriterInventoryRecord::new(basis, record.catalogs.clone(), rows.clone()).is_err() + ); + assert!( + OriginalWriterInventoryRecord::new( + record.basis, + record.catalogs, + vec![rows[0].clone(); MAX_ORIGINAL_WRITERS + 1] + ) + .is_err() + ); +} +#[test] +fn original_inventory_decoders_refuse_every_truncation_trailing_unknown_and_oversize_body() { + let (record, pages) = fixture(1); + let body = record.to_bytes().unwrap(); + for length in 0..body.len() { + assert!(OriginalWriterInventoryRecord::from_bytes(&body[..length]).is_err()); + } + let mut changed = body.clone(); + changed.push(0); + assert!(OriginalWriterInventoryRecord::from_bytes(&changed).is_err()); + let mut changed = body; + changed[0] ^= 1; + assert!(OriginalWriterInventoryRecord::from_bytes(&changed).is_err()); + let body = pages[0].to_bytes().unwrap(); + for length in 0..body.len() { + assert!(OriginalWriterInventoryPage::from_bytes(&body[..length]).is_err()); + } + let mut changed = body; + changed.push(0); + assert!(OriginalWriterInventoryPage::from_bytes(&changed).is_err()); + assert!( + OriginalWriterInventoryPage::from_bytes(&vec![0; MAX_PAGE_BYTES as usize + 1]).is_err() + ); +} +#[test] +fn original_controls_bind_code_schema_roots_and_observation_target() { + let (record, pages) = fixture(1); + let mut rows = pages[0].entries.clone(); + let mut changed = rows.clone(); + changed[0].control.cell = crate::identity::CellId::from_bytes([88; 32]); + assert!( + OriginalWriterInventoryRecord::new(record.basis.clone(), record.catalogs.clone(), changed) + .is_err() + ); + rows[0].control.state = ControlState::Serving; + rows[0].control.root = Some(crate::control::RootRef { + digest: Digest::from_bytes([89; 32]), + txid: 1, + checksum: cellule_ltx::types::CHECKSUM_FLAG | 1, + commit_sequence: 1, + }); + rows[0].control.code = Digest::from_bytes([88; 32]); + rows[0].control.schema = 2; + let (later, _) = + OriginalWriterInventoryRecord::new(record.basis.clone(), record.catalogs.clone(), rows) + .unwrap(); + assert_ne!(later.digest().unwrap(), record.digest().unwrap()); + let encoded = later.to_bytes().unwrap(); + assert_eq!( + OriginalWriterInventoryRecord::from_bytes(&encoded).unwrap(), + later + ); +} + +#[test] +fn repeated_original_epochs_preserve_one_ordered_incarnation_across_pages() { + let (record, pages) = fixture(64); + let mut rows = pages[0].entries.clone(); + let original = rows.last().unwrap().clone(); + let mut later = original.clone(); + later.control = original + .control + .takeover(Owner { + session: SessionId::from_bytes([88; 16]), + endpoint: "https://intermediate.example".into(), + }) + .unwrap() + .takeover(original.control.owner.clone().unwrap()) + .unwrap(); + rows.push(later.clone()); + let mut catalogs = record.catalogs.clone(); + catalogs[0].owners = 65; + let (complete, pages) = + OriginalWriterInventoryRecord::new(record.basis.clone(), catalogs.clone(), rows.clone()) + .unwrap(); + assert_eq!(pages.len(), 2); + complete.validate_pages(&pages).unwrap(); + for change in 0..3 { + let mut changed = rows.clone(); + let control = &mut changed.last_mut().unwrap().control; + match change { + 0 => control.incarnation = IncarnationId::from_bytes([7; 16]), + 1 => control.revision = original.control.revision, + _ => control.progress = original.control.progress, + } + assert!( + OriginalWriterInventoryRecord::new(record.basis.clone(), catalogs.clone(), changed,) + .is_err() + ); + } +} diff --git a/crates/cellule-runtime/src/fleet/operations/writer_inventory/validation.rs b/crates/cellule-runtime/src/fleet/operations/writer_inventory/validation.rs new file mode 100644 index 00000000..c19b308c --- /dev/null +++ b/crates/cellule-runtime/src/fleet/operations/writer_inventory/validation.rs @@ -0,0 +1,127 @@ +use super::*; + +pub(super) fn key( + row: &OriginalWriterObservation, +) -> ([u8; 16], [u8; 16], [u8; 32], [u8; 16], u64) { + ( + *row.target.application().as_bytes(), + *row.target.tenant().as_bytes(), + *row.control.cell.as_bytes(), + *row.control.incarnation.as_bytes(), + row.control.epoch, + ) +} +impl OriginalWriterInventoryRecord { + pub(in crate::fleet::operations) fn validate_basis(&self) -> Result<()> { + let basis = &self.basis; + basis.operation.validate()?; + basis.registry.confirm(basis.registry)?; + basis.boot.validate_replay(basis.boot.spec())?; + if basis.registry.bootstrap_revision().is_none() + || basis.registry.scope() != basis.boot.spec().scope + || !matches!(basis.boot.spec().role, EnrollmentRole::Node { .. }) + || basis.boot.spec().source.is_some() + || basis.boot.established_evidence().is_none() + || !matches!( + basis.boot.status(), + EnrollmentStatus::Established | EnrollmentStatus::Retired + ) + || basis.boot.spec().target.node != basis.operation.node() + || basis.boot.spec().target.session != basis.operation.session() + || basis.operation.phase() == MaintenancePhase::Completed + || basis.interval.0 < basis.operation.created_at_ms + || basis.interval.0 < basis.boot.updated_at_ms() + || basis.interval.1 < basis.interval.0 + || basis.interval.1 - basis.interval.0 > 30_000 + || basis.interval.1 >= basis.operation.deadline_ms() + || [ + basis.head_digest, + basis.process_request, + basis.process_witness, + basis.catalog_witness, + ] + .iter() + .any(|digest| !nonzero(digest.as_bytes())) + || self.count > MAX_ORIGINAL_WRITERS + || self.catalogs.len() > MAX_ORIGINAL_CATALOGS + { + return Err(OperationError::Invalid( + "invalid original writer inventory basis", + )); + } + let mut after = None; + let mut owners = 0_u64; + let mut cells = 0_u64; + for row in &self.catalogs { + let key = (*row.application.as_bytes(), *row.tenant.as_bytes()); + if [row.source, row.heads, row.histories] + .iter() + .any(|digest| !nonzero(digest.as_bytes())) + || row.cells > 10_000 + || row.owners > MAX_ORIGINAL_WRITERS as u64 + || (row.cells == 0 && row.owners != 0) + || after.is_some_and(|previous| previous >= key) + { + return Err(OperationError::Invalid("invalid original catalog witness")); + } + owners += row.owners; + cells += row.cells; + after = Some(key); + } + if cells > MAX_ORIGINAL_WRITERS as u64 || owners != self.count as u64 { + return Err(OperationError::Conflict); + } + Ok(()) + } + pub(in crate::fleet::operations) fn validate(&self) -> Result<()> { + self.validate_basis()?; + if self.pages.len() != self.count.div_ceil(WRITERS_PER_PAGE) + || self.pages.len() > MAX_WRITER_PAGES + || self.pages.iter().any(|digest| !nonzero(digest.as_bytes())) + { + return Err(OperationError::Invalid( + "invalid original writer inventory pages", + )); + } + Ok(()) + } +} +impl OriginalWriterInventoryPage { + pub(in crate::fleet::operations) fn validate(&self) -> Result<()> { + if !nonzero(self.basis.as_bytes()) + || self.ordinal as usize >= MAX_WRITER_PAGES + || self.entries.is_empty() + || self.entries.len() > WRITERS_PER_PAGE + { + return Err(OperationError::Invalid( + "invalid original writer inventory page", + )); + } + let mut after = None; + for row in &self.entries { + let body = row + .control + .encode() + .map_err(|source| OperationError::Control(Box::new(source)))?; + let target = CellTarget::new( + row.target.tenant(), + row.target.application(), + row.target.namespace(), + row.target.partition(), + ) + .map_err(|source| OperationError::Identity(Box::new(source)))?; + let key = key(row); + if target.cell_id() != row.control.cell + || row.control.owner.is_none() + || body.len() > 8 * 1024 + || after.is_some_and(|previous| previous >= key) + { + return Err(OperationError::Invalid( + "invalid original writer observation", + )); + } + after = Some(key); + } + Ok(()) + } +} diff --git a/crates/cellule-runtime/src/fleet/placement/mod.rs b/crates/cellule-runtime/src/fleet/placement/mod.rs index dc39f6c6..21006c75 100644 --- a/crates/cellule-runtime/src/fleet/placement/mod.rs +++ b/crates/cellule-runtime/src/fleet/placement/mod.rs @@ -4,7 +4,7 @@ use std::collections::HashSet; use crate::identity::NodeId; use crate::identity::{CellId, SessionId}; -use crate::node::{NodeAdvertisement, NodePlacementCapacity}; +use crate::node::{NodeAdvertisement, NodeMode, NodePlacementCapacity, NodePressure}; use crate::{Error, Result}; const MAX_OBSERVATION_AGE_MS: i64 = 30_000; @@ -21,9 +21,8 @@ const BALANCE_DEADBAND_PERCENT: u128 = 2; /// Pressure class supplied by the signed node observation. /// -/// The signed placement block carries measured counters rather than the node's -/// hysteretic tier, so derivation currently reaches only the critical class; -/// soft and hard pressure remain local until the tier is signed. +/// Schema 3 carries the stable local tier explicitly. Schema 2 bridge readers +/// retain the original zero-headroom derivation until the producer rollout. #[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord)] pub enum PlacementPressure { /// No pressure: the node accepts placements. @@ -89,6 +88,7 @@ impl PlacementObservation { now_ms: i64, current_owner: bool, ) -> Result { + advertisement.verify_placement()?; let placement = advertisement .placement_capacity() .ok_or(Error::Node("placement snapshot is missing"))?; @@ -118,7 +118,9 @@ impl PlacementObservation { Self { node: advertisement.node(), session: advertisement.session(), - observed_at_ms: advertisement.issued_at_ms(), + observed_at_ms: advertisement + .operational_sample() + .map_or(advertisement.issued_at_ms(), |sample| sample.observed_at_ms), memory_capacity_bytes: placement.memory_capacity_bytes, free_memory_bytes: capacity.free_memory_bytes, disk_capacity_bytes: placement.disk_capacity_bytes, @@ -130,14 +132,25 @@ impl PlacementObservation { publication_backlog: placement.publication_backlog, hydration_backlog: placement.hydration_backlog, primitive_backlog: placement.primitive_backlog, - // Only an empty ledger is visible to a peer while the hysteretic - // tier stays local. - pressure: if capacity.free_memory_bytes == 0 || capacity.free_disk_bytes == 0 { - PlacementPressure::Critical - } else { - PlacementPressure::Normal - }, - draining: capacity.free_memory_bytes == 0 || capacity.free_disk_bytes == 0, + pressure: advertisement.operational_sample().map_or_else( + || { + if capacity.free_memory_bytes == 0 || capacity.free_disk_bytes == 0 { + PlacementPressure::Critical + } else { + PlacementPressure::Normal + } + }, + |sample| match sample.pressure { + NodePressure::Normal => PlacementPressure::Normal, + NodePressure::Constrained => PlacementPressure::Constrained, + NodePressure::Shedding => PlacementPressure::Shedding, + NodePressure::Critical => PlacementPressure::Critical, + }, + ), + draining: advertisement.operational_sample().map_or_else( + || capacity.free_memory_bytes == 0 || capacity.free_disk_bytes == 0, + |sample| sample.mode != NodeMode::Active, + ), authenticated: true, current_owner, } @@ -157,6 +170,8 @@ pub enum PlacementEligibility { Draining, /// The node reports critical pressure. CriticalPressure, + /// The stable node tier pauses proactive receive before critical pressure. + Pressure, /// Free memory is below the placement reserve. NoMemoryHeadroom, /// Free disk is below the placement reserve. @@ -180,8 +195,9 @@ pub struct PlacementScore { pub eligibility: PlacementEligibility, } -/// Actor-sampled demand for one locally owned Cell. A missing or unsettled -/// sample cannot be used as a transfer hint; the actor must recheck on release. +/// Advisory demand for one locally owned Cell. Ordinary movement requires a +/// settled worker sample. Explicit maintenance can use the configured peak +/// envelope on a draining donor; the actor must quiesce and recheck on release. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub struct CellTransferDemand { /// Cell the sample describes. @@ -198,12 +214,20 @@ pub struct CellTransferDemand { pub job_credits: u32, /// Logical time the Cell became resident. pub resident_since_ms: i64, + /// Actor-observed logical time of the most recent use. Pressure movement + /// prefers recently used settled Cells while local eviction closes the + /// oldest idle Cells. This observation conveys no source reservation. + pub last_used_ms: i64, /// Logical time the Cell last moved, when it has. pub last_moved_at_ms: Option, /// Consecutive settled samples observed for this Cell. pub stable_observations: u8, /// Whether the sample is settled enough to act on. pub settled: bool, + /// Select busy maintenance eligibility under an exact retained intent. The + /// planner also requires a draining source. This flag is advisory; journal + /// authorization and canonical runtime release remain separate barriers. + pub maintenance: bool, } /// Advisory transfer proposal. It conveys neither release nor receiver admission. @@ -386,9 +410,14 @@ impl PlacementPlanner { })) } - /// Plans at most two settled transfers and 8 GiB of projected restore + /// Plans at most two transfers and 8 GiB of projected restore /// bytes from one authenticated fleet snapshot. Receiver capacity is /// projected across selected intents, then rechecked during activation. + /// Ordinary movement requires settled samples; explicit maintenance uses + /// configured peak costs and requires a draining donor. + /// Pressure on an active donor prefers recently used settled Cells, leaving + /// oldest idle Cells to independent local eviction. This reduces contention + /// but reserves no actor; exact release must still reject a racing close. /// /// `balance` is the ownership balance of the same snapshot. When it elects /// one of the demands' sources as the donor, its receivers may absorb that @@ -454,6 +483,21 @@ impl PlacementPlanner { }; priority(right) .cmp(&priority(left)) + .then_with(|| { + let recency = |demand: &CellTransferDemand| { + observations + .iter() + .find(|node| node.session == demand.source) + .filter(|node| { + !node.draining && node.pressure >= PlacementPressure::Shedding + }) + .map_or(0, |_| demand.last_used_ms) + }; + // The actor's emergency eviction is oldest-first. Advisory + // fleet movement preserves recently used settled actors on + // another node instead of preparing those same cold victims. + recency(right).cmp(&recency(left)) + }) .then_with(|| left.cell.as_bytes().cmp(right.cell.as_bytes())) }); let mut intents = Vec::new(); @@ -469,13 +513,16 @@ impl PlacementPlanner { else { continue; }; - if !demand.settled + if (demand.maintenance && !source.draining) + || (!demand.settled && !demand.maintenance) || demand.generation == 0 || demand.memory_bytes == 0 || demand.disk_bytes == 0 || demand.job_credits == 0 || demand.resident_since_ms < 0 || demand.resident_since_ms > now_ms + || demand.last_used_ms < 0 + || demand.last_used_ms > now_ms || demand .last_moved_at_ms .is_some_and(|at| at < 0 || at > now_ms) @@ -644,6 +691,9 @@ impl PlacementPlanner { if observation.pressure >= PlacementPressure::Critical { return PlacementEligibility::CriticalPressure; } + if observation.pressure != PlacementPressure::Normal { + return PlacementEligibility::Pressure; + } if observation.memory_capacity_bytes == 0 || observation.free_memory_bytes == 0 { return PlacementEligibility::NoMemoryHeadroom; } diff --git a/crates/cellule-runtime/src/fleet/placement/tests.rs b/crates/cellule-runtime/src/fleet/placement/tests.rs index 50922668..4b8c6c39 100644 --- a/crates/cellule-runtime/src/fleet/placement/tests.rs +++ b/crates/cellule-runtime/src/fleet/placement/tests.rs @@ -117,12 +117,52 @@ fn demand(cell: u8, source: SessionId) -> CellTransferDemand { disk_bytes: 400, job_credits: 1, resident_since_ms: 0, + last_used_ms: 0, last_moved_at_ms: None, stable_observations: 2, settled: true, + maintenance: false, } } +#[test] +fn busy_maintenance_requires_explicit_demand_draining_source_and_receiver_capacity() { + let planner = PlacementPlanner::default(); + let mut donor = observation(1); + let mut receiver = observation(2); + let mut candidate = demand(1, donor.session); + candidate.settled = false; + candidate.stable_observations = 0; + candidate.resident_since_ms = 100; + candidate.last_moved_at_ms = Some(100); + candidate.maintenance = true; + let plans = |source, destination, demand| { + planner + .plan_transfers(100, &[source, destination], &[demand], None) + .unwrap() + }; + assert!(plans(donor, receiver, candidate).is_empty()); + donor.draining = true; + candidate.maintenance = false; + assert!(plans(donor, receiver, candidate).is_empty()); + candidate.maintenance = true; + assert_eq!(plans(donor, receiver, candidate).len(), 1); + receiver.free_disk_bytes = candidate.disk_bytes - 1; + assert!(plans(donor, receiver, candidate).is_empty()); + receiver.free_disk_bytes = 700; + receiver.free_memory_bytes = candidate.memory_bytes - 1; + assert!(plans(donor, receiver, candidate).is_empty()); + receiver.free_memory_bytes = 700; + receiver.draining = true; + assert!(plans(donor, receiver, candidate).is_empty()); + receiver.draining = false; + donor.authenticated = false; + assert!(plans(donor, receiver, candidate).is_empty()); + donor.authenticated = true; + donor.observed_at_ms = -1; + assert!(plans(donor, receiver, candidate).is_empty()); +} + #[test] fn transfer_projection_prevents_receiver_overcommit() { let planner = PlacementPlanner::default(); @@ -430,3 +470,66 @@ fn balance_donation_requires_the_elected_donor() { .is_empty() ); } + +#[test] +fn pressure_transfers_prefer_recent_settled_cells_over_local_eviction_victims() { + let planner = PlacementPlanner::default(); + let mut donor = observation(1); + donor.pressure = PlacementPressure::Shedding; + let mut receiver = observation(2); + receiver.free_memory_bytes = 1_000; + receiver.free_disk_bytes = 1_000; + let oldest = demand(1, donor.session); + let mut recent = demand(2, donor.session); + recent.last_used_ms = 90; + let mut middle = demand(3, donor.session); + middle.last_used_ms = 50; + let plans = |source, values: &[CellTransferDemand]| { + planner + .plan_transfers(100, &[source, receiver], values, None) + .unwrap() + }; + let expected = vec![recent.cell, middle.cell]; + for values in [[oldest, middle, recent], [recent, oldest, middle]] { + let intents = plans(donor, &values); + assert_eq!( + intents.iter().map(|intent| intent.cell).collect::>(), + expected + ); + } + recent.settled = false; + assert_eq!(plans(donor, &[oldest, middle, recent])[0].cell, middle.cell); + // Draining remains deterministic in Cell order even under pressure. + donor.draining = true; + recent.settled = true; + assert_eq!(plans(donor, &[recent, middle, oldest])[0].cell, oldest.cell); + donor.draining = false; + donor.pressure = PlacementPressure::Normal; + donor.observed_at_ms = 100_000; + receiver.observed_at_ms = 100_000; + let ordinary = planner + .plan_transfers(100_000, &[donor, receiver], &[recent, middle, oldest], None) + .unwrap(); + assert_eq!(ordinary[0].cell, oldest.cell); +} + +#[test] +fn pressure_transfer_recency_is_validated_and_ties_use_cell_identity() { + let planner = PlacementPlanner::default(); + let mut donor = observation(1); + donor.pressure = PlacementPressure::Shedding; + let receiver = observation(2); + let first = demand(1, donor.session); + let second = demand(2, donor.session); + let plans = |values: &[CellTransferDemand]| { + planner + .plan_transfers(100, &[donor, receiver], values, None) + .unwrap() + }; + assert_eq!(plans(&[second, first])[0].cell, first.cell); + for at in [-1, 101] { + let mut invalid = first; + invalid.last_used_ms = at; + assert!(plans(&[invalid]).is_empty()); + } +} diff --git a/crates/cellule-runtime/src/fleet/resource.rs b/crates/cellule-runtime/src/fleet/resource.rs index d2da6b1d..5881fa0f 100644 --- a/crates/cellule-runtime/src/fleet/resource.rs +++ b/crates/cellule-runtime/src/fleet/resource.rs @@ -743,6 +743,35 @@ mod tests { assert_eq!(ledger.snapshot().unwrap().used.disk_bytes(), 0); } + #[test] + fn prepared_ltx_disk_credit_inherits_one_runtime_charge() { + let ledger = ResourceLedger::new(ResourceCost::zero().with_disk_bytes(10)); + let parent = cellule_ltx::DiskBudget::new(10); + parent + .install_admission(Arc::new(LedgerDiskAdmission::new( + crate::identity::SessionId::from_bytes([10; 16]), + &ledger, + ))) + .unwrap(); + let prepared = parent.try_reserve(10).unwrap().into_budget(); + let file = prepared.try_reserve(6).unwrap(); + assert_eq!(ledger.snapshot().unwrap().used.disk_bytes(), 10); + prepared.finish_preparation().unwrap(); + assert_eq!(ledger.snapshot().unwrap().used.disk_bytes(), 6); + let competing = parent.try_reserve(4).unwrap(); + assert!(file.try_grow(1).is_err()); + assert_eq!(file.bytes(), 6); + assert_eq!(prepared.used(), 6); + assert_eq!(ledger.snapshot().unwrap().used.disk_bytes(), 10); + drop(competing); + file.try_grow(4).unwrap(); + assert_eq!(ledger.snapshot().unwrap().used.disk_bytes(), 10); + drop(prepared); + assert_eq!(ledger.snapshot().unwrap().used.disk_bytes(), 10); + drop(file); + assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); + } + #[test] fn ltx_host_admissions_share_one_runtime_ledger() { let ledger = ResourceLedger::new( diff --git a/crates/cellule-runtime/src/follower/directory.rs b/crates/cellule-runtime/src/follower/directory.rs index 4faf97db..e09ad253 100644 --- a/crates/cellule-runtime/src/follower/directory.rs +++ b/crates/cellule-runtime/src/follower/directory.rs @@ -267,7 +267,7 @@ fn modified_at_ms(path: &Path) -> Result { i64::try_from(duration.as_millis()).map_err(|_| Error::Node("follower marker time exceeds i64")) } -fn parse_session_directory(path: &Path) -> Result { +pub(super) fn parse_session_directory(path: &Path) -> Result { let name = path .file_name() .and_then(|name| name.to_str()) @@ -290,7 +290,7 @@ fn parse_session_directory(path: &Path) -> Result { Ok(session) } -fn parse_epoch_directory(path: &Path) -> Result { +pub(super) fn parse_epoch_directory(path: &Path) -> Result { let epoch = path .file_name() .and_then(|name| name.to_str()) diff --git a/crates/cellule-runtime/src/follower/inventory/mod.rs b/crates/cellule-runtime/src/follower/inventory/mod.rs new file mode 100644 index 00000000..acc547fc --- /dev/null +++ b/crates/cellule-runtime/src/follower/inventory/mod.rs @@ -0,0 +1,340 @@ +//! Advisory persisted-lane inventory. It never retires or seals a tail. + +use super::*; +use crate::identity::{Digest, encode_hex}; +use crate::node::NodeMode; + +const MAX_INVENTORY_LANES: usize = 10_000; +const INVENTORY_BYTES: u64 = 1 << 20; +const MAX_PAGE_ENTRIES: usize = 128; + +/// Local continuation bound to one opened store and its persisted lane topology. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct FollowerInventoryCursor { + topology: Digest, + leader: SessionId, + epoch: u64, +} + +impl FollowerInventoryCursor { + /// Encodes one fixed-width continuation for application transports. + #[must_use] + pub fn to_bytes(self) -> [u8; 56] { + let mut bytes = [0; 56]; + bytes[..32].copy_from_slice(self.topology.as_bytes()); + bytes[32..48].copy_from_slice(self.leader.as_bytes()); + bytes[48..].copy_from_slice(&self.epoch.to_le_bytes()); + bytes + } + + /// Decodes a continuation; the store validates its topology on use. + pub fn from_bytes(bytes: &[u8]) -> Result { + let bytes: &[u8; 56] = bytes + .try_into() + .map_err(|_| Error::Node("invalid follower inventory cursor width"))?; + let mut topology = [0; 32]; + topology.copy_from_slice(&bytes[..32]); + let mut leader = [0; 16]; + leader.copy_from_slice(&bytes[32..48]); + let mut epoch = [0; 8]; + epoch.copy_from_slice(&bytes[48..]); + let cursor = Self { + topology: Digest::from_bytes(topology), + leader: SessionId::from_bytes(leader), + epoch: u64::from_le_bytes(epoch), + }; + validate_lane(Lane { + leader: cursor.leader, + epoch: cursor.epoch, + })?; + Ok(cursor) + } +} + +/// Persisted local append fence; it is separate from remote epoch authority. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[repr(u8)] +pub enum FollowerLaneState { + /// The lane can still receive authorized appends. + Open = 1, + /// A seal prevents appends, but retained data may still be required. + Sealed = 2, + /// A retirement fence exists; current authority must still be inspected. + Retired = 3, +} + +/// One persisted lane, including lanes not yet loaded into the in-memory index. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct FollowerLaneObservation { + /// Boot session that originally enrolled this lane, even if now expired. + pub leader: SessionId, + /// Exact node-log epoch; identities are never inferred from live membership. + pub epoch: u64, + /// Persisted append-fence state at the local scan barrier. + pub state: FollowerLaneState, + /// Highest sequence recorded in the seal marker, if sealed. + pub sealed_through: Option, + /// Object-coverage watermark recorded by retirement, if retired. + pub retired_through: Option, +} + +impl FollowerLaneObservation { + fn key(self) -> ([u8; 16], u64) { + (*self.leader.as_bytes(), self.epoch) + } +} + +/// One bounded local scan page retaining its follower-index memory admission. +pub struct FollowerInventoryPage { + topology: Digest, + mode: NodeMode, + observed_at_ms: i64, + total_lanes: usize, + unretired_lanes: usize, + quarantined_entries: usize, + entries: Vec, + next: Option, + _memory: IndexReservation, +} + +impl FollowerInventoryPage { + /// Returns the store and persisted topology fingerprint shared by all pages. + #[must_use] + pub const fn topology(&self) -> Digest { + self.topology + } + /// Returns the shared local admission mode observed at the scan barrier. + #[must_use] + pub const fn mode(&self) -> NodeMode { + self.mode + } + /// Returns the caller's capture time; it does not refresh remote evidence. + #[must_use] + pub const fn observed_at_ms(&self) -> i64 { + self.observed_at_ms + } + /// Counts all persisted lanes, including lanes outside this page. + #[must_use] + pub const fn total_lanes(&self) -> usize { + self.total_lanes + } + /// Counts all open or sealed lanes. Zero is not remote retirement proof. + #[must_use] + pub const fn unretired_lanes(&self) -> usize { + self.unretired_lanes + } + /// Counts isolated filesystem entries that still require diagnosis. + #[must_use] + pub const fn quarantined_entries(&self) -> usize { + self.quarantined_entries + } + /// Returns at most 128 observations, sorted by leader session and epoch. + #[must_use] + pub fn entries(&self) -> &[FollowerLaneObservation] { + &self.entries + } + /// Continues only while the opened store and persisted topology still match. + #[must_use] + pub const fn next(&self) -> Option { + self.next + } +} + +impl FollowerStore { + /// Observes every persisted lane in bounded pages without changing its fences. + /// + /// Each page retains one MiB from the existing follower-index memory budget. + /// The scan includes cold lanes after reopen and reports quarantine. A topology + /// change requires restarting pagination. Callers must separately inspect all + /// authoritative log references, including dead owners, before shutdown. + /// Cordon must be committed before using a scan as an enrollment barrier. + pub async fn fleet_lanes_page( + &self, + cursor: Option, + limit: usize, + now_ms: i64, + ) -> Result { + if !(1..=MAX_PAGE_ENTRIES).contains(&limit) || now_ms < 0 { + return Err(Error::Node("invalid follower inventory page bounds")); + } + let memory = IndexReservation::new(&self.index_used, INVENTORY_BYTES)?; + let store = self.clone(); + tokio::task::spawn_blocking(move || { + // All lane creation, append, seal, retire, and collection mutate under + // this same disk lane. A completed cordon plus this barrier includes + // enrollments accepted before cordon; queued new lanes cannot appear. + let _disk = store + .retained + .lock() + .map_err(|_| Error::Node("follower disk reservation lock poisoned"))?; + collect_page(&store, cursor, limit, now_ms, memory) + }) + .await + .map_err(Error::FollowerWorkerJoin)? + } +} + +fn collect_page( + store: &FollowerStore, + cursor: Option, + limit: usize, + now_ms: i64, + memory: IndexReservation, +) -> Result { + // Preallocate the hard cap: geometric Vec growth must not exceed admission. + let mut lanes = Vec::with_capacity(MAX_INVENTORY_LANES); + let followers = store.root.join("followers"); + let exists = match std::fs::symlink_metadata(&followers) { + Ok(_) => true, + Err(error) if error.kind() == std::io::ErrorKind::NotFound => false, + Err(error) => return Err(error.into()), + }; + if exists { + require_directory(&followers)?; + let mut leaders = 0; + for leader in std::fs::read_dir(&followers)? { + leaders += 1; + if leaders > MAX_INVENTORY_LANES { + return Err(Error::Capacity("follower inventory leader bound")); + } + let path = leader?.path(); + require_directory(&path)?; + let leader = parse_session_directory(&path)?; + if path.file_name().and_then(|name| name.to_str()) + != Some(encode_hex(leader.as_bytes()).as_str()) + { + return Err(Error::Node("noncanonical follower leader directory")); + } + for epoch in std::fs::read_dir(&path)? { + if lanes.len() == MAX_INVENTORY_LANES { + return Err(Error::Capacity("follower inventory lane bound")); + } + let path = epoch?.path(); + require_directory(&path)?; + let epoch = parse_epoch_directory(&path)?; + if path.file_name().and_then(|name| name.to_str()) + != Some(epoch.to_string().as_str()) + { + return Err(Error::Node("noncanonical follower epoch directory")); + } + let sealed_through = marker(&path.join("sealed"))?; + let retired_through = marker(&path.join("retired"))?; + // Canonical retirement removes chunks after persisting its fence. + // A crash can retain the covered chunks; absence is legal only + // when that durable retirement marker exists. + match std::fs::symlink_metadata(path.join("chunks")) { + Ok(metadata) if metadata.file_type().is_dir() => {} + Err(error) + if error.kind() == std::io::ErrorKind::NotFound + && retired_through.is_some() => {} + Err(error) => return Err(error.into()), + Ok(_) => { + return Err(Error::Node("follower inventory chunks are not a directory")); + } + } + for entry in std::fs::read_dir(&path)? { + let entry = entry?; + if !matches!( + entry.file_name().to_str(), + Some("chunks" | "sealed" | "retired") + ) { + return Err(Error::Node("follower inventory lane has an unknown entry")); + } + } + if matches!((sealed_through, retired_through), (Some(sealed), Some(retired)) if sealed > retired) + { + return Err(Error::Node("follower retirement does not cover seal")); + } + lanes.push(FollowerLaneObservation { + leader, + epoch, + state: if retired_through.is_some() { + FollowerLaneState::Retired + } else if sealed_through.is_some() { + FollowerLaneState::Sealed + } else { + FollowerLaneState::Open + }, + sealed_through, + retired_through, + }); + } + } + } + lanes.sort_unstable_by_key(|lane| lane.key()); + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule-follower-inventory-v1"); + hash.update(&store.inventory_scope); + for lane in &lanes { + hash.update(lane.leader.as_bytes()); + hash.update(&lane.epoch.to_le_bytes()); + hash.update(&[lane.state as u8]); + } + let topology = Digest::from_bytes(*hash.finalize().as_bytes()); + let start = match cursor { + None => 0, + Some(cursor) if cursor.topology == topology => { + let position = lanes + .binary_search_by_key(&(*cursor.leader.as_bytes(), cursor.epoch), |lane| { + lane.key() + }) + .map_err(|_| Error::Node("follower inventory cursor key is absent"))?; + position + 1 + } + Some(_) => { + return Err(Error::Node( + "follower inventory topology changed; restart scan", + )); + } + }; + let end = start.saturating_add(limit).min(lanes.len()); + let entries = lanes[start..end].to_vec(); + let next = if end < lanes.len() { + entries.last().map(|last| FollowerInventoryCursor { + topology, + leader: last.leader, + epoch: last.epoch, + }) + } else { + None + }; + Ok(FollowerInventoryPage { + topology, + mode: store.admission.mode()?, + observed_at_ms: now_ms, + total_lanes: lanes.len(), + unretired_lanes: lanes + .iter() + .filter(|lane| lane.state != FollowerLaneState::Retired) + .count(), + quarantined_entries: quarantine_entry_count(&store.root)?, + entries, + next, + _memory: memory, + }) +} + +fn require_directory(path: &Path) -> Result<()> { + if !std::fs::symlink_metadata(path)?.file_type().is_dir() { + return Err(Error::Node( + "follower inventory contains a special directory", + )); + } + Ok(()) +} + +fn marker(path: &Path) -> Result> { + match std::fs::symlink_metadata(path) { + Err(error) if error.kind() == std::io::ErrorKind::NotFound => Ok(None), + Err(error) => Err(error.into()), + Ok(metadata) if metadata.file_type().is_file() && metadata.len() == 8 => Ok(Some( + read_watermark(path, "invalid follower inventory marker")?, + )), + Ok(_) => Err(Error::Node( + "follower inventory marker is not an eight-byte file", + )), + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-runtime/src/follower/inventory/tests.rs b/crates/cellule-runtime/src/follower/inventory/tests.rs new file mode 100644 index 00000000..42dacfbc --- /dev/null +++ b/crates/cellule-runtime/src/follower/inventory/tests.rs @@ -0,0 +1,209 @@ +use super::*; + +fn store(root: &Path) -> FollowerStore { + FollowerStore::open( + root.to_owned(), + cellule_ltx::Limits::default(), + cellule_ltx::DiskBudget::new(128 << 20), + ) + .unwrap() +} + +fn leader() -> SessionId { + SessionId::from_bytes([1; 16]) +} + +fn enroll(root: &Path, epoch: u64) { + ensure_lane_directories( + root, + Lane { + leader: leader(), + epoch, + }, + ) + .unwrap(); +} + +fn used(store: &FollowerStore) -> u64 { + *store.index_used.lock().unwrap() +} + +#[tokio::test] +async fn persisted_cold_lanes_are_sorted_paged_and_charged_without_loading_tail_indexes() { + let root = tempfile::tempdir().unwrap(); + let store = store(root.path()); + for epoch in (1..=255).rev() { + enroll(root.path(), epoch); + } + assert!(store.lanes.lock().unwrap().is_empty()); + assert!( + (MAX_INVENTORY_LANES + MAX_PAGE_ENTRIES) * std::mem::size_of::() + < INVENTORY_BYTES as usize + ); + let first = store.fleet_lanes_page(None, 128, 10).await.unwrap(); + assert_eq!(first.total_lanes(), 255); + assert_eq!(first.unretired_lanes(), 255); + assert_eq!(first.entries().len(), 128); + assert_eq!(first.entries()[0].epoch, 1); + assert_eq!(first.entries()[127].epoch, 128); + assert_eq!(first.mode(), NodeMode::Active); + let cursor = FollowerInventoryCursor::from_bytes(&first.next().unwrap().to_bytes()).unwrap(); + let second = store.fleet_lanes_page(Some(cursor), 128, 11).await.unwrap(); + assert_eq!(second.topology(), first.topology()); + assert_eq!(second.entries().len(), 127); + assert_eq!(second.entries()[0].epoch, 129); + assert_eq!(second.entries()[126].epoch, 255); + assert!(second.next().is_none()); + assert_eq!(used(&store), 2 * INVENTORY_BYTES); + assert!(store.lanes.lock().unwrap().is_empty()); + assert_eq!(store.scan_count(), 0); + drop(first); + drop(second); + assert_eq!(used(&store), 0); +} + +#[tokio::test] +async fn topology_or_store_replacement_rejects_cursor_and_releases_admission() { + let root = tempfile::tempdir().unwrap(); + let first_store = store(root.path()); + enroll(root.path(), 1); + enroll(root.path(), 2); + let page = first_store.fleet_lanes_page(None, 1, 10).await.unwrap(); + let cursor = page.next().unwrap(); + drop(page); + first_store.seal(leader(), 1).await.unwrap(); + assert!( + first_store + .fleet_lanes_page(Some(cursor), 1, 11) + .await + .is_err() + ); + assert_eq!(used(&first_store), 0); + let sealed = first_store.fleet_lanes_page(None, 1, 12).await.unwrap(); + assert_eq!(sealed.entries()[0].state, FollowerLaneState::Sealed); + assert_eq!(sealed.entries()[0].sealed_through, Some(0)); + let cursor = sealed.next().unwrap(); + drop(sealed); + let reopened = store(root.path()); + assert!( + reopened + .fleet_lanes_page(Some(cursor), 1, 13) + .await + .is_err() + ); + assert_eq!(used(&reopened), 0); + first_store.retire(leader(), 1, 0).await.unwrap(); + let retired = first_store.fleet_lanes_page(None, 128, 14).await.unwrap(); + assert_eq!(retired.entries()[0].state, FollowerLaneState::Retired); + assert_eq!(retired.entries()[0].retired_through, Some(0)); + assert_eq!(retired.unretired_lanes(), 1); + drop(retired); + assert_eq!(used(&first_store), 0); +} + +#[tokio::test] +async fn cordon_and_quarantine_remain_visible_without_manufacturing_empty_safety() { + let root = tempfile::tempdir().unwrap(); + let gate = crate::fleet::admission::NodeAdmission::default(); + let store = store(root.path()).with_node_admission(gate.clone()); + enroll(root.path(), 1); + gate.cordon().unwrap(); + std::fs::create_dir(root.path().join(FOLLOWER_QUARANTINE)).unwrap(); + std::fs::write( + root.path().join(FOLLOWER_QUARANTINE).join("1.bad"), + b"unknown", + ) + .unwrap(); + let page = store.fleet_lanes_page(None, 128, 20).await.unwrap(); + assert_eq!(page.mode(), NodeMode::Cordoned); + assert_eq!(page.total_lanes(), 1); + assert_eq!(page.unretired_lanes(), 1); + assert_eq!(page.quarantined_entries(), 1); + assert_eq!(page.observed_at_ms(), 20); + assert_eq!(page.entries()[0].leader, leader()); +} + +#[tokio::test] +async fn malformed_or_oversized_markers_fail_without_unbounded_read_or_leaked_page() { + let root = tempfile::tempdir().unwrap(); + let store = store(root.path()); + enroll(root.path(), 1); + let directory = lane_directory( + root.path(), + Lane { + leader: leader(), + epoch: 1, + }, + ); + std::fs::write(directory.join("sealed"), [0; 9]).unwrap(); + assert!(store.fleet_lanes_page(None, 128, 1).await.is_err()); + assert_eq!(used(&store), 0); + std::fs::write(directory.join("sealed"), 5_u64.to_le_bytes()).unwrap(); + std::fs::write(directory.join("retired"), 4_u64.to_le_bytes()).unwrap(); + assert!(store.fleet_lanes_page(None, 128, 1).await.is_err()); + assert_eq!(used(&store), 0); + for limit in [0, 129, usize::MAX] { + assert!(store.fleet_lanes_page(None, limit, 1).await.is_err()); + } + assert!(store.fleet_lanes_page(None, 1, -1).await.is_err()); + assert!(FollowerInventoryCursor::from_bytes(&[0; 56]).is_err()); + assert!(FollowerInventoryCursor::from_bytes(&[1; 57]).is_err()); +} + +#[cfg(unix)] +#[tokio::test] +async fn special_directory_or_marker_cannot_be_reported_as_an_empty_or_retired_store() { + let root = tempfile::tempdir().unwrap(); + let store = store(root.path()); + std::os::unix::fs::symlink(root.path().join("missing"), root.path().join("followers")).unwrap(); + assert!(store.fleet_lanes_page(None, 128, 1).await.is_err()); + assert_eq!(used(&store), 0); + std::fs::remove_file(root.path().join("followers")).unwrap(); + enroll(root.path(), 1); + let directory = lane_directory( + root.path(), + Lane { + leader: leader(), + epoch: 1, + }, + ); + std::os::unix::fs::symlink(root.path().join("missing"), directory.join("retired")).unwrap(); + assert!(store.fleet_lanes_page(None, 128, 1).await.is_err()); + assert_eq!(used(&store), 0); +} + +#[tokio::test] +async fn cancelled_page_keeps_memory_until_its_local_scan_finishes() { + let root = tempfile::tempdir().unwrap(); + let store = store(root.path()); + let (ready_tx, ready_rx) = std::sync::mpsc::channel(); + let (release_tx, release_rx) = std::sync::mpsc::channel(); + let holder = { + let retained = Arc::clone(&store.retained); + std::thread::spawn(move || { + let _disk = retained.lock().unwrap(); + ready_tx.send(()).unwrap(); + release_rx.recv().unwrap(); + }) + }; + ready_rx.recv().unwrap(); + let task = { + let store = store.clone(); + tokio::spawn(async move { store.fleet_lanes_page(None, 1, 1).await }) + }; + while used(&store) == 0 { + tokio::task::yield_now().await; + } + task.abort(); + assert_eq!(used(&store), INVENTORY_BYTES); + release_tx.send(()).unwrap(); + let _ = task.await; + tokio::time::timeout(std::time::Duration::from_secs(3), async { + while used(&store) != 0 { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + holder.join().unwrap(); +} diff --git a/crates/cellule-runtime/src/follower/mod.rs b/crates/cellule-runtime/src/follower/mod.rs index ff5ce992..ec085307 100644 --- a/crates/cellule-runtime/src/follower/mod.rs +++ b/crates/cellule-runtime/src/follower/mod.rs @@ -12,8 +12,13 @@ use crate::identity::SessionId; use crate::{Error, Result}; mod directory; +mod inventory; mod records; +pub use inventory::{ + FollowerInventoryCursor, FollowerInventoryPage, FollowerLaneObservation, FollowerLaneState, +}; + use directory::*; use records::*; @@ -68,6 +73,11 @@ struct Lane { epoch: u64, } +enum RetirementWatermark { + Covered(u64), + Recovered { active: bool }, +} + type LaneState = Arc>>; type LaneMap = Arc>>; @@ -137,6 +147,8 @@ pub struct FollowerStore { index_used: Arc>, quarantined_entries: usize, scan_counter: ScanCounter, + admission: crate::fleet::admission::NodeAdmission, + inventory_scope: [u8; 16], } impl FollowerStore { @@ -167,9 +179,23 @@ impl FollowerStore { index_used: Arc::new(Mutex::new(0)), quarantined_entries, scan_counter: new_scan_counter(), + admission: crate::fleet::admission::NodeAdmission::default(), + inventory_scope: rand::random(), }) } + /// Installs the runtime's shared gate before the store is shared or serves + /// traffic. Cordon blocks entirely new lanes while existing acknowledged + /// tails remain appendable under their normal epoch authorization. + #[must_use] + pub fn with_node_admission( + mut self, + admission: crate::fleet::admission::NodeAdmission, + ) -> Self { + self.admission = admission; + self + } + #[cfg(test)] pub(crate) fn scan_count(&self) -> usize { self.scan_counter.load(Ordering::Relaxed) @@ -215,6 +241,10 @@ impl FollowerStore { return Err(Error::Node("invalid follower append batch")); } let lane = Lane { leader, epoch }; + validate_lane(lane)?; + if !lane_directory(&self.root, lane).exists() { + self.admission.check_new_role()?; + } let lock = self.lane_lock(lane)?; let root = self.root.clone(); let limits = self.limits; @@ -224,6 +254,8 @@ impl FollowerStore { let growth = encoded_bytes .and_then(|bytes| bytes.checked_add((frames.len() * RECORD_HEADER_BYTES) as u64)) .ok_or(Error::Node("follower append byte count overflow"))?; + let admission = self.admission.clone(); + let lanes = Arc::clone(&self.lanes); tokio::task::spawn_blocking(move || { let retained = retained .lock() @@ -232,20 +264,36 @@ impl FollowerStore { let mut state = lock .lock() .map_err(|_| Error::Node("follower lane lock poisoned"))?; - let result = append_sync( - &root, - lane, - frames, - covered_through, - limits, - &index_used, - &mut state, - &scan_counter, - ); + let result = (|| { + if !lane_directory(&root, lane).exists() { + admission.admit(|| ensure_lane_directories(&root, lane))?; + } + append_sync( + &root, + lane, + frames, + covered_through, + limits, + &index_used, + &mut state, + &scan_counter, + ) + })(); let resize = follower_bytes(&root).and_then(|bytes| retained.resize(bytes).map_err(Error::from)); if result.is_err() { *state = None; + if !lane_directory(&root, lane).exists() + && let Ok(mut lanes) = lanes.lock() + { + // Cordon may win after the precheck but before enrollment. + // Rejected transient lanes must not accumulate in memory. + if lanes.get(&lane).is_some_and(|registered| { + Arc::ptr_eq(registered, &lock) && Arc::strong_count(registered) == 2 + }) { + lanes.remove(&lane); + } + } } settle_disk_reservation(result, resize) }) @@ -293,6 +341,41 @@ impl FollowerStore { covered_through: u64, ) -> Result { let lane = Lane { leader, epoch }; + self.retire_lane(lane, RetirementWatermark::Covered(covered_through)) + .await + } + + /// Retires a canonically recovered lane through its existing durable fence. + /// The application binds `member` to this receiver and authenticates the + /// requester before obtaining fresh directory authorization. Active lanes + /// must retain their actual native seal; no caller watermark is accepted. + pub async fn retire_recovered( + &self, + member: crate::identity::NodeId, + authorization: crate::node::RecoveredLogRetirementAuthorization, + ) -> Result { + if authorization.member() != member { + return Err(Error::Fenced); + } + let sealed = authorization.sealed(); + let lane = Lane { + leader: sealed.session(), + epoch: sealed.log().epoch(), + }; + self.retire_lane( + lane, + RetirementWatermark::Recovered { + active: sealed.log().active(), + }, + ) + .await + } + + async fn retire_lane( + &self, + lane: Lane, + watermark: RetirementWatermark, + ) -> Result { let lock = self.lane_lock(lane)?; let root = self.root.clone(); let limits = self.limits; @@ -308,7 +391,7 @@ impl FollowerStore { if !lane_directory(&root, lane).join("retired").exists() { retained.try_grow(8)?; } - let result = retire_sync(&root, lane, covered_through, limits, &scan_counter); + let result = retire_sync(&root, lane, watermark, limits, &scan_counter); let resize = follower_bytes(&root).and_then(|bytes| retained.resize(bytes).map_err(Error::from)); *state = None; diff --git a/crates/cellule-runtime/src/follower/records/append.rs b/crates/cellule-runtime/src/follower/records/append.rs index 27e0da06..44e7121f 100644 --- a/crates/cellule-runtime/src/follower/records/append.rs +++ b/crates/cellule-runtime/src/follower/records/append.rs @@ -208,7 +208,7 @@ pub(in crate::follower) fn seal_sync( pub(in crate::follower) fn retire_sync( root: &Path, lane: Lane, - covered_through: u64, + watermark: RetirementWatermark, limits: cellule_ltx::Limits, scan_counter: &ScanCounter, ) -> Result { @@ -218,6 +218,33 @@ pub(in crate::follower) fn retire_sync( let chunks = directory.join("chunks"); let retained = scan_lane_counted(&chunks, lane, limits, scan_counter)?; let durable_through = retained.keys().next_back().copied().unwrap_or(0); + let covered_through = match watermark { + RetirementWatermark::Covered(value) => value, + RetirementWatermark::Recovered { active } => { + if !active && durable_through != 0 { + return Err(Error::Node("follower lane has uncovered records")); + } + if directory.join("retired").exists() { + read_watermark( + &directory.join("retired"), + "follower retire marker is invalid", + )? + } else if directory.join("sealed").exists() { + let sealed = + read_watermark(&directory.join("sealed"), "follower seal marker is invalid")?; + if sealed != durable_through { + return Err(Error::Node("follower seal watermark differs")); + } + sealed + } else if !active { + // Inactive enrollment cannot acknowledge a tail. The shared + // coverage check still refuses unexpected native records. + 0 + } else { + return Err(Error::Node("recovered follower lane has no native seal")); + } + } + }; if durable_through > covered_through { return Err(Error::Node("follower lane has uncovered records")); } diff --git a/crates/cellule-runtime/src/follower/records/mod.rs b/crates/cellule-runtime/src/follower/records/mod.rs index c25ce0c0..1f5561fb 100644 --- a/crates/cellule-runtime/src/follower/records/mod.rs +++ b/crates/cellule-runtime/src/follower/records/mod.rs @@ -25,7 +25,7 @@ pub(in crate::follower) struct IndexReservation { bytes: u64, } impl IndexReservation { - fn new(used: &Arc>, bytes: u64) -> Result { + pub(in crate::follower) fn new(used: &Arc>, bytes: u64) -> Result { let mut current = used .lock() .map_err(|_| Error::Node("follower index reservation lock poisoned"))?; diff --git a/crates/cellule-runtime/src/follower/tests/append.rs b/crates/cellule-runtime/src/follower/tests/append.rs index 432b68bb..ba7b533e 100644 --- a/crates/cellule-runtime/src/follower/tests/append.rs +++ b/crates/cellule-runtime/src/follower/tests/append.rs @@ -2,6 +2,79 @@ use super::*; +#[tokio::test] +async fn cordon_blocks_new_lanes_while_existing_tail_appends_and_restart_remain_safe() { + let limits = cellule_ltx::Limits::default(); + let source = tempfile::TempDir::new().unwrap(); + let mut database = Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); + database + .transaction(|transaction| { + transaction.execute_batch( + "CREATE TABLE events(id INTEGER PRIMARY KEY, body TEXT NOT NULL);\ + INSERT INTO events(body) VALUES ('one')", + ) + }) + .unwrap(); + let capture = database.capture().unwrap(); + let segment = capture.segments.first().unwrap(); + let root = tempfile::TempDir::new().unwrap(); + let disk = cellule_ltx::DiskBudget::new(1 << 30); + let gate = crate::fleet::admission::NodeAdmission::default(); + let store = FollowerStore::open(root.path().to_owned(), limits, disk.clone()) + .unwrap() + .with_node_admission(gate.clone()); + let leader = SessionId::from_bytes([1; 16]); + store + .append(leader, 2, vec![frame(1, segment, limits)], 0) + .await + .unwrap(); + gate.observe(crate::fleet::pressure::PressureState::Shedding, 100) + .unwrap(); + assert_eq!( + store + .append(leader, 2, vec![frame(2, segment, limits)], 0) + .await + .unwrap() + .durable_through, + 2 + ); + assert!(matches!( + store + .append(leader, 3, vec![frame(1, segment, limits)], 0) + .await, + Err(Error::Capacity("node pressure")) + )); + assert!(!lane_directory(root.path(), Lane { leader, epoch: 3 }).exists()); + assert_eq!(store.lanes.lock().unwrap().len(), 1); + gate.cordon().unwrap(); + gate.observe(crate::fleet::pressure::PressureState::Normal, 200) + .unwrap(); + assert!(matches!( + store + .append(leader, 3, vec![frame(1, segment, limits)], 0) + .await, + Err(Error::CellDraining) + )); + drop(store); + let restarted = FollowerStore::open(root.path().to_owned(), limits, disk) + .unwrap() + .with_node_admission(gate); + assert_eq!( + restarted + .append(leader, 2, vec![frame(2, segment, limits)], 0) + .await + .unwrap() + .durable_through, + 2 + ); + assert!(matches!( + restarted + .append(leader, 3, vec![frame(1, segment, limits)], 0) + .await, + Err(Error::CellDraining) + )); +} + #[tokio::test] async fn append_recovers_torn_suffix_deduplicates_and_seals() { let limits = cellule_ltx::Limits::default(); diff --git a/crates/cellule-runtime/src/lib.rs b/crates/cellule-runtime/src/lib.rs index 1304341f..4c6d166f 100644 --- a/crates/cellule-runtime/src/lib.rs +++ b/crates/cellule-runtime/src/lib.rs @@ -54,7 +54,7 @@ pub use cell::executor::{MutationIdentity, Resolution}; pub use cell::worker::SqlWorkerPool; pub use client::{ CellClient, CellReadReplica, Committed, InvocationError, Observed, PendingMutation, - PreparedCommand, Receipt, + PreparedCommand, ReadReplicaSource, Receipt, }; pub use error::{Error, Result}; diff --git a/crates/cellule-runtime/src/ltx.rs b/crates/cellule-runtime/src/ltx.rs index 420fdcaf..b5f9b6bf 100644 --- a/crates/cellule-runtime/src/ltx.rs +++ b/crates/cellule-runtime/src/ltx.rs @@ -5,5 +5,6 @@ pub use cellule_ltx::{ CaptureTiming, CellObjectKind, CellReplica, CellStorageLayout, DiskBudget, DiskReservation, - Host, Limits, LtxPhase, LtxReadOrigin, LtxRequestOutcome, ScratchMonitor, + Host, Limits, LtxPhase, LtxReadOrigin, LtxRequestOutcome, NodeFrameScope, RootRef, + ScratchMonitor, encode_node_frame, }; diff --git a/crates/cellule-runtime/src/node/advertisement/codec.rs b/crates/cellule-runtime/src/node/advertisement/codec.rs index 6d97d0f6..648f2d6b 100644 --- a/crates/cellule-runtime/src/node/advertisement/codec.rs +++ b/crates/cellule-runtime/src/node/advertisement/codec.rs @@ -111,6 +111,8 @@ pub(in crate::node) struct RawAdvertisement { pub(in crate::node) placement_version: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub(in crate::node) placement_signature: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub(in crate::node) operational_sample: Option, } #[derive(Deserialize, Serialize)] @@ -157,10 +159,25 @@ impl From<&NodeAdvertisement> for RawAdvertisement { .iter() .any(|byte| *byte != 0) .then(|| encode_hex(&value.placement_signature)), + operational_sample: value.operational_sample.map(|sample| RawOperationalSample { + mode: sample.mode, + pressure: sample.pressure, + sequence: sample.sequence.to_string(), + observed_at_ms: sample.observed_at_ms.to_string(), + }), } } } +#[derive(Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub(in crate::node) struct RawOperationalSample { + pub(in crate::node) mode: NodeMode, + pub(in crate::node) pressure: NodePressure, + pub(in crate::node) sequence: String, + pub(in crate::node) observed_at_ms: String, +} + #[derive(Deserialize, Serialize)] #[serde(deny_unknown_fields)] pub(in crate::node) struct RawLease { @@ -306,6 +323,18 @@ impl TryFrom for NodeAdvertisement { .map(|signature| decode_hex(&signature)) .transpose()? .unwrap_or([0; 64]), + operational_sample: value + .operational_sample + .map(|sample| { + NodeOperationalSample { + mode: sample.mode, + pressure: sample.pressure, + sequence: canonical_u64(&sample.sequence)?, + observed_at_ms: canonical_i64(&sample.observed_at_ms)?, + } + .validated() + }) + .transpose()?, }) } } diff --git a/crates/cellule-runtime/src/node/advertisement/mod.rs b/crates/cellule-runtime/src/node/advertisement/mod.rs index be075aac..f94b6b7a 100644 --- a/crates/cellule-runtime/src/node/advertisement/mod.rs +++ b/crates/cellule-runtime/src/node/advertisement/mod.rs @@ -31,6 +31,7 @@ pub struct NodeAdvertisement { pub(super) placement_version: u32, pub(super) placement_signature: [u8; 64], pub(super) placement: Option, + pub(super) operational_sample: Option, } impl NodeAdvertisement { @@ -78,6 +79,7 @@ impl NodeAdvertisement { placement_version: 0, placement_signature: [0; 64], placement: None, + operational_sample: None, }; advertisement.validate_shape()?; advertisement.signature = signing_key.sign(&advertisement.signing_bytes()?).to_bytes(); @@ -192,25 +194,91 @@ impl NodeAdvertisement { placement: NodePlacementCapacity, signing_key: &SigningKey, ) -> Result { + if signing_key.verifying_key().to_bytes() != self.public_key { + return Err(Error::Node( + "placement key differs from the boot-session key", + )); + } placement.validated()?; self.placement = Some(placement); self.placement_version = PLACEMENT_SCHEMA_VERSION; + self.operational_sample = None; + self.placement_signature = signing_key + .sign(&self.placement_signing_bytes()?) + .to_bytes(); + Ok(self) + } + + /// Emits schema 3 only after the application has completed reader rollout. + /// The sample must come from local admission and classifier state. The + /// schema 2 builder remains the default bridge writer with identical bytes. + pub fn with_operational_placement( + mut self, + placement: NodePlacementCapacity, + sample: NodeOperationalSample, + signing_key: &SigningKey, + ) -> Result { + if signing_key.verifying_key().to_bytes() != self.public_key { + return Err(Error::Node( + "placement key differs from the boot-session key", + )); + } + self.placement = Some(placement.validated()?); + self.operational_sample = Some(sample.validated()?); + self.placement_version = OPERATIONAL_PLACEMENT_SCHEMA_VERSION; + self.validate_shape()?; self.placement_signature = signing_key .sign(&self.placement_signing_bytes()?) .to_bytes(); Ok(self) } + /// Returns the explicit operational sample only for an understood schema. + #[must_use] + pub const fn operational_sample(&self) -> Option { + if self.placement_version == OPERATIONAL_PLACEMENT_SCHEMA_VERSION { + self.operational_sample + } else { + None + } + } + + /// Checks signed mode/pressure for all new remote role selections. Identity + /// advertisements without an operational block retain bridge behavior. + #[must_use] + pub fn accepts_new_roles(&self, now_ms: i64) -> bool { + if now_ms < self.issued_at_ms + || now_ms >= self.expires_at_ms + || self.placement_version > OPERATIONAL_PLACEMENT_SCHEMA_VERSION + { + return false; + } + self.operational_sample.is_none_or(|sample| { + self.placement_version == OPERATIONAL_PLACEMENT_SCHEMA_VERSION + && self.has_signed_placement() + && sample.accepts_roles() + && sample.observed_at_ms <= now_ms + && now_ms.saturating_sub(sample.observed_at_ms) <= MAX_ADVERTISEMENT_LIFETIME_MS + }) + } + /// Reports whether this advertisement carries an authenticated placement /// snapshot. Legacy identity-only records remain readable but are never /// eligible for weighted ownership placement. #[must_use] pub fn has_signed_placement(&self) -> bool { - self.placement_version == PLACEMENT_SCHEMA_VERSION - && self.placement.is_some() + matches!( + self.placement_version, + PLACEMENT_SCHEMA_VERSION | OPERATIONAL_PLACEMENT_SCHEMA_VERSION + ) && self.placement.is_some() && self.placement_signature.iter().any(|byte| *byte != 0) } + pub(crate) fn verify_placement(&self) -> Result<()> { + self.validate_shape()?; + self.verify_signature() + } + /// Returns the node-log status, when the node has an enrolled log. #[must_use] pub const fn log(&self) -> Option<&NodeLogStatus> { @@ -220,6 +288,13 @@ impl NodeAdvertisement { pub(super) fn encode(&self) -> Result> { self.validate_shape()?; self.verify_signature()?; + self.canonical_bytes() + } + + // Both callers first verify this immutable value. Re-encoding for canonical + // byte equality needs the same serializer, without a second signature pass. + // Storage producers still enter through encode and verify before emission. + fn canonical_bytes(&self) -> Result> { let encoded = serde_json::to_vec(&RawAdvertisement::from(self))?; if encoded.len() as u64 > MAX_NODE_BYTES { return Err(Error::Node("advertisement exceeds 64 KiB")); @@ -235,7 +310,7 @@ impl NodeAdvertisement { let advertisement = Self::try_from(raw)?; advertisement.validate_shape()?; advertisement.verify_signature()?; - if advertisement.encode()?.as_slice() != bytes { + if advertisement.canonical_bytes()?.as_slice() != bytes { return Err(Error::Node("advertisement JSON is not canonical")); } Ok(advertisement) @@ -288,6 +363,25 @@ impl NodeAdvertisement { if let Some(placement) = self.placement { placement.validated()?; } + match (self.placement_version, self.operational_sample) { + (OPERATIONAL_PLACEMENT_SCHEMA_VERSION, Some(sample)) => { + sample.validated()?; + if sample.observed_at_ms > self.issued_at_ms || self.placement.is_none() { + return Err(Error::Node( + "operational sample is ahead of its advertisement", + )); + } + } + (OPERATIONAL_PLACEMENT_SCHEMA_VERSION, None) => { + return Err(Error::Node("operational placement sample is missing")); + } + (0..=PLACEMENT_SCHEMA_VERSION, Some(_)) => { + return Err(Error::Node( + "operational sample uses a legacy placement schema", + )); + } + _ => {} + } if self.module_digests.is_empty() || self.module_digests.len() > MAX_MODULES || !self @@ -305,6 +399,8 @@ impl NodeAdvertisement { } pub(super) fn verify_signature(&self) -> Result<()> { + #[cfg(test)] + crate::node::tests::record_signature_pass(); self.verifying_key()? .verify( &self.signing_bytes()?, @@ -317,7 +413,7 @@ impl NodeAdvertisement { return Err(Error::Node("legacy placement record carries a signature")); } } - PLACEMENT_SCHEMA_VERSION => { + PLACEMENT_SCHEMA_VERSION | OPERATIONAL_PLACEMENT_SCHEMA_VERSION => { if !self.has_signed_placement() { return Err(Error::Node("placement signature is missing")); } @@ -458,7 +554,10 @@ impl NodeTakeoverProof { let recovering = fenced.log.as_ref().ok_or(Error::Fenced)?; let sealed_log = sealed.log(); if sealed.session() != fenced.session - || sealed_log.phase() != NodeLogPhase::Sealed + || !matches!( + sealed_log.phase(), + NodeLogPhase::Sealed | NodeLogPhase::Retired + ) || sealed_log.epoch() != recovering.epoch() || sealed_log.members() != recovering.members() || sealed_log.active() != recovering.active() @@ -506,6 +605,29 @@ pub(super) fn validate_successor( current: &NodeAdvertisement, next: &NodeAdvertisement, ) -> Result<()> { + if let Some(previous) = current.operational_sample { + match next.operational_sample { + Some(sample) => { + if sample.sequence < previous.sequence + || sample.observed_at_ms < previous.observed_at_ms + || (sample.sequence == previous.sequence + && (sample != previous + || current.capacity != next.capacity + || current.placement != next.placement)) + { + return Err(Error::Node( + "operational sample regressed or changed without a sequence", + )); + } + } + None if !previous.accepts_roles() => { + return Err(Error::Node( + "operational withdrawal would hide closed admission", + )); + } + None => {} + } + } if !same_boot_identity(current, next) || current.log != next.log || current.generation.checked_add(1) != Some(next.generation) diff --git a/crates/cellule-runtime/src/node/capacity.rs b/crates/cellule-runtime/src/node/capacity.rs index 12553475..49ebb4d8 100644 --- a/crates/cellule-runtime/src/node/capacity.rs +++ b/crates/cellule-runtime/src/node/capacity.rs @@ -2,6 +2,78 @@ use super::*; +/// Signed role-admission mode, independent of measured resource capacity. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Deserialize, Serialize)] +#[serde(rename_all = "snake_case")] +pub enum NodeMode { + /// The node may admit new roles if its pressure and ledgers permit them. + #[default] + Active, + /// Existing roles continue while new role acquisition is closed. + Cordoned, + /// Accepted work and role obligations are being settled for shutdown. + Draining, +} + +/// Stable pressure tier published by the node's existing classifier. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Deserialize, Serialize)] +#[serde(rename_all = "snake_case")] +pub enum NodePressure { + /// Normal measured pressure. + #[default] + Normal, + /// Optional receive is paused, including when the local sample is stale. + Constrained, + /// Sustained pressure on at least one resource dimension. + Shedding, + /// Sustained critical memory and disk pressure. + Critical, +} + +impl TryFrom for NodePressure { + type Error = Error; + + fn try_from(state: crate::fleet::pressure::PressureState) -> Result { + use crate::fleet::pressure::PressureState; + match state { + PressureState::Normal => Ok(Self::Normal), + PressureState::Constrained => Ok(Self::Constrained), + PressureState::Shedding => Ok(Self::Shedding), + PressureState::Critical => Ok(Self::Critical), + PressureState::Recovering => Err(Error::Node("transient pressure is not a wire tier")), + } + } +} + +/// Schema 3 operational sample; republication must preserve its sample identity. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct NodeOperationalSample { + /// Role-admission mode at the time of the sample. + pub mode: NodeMode, + /// Stable output of the local hysteretic pressure classifier. + pub pressure: NodePressure, + /// Strictly increasing sequence scoped to the boot session. + pub sequence: u64, + /// Logical time of measurement, not time of heartbeat republication. + pub observed_at_ms: i64, +} + +impl NodeOperationalSample { + /// Validates sample identity and nonnegative observation time. + pub const fn validated(self) -> Result { + if self.sequence == 0 || self.observed_at_ms < 0 { + return Err(Error::Node("operational sample identity is invalid")); + } + Ok(self) + } + + /// Whether this sample admits proactive writer, reader, or follower receive. + #[must_use] + pub fn accepts_roles(self) -> bool { + self.mode == NodeMode::Active && self.pressure == NodePressure::Normal + } +} + /// Capacity hints published by one node boot session. #[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] pub struct NodeCapacity { diff --git a/crates/cellule-runtime/src/node/directory/advertisement.rs b/crates/cellule-runtime/src/node/directory/advertisement.rs index 768ea429..7aa98cad 100644 --- a/crates/cellule-runtime/src/node/directory/advertisement.rs +++ b/crates/cellule-runtime/src/node/directory/advertisement.rs @@ -80,6 +80,7 @@ impl NodeDirectory { .filter(|candidate| { let capacity = candidate.capacity(); candidate.expires_at_ms() > now_ms + && candidate.accepts_new_roles(now_ms) && candidate.node() != owner_node && candidate.module_digests().contains(&code) && capacity.free_memory_bytes @@ -256,6 +257,21 @@ impl NodeDirectory { now_ms: i64, limit: usize, ) -> Result> { + Ok(self + .select_log_advertisements(leader, required_follower_bytes, now_ms, limit) + .await? + .into_iter() + .map(|advertisement| advertisement.node) + .collect()) + } + + pub(super) async fn select_log_advertisements( + &self, + leader: SessionId, + required_follower_bytes: u64, + now_ms: i64, + limit: usize, + ) -> Result> { if required_follower_bytes == 0 { return Err(Error::Node("node-log follower byte requirement is zero")); } @@ -275,11 +291,7 @@ impl NodeDirectory { .iter() .filter(|candidate| { candidate.node != leader.node - && candidate.capacity.log_protocol == NODE_LOG_PROTOCOL_VERSION - && candidate.capacity.follower_free_bytes >= required_follower_bytes - && candidate.capacity.free_memory_bytes != 0 - && candidate.capacity.free_disk_bytes != 0 - && candidate.capacity.job_credits != 0 + && accepts_log_enrollment(candidate, required_follower_bytes, now_ms) }) .map(|candidate| { let mut hasher = blake3::Hasher::new(); @@ -307,9 +319,9 @@ impl NodeDirectory { } let mut selected = selected_advertisements .into_iter() - .map(|advertisement| advertisement.node) + .cloned() .collect::>(); - selected.sort_unstable_by(|left, right| left.as_bytes().cmp(right.as_bytes())); + selected.sort_unstable_by(|left, right| left.node.as_bytes().cmp(right.node.as_bytes())); Ok(selected) } @@ -506,7 +518,18 @@ impl NodeDirectory { Ok(_) => Ok(()), Err(update_error) => match self.load_record_at(&path).await? { None => Ok(()), - Some((NodeRecord::Tombstone(current), _)) if current.claimant.is_none() => Ok(()), + Some((NodeRecord::Tombstone(current), _)) if current.claimant.is_none() => { + // Stale collection fences the boot but retains its log. + // A token captured before recruitment must not turn that + // unresolved authority into successful planned withdrawal. + if current.log.is_some() { + Err(Error::Node( + "node log must be sealed before session withdrawal", + )) + } else { + Ok(()) + } + } Some((NodeRecord::Tombstone(_), _)) => Err(Error::Fenced), Some((NodeRecord::Advertisement(current), _)) if *current == observed.advertisement => @@ -675,3 +698,16 @@ impl NodeDirectory { Err(Error::Node("node session changed during heartbeat refresh")) } } + +pub(super) fn accepts_log_enrollment( + candidate: &NodeAdvertisement, + required_follower_bytes: u64, + now_ms: i64, +) -> bool { + candidate.accepts_new_roles(now_ms) + && candidate.capacity.log_protocol == NODE_LOG_PROTOCOL_VERSION + && candidate.capacity.follower_free_bytes >= required_follower_bytes + && candidate.capacity.free_memory_bytes != 0 + && candidate.capacity.free_disk_bytes != 0 + && candidate.capacity.job_credits != 0 +} diff --git a/crates/cellule-runtime/src/node/directory/closure/fence.rs b/crates/cellule-runtime/src/node/directory/closure/fence.rs new file mode 100644 index 00000000..2f40a528 --- /dev/null +++ b/crates/cellule-runtime/src/node/directory/closure/fence.rs @@ -0,0 +1,94 @@ +use super::*; + +/// Permanent canonical fence of an exact physical boot, independent of recovery. +/// +/// This proves no process termination, accepted-work joining, log retirement, +/// Cell relocation or withdrawal. Applications obtain original process evidence +/// separately before using a fence to retain affected Cells. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct NodeSessionFence { + node: NodeId, + session: SessionId, + expires_at_ms: i64, + retired_at_ms: i64, +} + +impl NodeSessionFence { + /// Physical node retained in the permanent tombstone. + #[must_use] + pub const fn node(&self) -> NodeId { + self.node + } + /// Original boot session that cannot renew its advertisement. + #[must_use] + pub const fn session(&self) -> SessionId { + self.session + } + /// Original advertised expiry, without claiming process closure. + #[must_use] + pub const fn expires_at_ms(&self) -> i64 { + self.expires_at_ms + } + /// First permanent fencing time, unaffected by claims or recovery. + #[must_use] + pub const fn retired_at_ms(&self) -> i64 { + self.retired_at_ms + } + + pub(super) fn from_record(record: &NodeTombstone) -> Self { + Self { + node: record.node, + session: record.session, + expires_at_ms: record.expires_at_ms, + retired_at_ms: record.retired_at_ms, + } + } + + pub(super) fn from_closure(closure: &NodeSessionClosure) -> Self { + Self { + node: closure.node, + session: closure.session, + expires_at_ms: closure.expires_at_ms, + retired_at_ms: closure.retired_at_ms, + } + } +} + +impl NodeDirectory { + /// Reads the permanent original boot fence while its leader log may still + /// need recovery. It starts no claim, recovery, retirement or process effect. + /// Authenticate the live claimant and original signing identity separately. + pub async fn fenced_session( + &self, + node: NodeId, + session: SessionId, + claimant: SessionId, + now_ms: i64, + ) -> Result { + let current = self + .fenced_session_record(node, session, claimant, now_ms) + .await?; + Ok(NodeSessionFence::from_record(¤t)) + } + + pub(super) async fn fenced_session_record( + &self, + node: NodeId, + session: SessionId, + claimant: SessionId, + now_ms: i64, + ) -> Result> { + if now_ms < 0 || claimant == session { + return Err(Error::Fenced); + } + self.load(claimant, now_ms).await?.ok_or(Error::Fenced)?; + let path = self.layout.node_path(session.as_bytes()); + let Some((NodeRecord::Tombstone(current), _)) = self.load_record_at(&path).await? else { + return Err(Error::Control("node session has no permanent fence")); + }; + if current.node != node || current.session != session || current.retired_at_ms > now_ms { + return Err(Error::Fenced); + } + Ok(current) + } +} diff --git a/crates/cellule-runtime/src/node/directory/closure/mod.rs b/crates/cellule-runtime/src/node/directory/closure/mod.rs new file mode 100644 index 00000000..2160894a --- /dev/null +++ b/crates/cellule-runtime/src/node/directory/closure/mod.rs @@ -0,0 +1,84 @@ +//! Permanent session fencing with no outstanding leader-log retirement. +use super::*; + +mod fence; +pub use fence::NodeSessionFence; + +/// Canonical permanent fence for one original physical boot and its leader log. +/// A Retired log retains its original ensemble and pinned manifest. This proves +/// neither process termination nor Cell relocation, foreign roles or withdrawal. +/// Applications must separately prove those before completing maintenance. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct NodeSessionClosure { + node: NodeId, + session: SessionId, + expires_at_ms: i64, + retired_at_ms: i64, + log: Option, +} + +impl NodeSessionClosure { + /// The same original permanent fence, without terminal-log assertions. + #[must_use] + pub fn fence(&self) -> NodeSessionFence { + NodeSessionFence::from_closure(self) + } + /// Physical identity retained by the canonical tombstone. + #[must_use] + pub const fn node(&self) -> NodeId { + self.node + } + /// Original fenced session, never a replacement boot. + #[must_use] + pub const fn session(&self) -> SessionId { + self.session + } + /// Original advertised expiry; expiry alone supplies no closure. + #[must_use] + pub const fn expires_at_ms(&self) -> i64 { + self.expires_at_ms + } + /// First permanent fencing time, independent of recovery claim renewal. + #[must_use] + pub const fn retired_at_ms(&self) -> i64 { + self.retired_at_ms + } + /// Exact terminal original log, or no enrolled log in the canonical fence. + #[must_use] + pub const fn log(&self) -> Option<&NodeLogStatus> { + self.log.as_ref() + } +} + +impl NodeDirectory { + /// Freshly confirms the exact physical boot's permanent fence and terminal + /// leader log. Missing/live/expired advertisements and Open/Recovering/Sealed + /// logs refuse; even an inactive enrolled log must complete retirement. + /// Authenticate the live claimant before calling. The retained claimant and + /// its claim expiry are excluded from this immutable session closure. + pub async fn closed_session( + &self, + node: NodeId, + session: SessionId, + claimant: SessionId, + now_ms: i64, + ) -> Result { + let current = self + .fenced_session_record(node, session, claimant, now_ms) + .await?; + if current + .log + .as_ref() + .is_some_and(|log| log.phase() != NodeLogPhase::Retired) + { + return Err(Error::Control("node session log retirement is incomplete")); + } + Ok(NodeSessionClosure { + node: current.node, + session: current.session, + expires_at_ms: current.expires_at_ms, + retired_at_ms: current.retired_at_ms, + log: current.log.clone(), + }) + } +} diff --git a/crates/cellule-runtime/src/node/directory/enrollment.rs b/crates/cellule-runtime/src/node/directory/enrollment.rs new file mode 100644 index 00000000..a434c17b --- /dev/null +++ b/crates/cellule-runtime/src/node/directory/enrollment.rs @@ -0,0 +1,359 @@ +//! Read-only follower selection and exact CAS inputs for durable enrollment owners. +use super::*; + +/// A complete selected ensemble and its original signed physical boot identities. +/// Preparation sends no frames and changes no authority. Metadata grants no +/// admission: an embedding fleet producer must journal every member Pending +/// before committing the retained attempt. +#[derive(Clone)] +pub struct PreparedNodeLogEnrollment { + directory_scope: [u8; 16], + source: NodeAdvertisement, + followers: Vec, + log: NodeLogStatus, + required_follower_bytes: u64, +} + +impl PreparedNodeLogEnrollment { + /// Returns the original signed leader boot, including its physical identity. + #[must_use] + pub const fn source(&self) -> &NodeAdvertisement { + &self.source + } + + /// Returns every selected signed follower boot in canonical physical-node order. + #[must_use] + pub fn followers(&self) -> &[NodeAdvertisement] { + &self.followers + } + + /// Returns the fixed epoch and complete sorted member set to be enrolled. + #[must_use] + pub const fn log(&self) -> &NodeLogStatus { + &self.log + } +} + +/// One immutable conditional enrollment write, prepared without side effects. +/// Retain this exact attempt before awaiting the CAS. Retry/inspection never +/// chooses another ensemble or boot. A missing log is not a refusal proof. +#[derive(Clone)] +pub struct NodeLogEnrollmentAttempt { + prepared: PreparedNodeLogEnrollment, + observed: VersionedNodeAdvertisement, + next: NodeAdvertisement, +} + +impl NodeLogEnrollmentAttempt { + /// Digests the original signed source version, selected boot snapshots and + /// epoch for a durable producer's request-bound evidence. Grants no admission. + pub fn evidence_digest(&self) -> Result { + enrollment_digest( + &self.prepared, + &self.observed, + b"cellule.node-log.attempt.v1\0", + ) + } + /// Returns the complete original selection retained by this attempt. + #[must_use] + pub const fn prepared(&self) -> &PreparedNodeLogEnrollment { + &self.prepared + } + + /// Returns the fresh signed source and CAS version captured before dispatch. + #[must_use] + pub const fn observed(&self) -> &VersionedNodeAdvertisement { + &self.observed + } +} + +/// Canonical observation of this exact attempt's leader boot, epoch and ensemble. +/// It proves enrollment, not follower fsync, fleet registry publication, current +/// redundancy policy, or retirement. The original selected follower boots remain +/// pinned even if they subsequently expire or are replaced. +#[derive(Clone)] +pub struct NodeLogEnrollmentProof { + prepared: PreparedNodeLogEnrollment, + enrollment: VersionedNodeAdvertisement, +} + +impl NodeLogEnrollmentProof { + /// Digests the original selected scope and its checked canonical observation. + pub fn evidence_digest(&self) -> Result { + enrollment_digest( + &self.prepared, + &self.enrollment, + b"cellule.node-log.enrolled.v1\0", + ) + } + /// Returns the original scope and every selected follower boot. + #[must_use] + pub const fn prepared(&self) -> &PreparedNodeLogEnrollment { + &self.prepared + } + + /// Returns the exact canonical leader version establishing the enrollment. + #[must_use] + pub const fn enrollment(&self) -> &VersionedNodeAdvertisement { + &self.enrollment + } +} + +/// Proof that a conditional fence won against this exact enrollment attempt. +/// The fence and enrollment use the same original source version, so this +/// attempt's delayed CAS can no longer enroll followers. This proves neither +/// the absence of other enrollments nor retirement of any existing epoch. +#[derive(Clone)] +pub struct NodeLogEnrollmentRefusalProof { + prepared: PreparedNodeLogEnrollment, + refusal: VersionedNodeAdvertisement, +} + +impl NodeLogEnrollmentRefusalProof { + /// Digests the selected scope and original-token conditional refusal receipt. + pub fn evidence_digest(&self) -> Result { + enrollment_digest( + &self.prepared, + &self.refusal, + b"cellule.node-log.refused.v1\0", + ) + } + /// Returns the original selected scope whose one attempt is now fenced. + #[must_use] + pub const fn prepared(&self) -> &PreparedNodeLogEnrollment { + &self.prepared + } + + /// Returns the canonical source version produced by the conditional fence. + #[must_use] + pub const fn refusal(&self) -> &VersionedNodeAdvertisement { + &self.refusal + } +} + +fn enrollment_digest( + prepared: &PreparedNodeLogEnrollment, + source: &VersionedNodeAdvertisement, + domain: &[u8], +) -> Result { + let mut digest = blake3::Hasher::new(); + digest.update(domain); + digest.update(&prepared.source.encode()?); + digest.update(&source.advertisement.encode()?); + // Preserve the immutable conditional-write token as well as signed bytes. + // Lengths and option markers keep arbitrary provider tokens unambiguous. + for token in [&source.token.e_tag, &source.token.version] { + match token { + Some(token) => { + digest.update(&[1]); + digest.update(&(token.len() as u64).to_le_bytes()); + digest.update(token.as_bytes()); + } + None => { + digest.update(&[0]); + } + } + } + digest.update(&prepared.log.epoch().to_le_bytes()); + for follower in &prepared.followers { + digest.update(&follower.encode()?); + } + Ok(Digest::from_bytes(*digest.finalize().as_bytes())) +} + +impl NodeDirectory { + /// Selects one complete ensemble without executing its authority CAS. + /// A producer can now persist every selected follower obligation before the + /// first native effect, rather than registering only after recruitment. + /// Providers must assign a never-reused, monotonically advancing epoch for + /// this signed leader boot, including after closure and ambiguous results. + pub async fn prepare_log_enrollment( + &self, + observed: &VersionedNodeAdvertisement, + log_epoch: u64, + required_follower_bytes: u64, + live_node_limit: usize, + now_ms: i64, + ) -> Result> { + self.validate(&observed.advertisement, now_ms)?; + if observed.advertisement.log.is_some() { + return Err(Error::Node("node session already has an enrolled log")); + } + let followers = self + .select_log_advertisements( + observed.advertisement.session, + required_follower_bytes, + now_ms, + live_node_limit, + ) + .await?; + if followers.is_empty() { + return Ok(None); + } + let log = NodeLogStatus::open( + observed.advertisement.node, + log_epoch, + followers.iter().map(|member| member.node).collect(), + )?; + Ok(Some(PreparedNodeLogEnrollment { + directory_scope: self.inventory_scope, + source: observed.advertisement.clone(), + followers, + log, + required_follower_bytes, + })) + } + + /// Revalidates the original boots and captures a fresh source CAS version. + /// Heartbeats may change mutable measurements; replacement boots and an + /// already-enrolled source are rejected. Members are never reselected. + /// Prepare this before journal acceptance, then retain it through settlement. + pub async fn prepare_log_enrollment_attempt( + &self, + prepared: &PreparedNodeLogEnrollment, + now_ms: i64, + ) -> Result { + self.validate_enrollment_scope(prepared)?; + let observed = self + .load_if_live(prepared.source.session, now_ms) + .await? + .ok_or(Error::Node("node-log leader is not live"))?; + if !same_boot_identity(&prepared.source, &observed.advertisement) { + return Err(Error::Node("node-log leader boot differs from preparation")); + } + if observed.advertisement.log.is_some() { + return Err(Error::Node("node session already has an enrolled log")); + } + for original in &prepared.followers { + let current = self + .load_if_live(original.session, now_ms) + .await? + .ok_or(Error::Node("prepared node-log follower is not live"))?; + let current = ¤t.advertisement; + if !same_boot_identity(original, current) { + return Err(Error::Node( + "node-log follower boot differs from preparation", + )); + } + if !advertisement::accepts_log_enrollment( + current, + prepared.required_follower_bytes, + now_ms, + ) { + return Err(Error::Node( + "prepared node-log follower cannot receive enrollment", + )); + } + } + self.enrollment_attempt(prepared, observed) + } + + pub(super) fn enrollment_attempt( + &self, + prepared: &PreparedNodeLogEnrollment, + observed: VersionedNodeAdvertisement, + ) -> Result { + let mut next = observed.advertisement.clone(); + next.generation = next + .generation + .checked_add(1) + .ok_or(Error::Node("node session generation overflow"))?; + next.log = Some(prepared.log.clone()); + Ok(NodeLogEnrollmentAttempt { + prepared: prepared.clone(), + observed, + next, + }) + } + + /// Executes only the retained attempt's conditional write and fixed ensemble. + /// Fleet producers must first accept Pending for every original participant. + /// An error, timeout or dropped waiter is ambiguous; retain the same attempt + /// and inspect it. Starting a different enrollment cannot settle this one. + pub async fn commit_log_enrollment( + &self, + attempt: &NodeLogEnrollmentAttempt, + now_ms: i64, + ) -> Result { + self.validate_enrollment_scope(&attempt.prepared)?; + self.validate(&attempt.observed.advertisement, now_ms)?; + let enrollment = self + .update_advertisement(&attempt.observed, attempt.next.clone(), now_ms) + .await?; + self.enrollment_proof(&attempt.prepared, enrollment)? + .ok_or(Error::Node( + "node-log enrollment CAS returned a different ensemble", + )) + } + + /// Freshly reconciles the exact original attempt across activation/heartbeats. + /// `None` leaves the outcome unknown. Absence, another epoch, expiry or a + /// tombstone cannot prove nonexecution or settle any registry obligation. + pub async fn inspect_log_enrollment( + &self, + attempt: &NodeLogEnrollmentAttempt, + now_ms: i64, + ) -> Result> { + self.validate_enrollment_scope(&attempt.prepared)?; + let Some(current) = self + .load_if_live(attempt.prepared.source.session, now_ms) + .await? + else { + return Ok(None); + }; + self.enrollment_proof(&attempt.prepared, current) + } + + /// Conditionally fences only the original attempt using its exact CAS token. + /// The no-log successor competes with the enrollment write on the same + /// version; at most one can succeed. Never rebase a refusal fence. A failure, + /// lost reply, newer empty record or missing source leaves the outcome unknown. + pub async fn fence_log_enrollment( + &self, + attempt: &NodeLogEnrollmentAttempt, + now_ms: i64, + ) -> Result { + self.validate_enrollment_scope(&attempt.prepared)?; + self.validate(&attempt.observed.advertisement, now_ms)?; + let mut next = attempt.observed.advertisement.clone(); + next.generation = attempt.next.generation; + // The original observed source is unenrolled. Publishing its unchanged + // body with the next generation invalidates the delayed enrollment CAS. + let refusal = self + .update_advertisement(&attempt.observed, next, now_ms) + .await?; + Ok(NodeLogEnrollmentRefusalProof { + prepared: attempt.prepared.clone(), + refusal, + }) + } + + fn validate_enrollment_scope(&self, prepared: &PreparedNodeLogEnrollment) -> Result<()> { + if prepared.directory_scope != self.inventory_scope { + return Err(Error::Node( + "node-log enrollment belongs to another directory", + )); + } + Ok(()) + } + + fn enrollment_proof( + &self, + prepared: &PreparedNodeLogEnrollment, + current: VersionedNodeAdvertisement, + ) -> Result> { + if !same_boot_identity(&prepared.source, ¤t.advertisement) { + return Err(Error::Node("node-log leader boot differs from preparation")); + } + let Some(log) = ¤t.advertisement.log else { + return Ok(None); + }; + if log.epoch() != prepared.log.epoch() || log.members() != prepared.log.members() { + return Ok(None); + } + Ok(Some(NodeLogEnrollmentProof { + prepared: prepared.clone(), + enrollment: current, + })) + } +} diff --git a/crates/cellule-runtime/src/node/directory/inventory/mod.rs b/crates/cellule-runtime/src/node/directory/inventory/mod.rs new file mode 100644 index 00000000..c7699645 --- /dev/null +++ b/crates/cellule-runtime/src/node/directory/inventory/mod.rs @@ -0,0 +1,248 @@ +//! Bounded discovery of authoritative log references, including failed owners. + +use std::collections::{BTreeMap, HashSet}; + +use super::*; + +const MAX_PAGE_ENTRIES: usize = 128; + +/// Whether the authoritative log belongs to a live, expired, or fenced session. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum LogLeaderState { + /// A signed advertisement remains within its declared lifetime. + Live, + /// The signed advertisement expired; its log remains an obligation. + Expired, + /// An authoritative tombstone fences the original owner. + Fenced, +} + +/// One authoritative current epoch naming the inspected physical follower node. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct FollowerLogObservation { + /// Original leader boot session, retained through tombstone recovery. + pub leader: SessionId, + /// Physical node that enrolled the log. + pub leader_node: NodeId, + /// Liveness classification is advisory; recovery requires its ordinary claim. + pub leader_state: LogLeaderState, + /// Exact current log authority, including phase, members, and coverage. + pub log: NodeLogStatus, +} + +/// Continuation bound to one directory instance, member, and discovered topology. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct LogInventoryCursor { + topology: Digest, + after: SessionId, +} + +impl LogInventoryCursor { + /// Encodes one fixed-width application continuation. + #[must_use] + pub fn to_bytes(self) -> [u8; 48] { + let mut bytes = [0; 48]; + bytes[..32].copy_from_slice(self.topology.as_bytes()); + bytes[32..].copy_from_slice(self.after.as_bytes()); + bytes + } + /// Decodes the fixed width; the directory rechecks topology during use. + pub fn from_bytes(bytes: &[u8]) -> Result { + let bytes: &[u8; 48] = bytes + .try_into() + .map_err(|_| Error::Node("invalid log inventory cursor width"))?; + let mut topology = [0; 32]; + topology.copy_from_slice(&bytes[..32]); + let mut after = [0; 16]; + after.copy_from_slice(&bytes[32..]); + if after == [0; 16] { + return Err(Error::Node("zero log inventory continuation session")); + } + Ok(Self { + topology: Digest::from_bytes(topology), + after: SessionId::from_bytes(after), + }) + } +} + +/// Bounded discovered references. This is not an atomic membership snapshot. +pub struct LogInventoryPage { + member: NodeId, + topology: Digest, + observed_at_ms: i64, + total_logs: usize, + entries: Vec, + next: Option, +} + +impl LogInventoryPage { + /// Returns the physical follower node whose references were discovered. + #[must_use] + pub const fn member(&self) -> NodeId { + self.member + } + /// Returns the fingerprint for this directory and discovered log topology. + #[must_use] + pub const fn topology(&self) -> Digest { + self.topology + } + /// Returns the caller's capture time, without renewing any owner lease. + #[must_use] + pub const fn observed_at_ms(&self) -> i64 { + self.observed_at_ms + } + /// Counts all matching current logs, including logs outside this page. + #[must_use] + pub const fn total_logs(&self) -> usize { + self.total_logs + } + /// Returns at most 128 references, sorted by leader session. + #[must_use] + pub fn entries(&self) -> &[FollowerLogObservation] { + &self.entries + } + /// Continues a scan only while discovered epoch membership still matches. + #[must_use] + pub const fn next(&self) -> Option { + self.next + } +} + +impl NodeDirectory { + /// Discovers current log references to a follower, including expired records + /// and fenced/recovering tombstones that `live` intentionally omits. + /// + /// Scans stream at most 10,000 records and retain only the requested window. + /// Cursor changes require a restart. Object listings alone are not atomic: + /// the fleet observer must additionally prove its membership/enrollment + /// barrier and reread exact epoch authority before maintenance finalization. + /// Wrap the call in the caller's deadline; failure yields no partial page. + pub async fn follower_logs_page( + &self, + member: NodeId, + cursor: Option, + limit: usize, + now_ms: i64, + ) -> Result { + if member.as_bytes() == &[0; 16] || !(1..=MAX_PAGE_ENTRIES).contains(&limit) || now_ms < 0 { + return Err(Error::Node("invalid follower log inventory bounds")); + } + let prefix = self.layout.node_directory_path(); + let mut objects = self.layout.store().inner().list(Some(&prefix)); + let mut seen = HashSet::with_capacity(MAX_LIVE_NODE_RECORDS); + let mut window = BTreeMap::new(); + let mut combined = [0_u8; 32]; + let mut total_logs = 0; + let mut cursor_found = cursor.is_none(); + while let Some(object) = objects.next().await { + if seen.len() == MAX_LIVE_NODE_RECORDS { + return Err(Error::Capacity("follower log inventory record bound")); + } + let meta = object.map_err(|error| map_object_store_error(error, prefix.as_ref()))?; + let Some((record, _)) = self.load_record_at(&meta.location).await? else { + // A disappearing record does not establish a complete scan. + return Err(Error::Node("log inventory record changed during scan")); + }; + let session = record.session(); + validate_record_path(&self.layout, session, &meta.location)?; + if !seen.insert(session) { + return Err(Error::Node("duplicate log inventory session")); + } + let (node, leader_state) = match &record { + NodeRecord::Advertisement(advertisement) => { + advertisement.validate_shape()?; + advertisement.verify_signature()?; + if advertisement.fleet != self.fleet + || advertisement.issued_at_ms > now_ms.saturating_add(MAX_CLOCK_SKEW_MS) + { + return Err(Error::Node("log inventory fleet or issue time differs")); + } + ( + advertisement.node, + if advertisement.expires_at_ms > now_ms { + LogLeaderState::Live + } else { + LogLeaderState::Expired + }, + ) + } + NodeRecord::Tombstone(tombstone) => (tombstone.node, LogLeaderState::Fenced), + }; + let Some(log) = record.log().filter(|log| log.members().contains(&member)) else { + continue; + }; + total_logs += 1; + // Commutative digest makes arbitrary object listing order harmless. + // Duplicate sessions are rejected separately. Volatile coverage and + // heartbeat times do not reset topology; exact actions recheck them. + let mut hash = blake3::Hasher::new(); + hash.update(session.as_bytes()); + hash.update(node.as_bytes()); + hash.update(&log.epoch().to_le_bytes()); + hash.update(&[match log.phase() { + NodeLogPhase::Open => 1, + NodeLogPhase::Recovering => 2, + NodeLogPhase::Sealed => 3, + NodeLogPhase::Retired => 4, + }]); + for enrolled in log.members() { + hash.update(enrolled.as_bytes()); + } + for (combined, byte) in combined.iter_mut().zip(hash.finalize().as_bytes()) { + *combined ^= byte; + } + if cursor.is_some_and(|cursor| cursor.after == session) { + cursor_found = true; + } + if cursor.is_some_and(|cursor| session.as_bytes() <= cursor.after.as_bytes()) { + continue; + } + window.insert( + *session.as_bytes(), + FollowerLogObservation { + leader: session, + leader_node: node, + leader_state, + log: log.clone(), + }, + ); + if window.len() > limit + 1 { + window.pop_last(); + } + } + let mut hash = blake3::Hasher::new(); + hash.update(b"cellule-authoritative-log-inventory-v1"); + hash.update(&self.inventory_scope); + hash.update(member.as_bytes()); + hash.update(&(total_logs as u64).to_le_bytes()); + hash.update(&combined); + let topology = Digest::from_bytes(*hash.finalize().as_bytes()); + if !cursor_found || cursor.is_some_and(|cursor| cursor.topology != topology) { + return Err(Error::Node("log inventory topology changed; restart scan")); + } + let more = window.len() > limit; + if more { + window.pop_last(); + } + let entries = window.into_values().collect::>(); + let next = if more { + entries.last().map(|last| LogInventoryCursor { + topology, + after: last.leader, + }) + } else { + None + }; + Ok(LogInventoryPage { + member, + topology, + observed_at_ms: now_ms, + total_logs, + entries, + next, + }) + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-runtime/src/node/directory/inventory/tests.rs b/crates/cellule-runtime/src/node/directory/inventory/tests.rs new file mode 100644 index 00000000..f921317b --- /dev/null +++ b/crates/cellule-runtime/src/node/directory/inventory/tests.rs @@ -0,0 +1,254 @@ +use super::*; + +use ed25519_dalek::SigningKey; +use object_store::{memory::InMemory, path::Path}; + +const NOW: i64 = 1_000_000; + +fn directory() -> NodeDirectory { + NodeDirectory::new( + CellStorageLayout::new( + cellule_store::Store::new(Arc::new(InMemory::new())), + Path::from("root"), + [9; 16], + ), + Digest::from_bytes([2; 32]), + Digest::from_bytes([4; 32]), + Digest::from_bytes([5; 32]), + ) +} + +fn member() -> NodeId { + NodeId::from_bytes([9; 16]) +} + +fn advertisement(id: u8, at: i64) -> NodeAdvertisement { + let session = SessionId::from_bytes([id; 16]); + NodeAdvertisement::sign( + NodeId::from_bytes([id; 16]), + session, + "https://node.internal:8789".into(), + Digest::from_bytes([2; 32]), + Digest::from_bytes([3; 32]), + Digest::from_bytes([4; 32]), + Digest::from_bytes([5; 32]), + &SigningKey::from_bytes(&[7; 32]), + 1, + at, + at + 10_000, + vec![Digest::from_bytes([6; 32])], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + free_memory_bytes: 1_000, + free_disk_bytes: 2_000, + follower_free_bytes: if id == 9 { 2_000 } else { 0 }, + follower_retained_bytes: 0, + job_credits: 3, + log_protocol: NODE_LOG_PROTOCOL_VERSION, + }, + ) + .unwrap() +} + +async fn enroll(directory: &NodeDirectory, id: u8, at: i64) -> VersionedNodeAdvertisement { + let created = directory.create(advertisement(id, at), at).await.unwrap(); + // Inventory fixtures pin one retained ensemble independently of how live + // membership subsequently changes. Recruitment has its own public tests. + let mut next = created.advertisement().clone(); + next.generation += 1; + next.log = Some(NodeLogStatus::open(next.node, 4, vec![member()]).unwrap()); + directory + .update_advertisement(&created, next, at + 1) + .await + .unwrap() +} + +#[tokio::test] +async fn discovery_pages_include_inactive_expired_and_fenced_epochs() { + let directory = directory(); + directory.create(advertisement(9, NOW), NOW).await.unwrap(); + for id in [3, 1, 2] { + enroll(&directory, id, NOW).await; + } + let first = directory + .follower_logs_page(member(), None, 1, NOW + 2) + .await + .unwrap(); + assert_eq!(first.total_logs(), 3); + assert_eq!(first.entries()[0].leader, SessionId::from_bytes([1; 16])); + assert_eq!(first.entries()[0].leader_state, LogLeaderState::Live); + assert!(!first.entries()[0].log.active()); + let cursor = LogInventoryCursor::from_bytes(&first.next().unwrap().to_bytes()).unwrap(); + let expired = directory + .follower_logs_page(member(), Some(cursor), 128, NOW + 20_000) + .await + .unwrap(); + assert_eq!(expired.topology(), first.topology()); + assert_eq!(expired.entries().len(), 2); + assert!( + expired + .entries() + .iter() + .all(|row| row.leader_state == LogLeaderState::Expired) + ); + assert!(expired.next().is_none()); + // Expiry alone never removes obligations. Stale fencing retains their logs. + assert_eq!(directory.collect_stale(NOW + 340_001, 16).await.unwrap(), 4); + let fenced = directory + .follower_logs_page(member(), None, 128, NOW + 340_002) + .await + .unwrap(); + assert_eq!(fenced.total_logs(), 3); + assert!( + fenced + .entries() + .iter() + .all(|row| row.leader_state == LogLeaderState::Fenced) + ); + assert!( + fenced + .entries() + .iter() + .all(|row| row.log.members().contains(&member())) + ); +} + +#[tokio::test] +async fn epoch_change_member_change_or_directory_reconstruction_restarts_pagination() { + let directory = directory(); + directory.create(advertisement(9, NOW), NOW).await.unwrap(); + let first_owner = enroll(&directory, 1, NOW).await; + enroll(&directory, 2, NOW).await; + let first = directory + .follower_logs_page(member(), None, 1, NOW + 2) + .await + .unwrap(); + let cursor = first.next().unwrap(); + assert!( + directory + .follower_logs_page(NodeId::from_bytes([8; 16]), Some(cursor), 1, NOW + 3) + .await + .is_err() + ); + let reconstructed = NodeDirectory::new( + directory.layout.clone(), + directory.fleet, + directory.image, + directory.release, + ); + assert!( + reconstructed + .follower_logs_page(member(), Some(cursor), 1, NOW + 3) + .await + .is_err() + ); + let mut next = first_owner.advertisement().clone(); + next.generation += 1; + next.log = Some(NodeLogStatus::open(next.node, 5, vec![member()]).unwrap()); + directory + .update_advertisement(&first_owner, next, NOW + 3) + .await + .unwrap(); + assert!( + directory + .follower_logs_page(member(), Some(cursor), 1, NOW + 4) + .await + .is_err() + ); +} + +#[tokio::test] +async fn coverage_updates_preserve_topology_but_are_returned_fresh() { + let directory = directory(); + directory.create(advertisement(9, NOW), NOW).await.unwrap(); + enroll(&directory, 1, NOW).await; + let second = enroll(&directory, 2, NOW).await; + let page = directory + .follower_logs_page(member(), None, 1, NOW + 2) + .await + .unwrap(); + directory + .advance_log_coverage(&second, 27, NOW + 3) + .await + .unwrap(); + let next = directory + .follower_logs_page(member(), page.next(), 1, NOW + 4) + .await + .unwrap(); + assert_eq!(next.topology(), page.topology()); + assert_eq!(next.entries()[0].log.tiered_through(), 27); +} + +#[tokio::test] +async fn malformed_records_or_page_bounds_never_return_a_partial_success() { + let directory = directory(); + for limit in [0, 129, usize::MAX] { + assert!( + directory + .follower_logs_page(member(), None, limit, NOW) + .await + .is_err() + ); + } + assert!( + directory + .follower_logs_page(NodeId::from_bytes([0; 16]), None, 1, NOW) + .await + .is_err() + ); + assert!( + directory + .follower_logs_page(member(), None, 1, -1) + .await + .is_err() + ); + assert!(LogInventoryCursor::from_bytes(&[0; 48]).is_err()); + assert!(LogInventoryCursor::from_bytes(&[1; 49]).is_err()); + directory + .layout + .store() + .create_strict_with_etag( + &directory.layout.node_path(&[1; 16]), + Bytes::from_static(b"invalid record"), + ) + .await + .unwrap(); + assert!( + directory + .follower_logs_page(member(), None, 128, NOW) + .await + .is_err() + ); +} + +#[tokio::test] +async fn ordinary_recovery_claim_remains_visible_as_a_follower_obligation() { + let directory = directory(); + directory.create(advertisement(9, NOW), NOW).await.unwrap(); + enroll(&directory, 1, NOW).await; + let expired_at = NOW + 20_000; + directory + .create(advertisement(8, expired_at), expired_at) + .await + .unwrap(); + directory + .claim_expired( + SessionId::from_bytes([1; 16]), + SessionId::from_bytes([8; 16]), + expired_at, + ) + .await + .unwrap(); + let page = directory + .follower_logs_page(member(), None, 128, expired_at + 1) + .await + .unwrap(); + assert_eq!(page.total_logs(), 1); + assert_eq!(page.entries()[0].leader_state, LogLeaderState::Fenced); + assert_eq!(page.entries()[0].log.phase(), NodeLogPhase::Recovering); + assert_eq!( + page.entries()[0].log.recovery().unwrap().claimant(), + SessionId::from_bytes([8; 16]) + ); +} diff --git a/crates/cellule-runtime/src/node/directory/log.rs b/crates/cellule-runtime/src/node/directory/log.rs index 263e953c..d24bce64 100644 --- a/crates/cellule-runtime/src/node/directory/log.rs +++ b/crates/cellule-runtime/src/node/directory/log.rs @@ -43,11 +43,12 @@ impl NodeDirectory { Ok(log) } - /// Reports whether the authoritative session record still names one log epoch. + /// Reports whether authority still requires local copies of one log epoch. /// /// A missing record is corruption rather than collection authority and fails /// closed. Callers may delete an exact grace-aged retired follower lane only - /// when this returns `false`. + /// when this returns `false`. Recovered Retired tombstones preserve the epoch + /// and manifest as history after every original member confirms retirement. pub async fn log_epoch_referenced(&self, session: SessionId, epoch: u64) -> Result { if epoch == 0 { return Err(Error::Node("node-log epoch is zero")); @@ -64,7 +65,9 @@ impl NodeDirectory { advertisement.validate_shape()?; advertisement.verify_signature()?; } - Ok(record.log().is_some_and(|log| log.epoch() == epoch)) + Ok(record + .log() + .is_some_and(|log| log.epoch() == epoch && log.phase() != NodeLogPhase::Retired)) } /// Verifies a live claimant may seal or read this follower's failed-owner lane. @@ -119,30 +122,24 @@ impl NodeDirectory { live_node_limit: usize, now_ms: i64, ) -> Result> { - self.validate(&observed.advertisement, now_ms)?; - if observed.advertisement.log.is_some() { - return Err(Error::Node("node session already has an enrolled log")); - } - let members = self - .select_log_members( - observed.advertisement.session, + let Some(prepared) = self + .prepare_log_enrollment( + observed, + log_epoch, required_follower_bytes, - now_ms, live_node_limit, + now_ms, ) - .await?; - if members.is_empty() { + .await? + else { return Ok(None); - } - let mut next = observed.advertisement.clone(); - next.generation = next - .generation - .checked_add(1) - .ok_or(Error::Node("node session generation overflow"))?; - next.log = Some(NodeLogStatus::open(next.node, log_epoch, members)?); - self.update_advertisement(observed, next, now_ms) + }; + // Preserve the ordinary API's observed-version CAS contract. Managed + // producers explicitly rebase before accepting their registry records. + let attempt = self.enrollment_attempt(&prepared, observed.clone())?; + self.commit_log_enrollment(&attempt, now_ms) .await - .map(Some) + .map(|proof| Some(proof.enrollment().clone())) } /// CAS-activates the exact enrolled epoch after every member fsyncs its first batch. diff --git a/crates/cellule-runtime/src/node/directory/mod.rs b/crates/cellule-runtime/src/node/directory/mod.rs index ac379131..8ad27126 100644 --- a/crates/cellule-runtime/src/node/directory/mod.rs +++ b/crates/cellule-runtime/src/node/directory/mod.rs @@ -10,8 +10,20 @@ use crate::node::advertisement::validate_successor; use super::*; mod advertisement; +mod closure; +mod enrollment; +mod inventory; mod log; +mod recovered; mod recovery; +pub use closure::{NodeSessionClosure, NodeSessionFence}; +pub use recovered::RecoveredLogRetirementAuthorization; + +pub use enrollment::{ + NodeLogEnrollmentAttempt, NodeLogEnrollmentProof, NodeLogEnrollmentRefusalProof, + PreparedNodeLogEnrollment, +}; +pub use inventory::{FollowerLogObservation, LogInventoryCursor, LogInventoryPage, LogLeaderState}; /// Object-store directory for one fleet and compiled release. #[derive(Clone)] @@ -27,6 +39,7 @@ pub struct NodeDirectory { // Reader placement is advisory. Authority, session authentication and // maintenance continue to read their canonical records directly. reader_membership: Arc>>, + inventory_scope: [u8; 16], } /// Request verifier bound to one mTLS-authenticated enrollment observation. @@ -81,6 +94,7 @@ impl NodeDirectory { release, recovery_scan: Arc::new(RwLock::new(None)), reader_membership: Arc::new(RwLock::new(None)), + inventory_scope: rand::random(), } } @@ -201,7 +215,9 @@ impl NodeDirectory { /// Reports whether an exact session has a permanent canonical tombstone. /// /// Missing or advertised sessions return false. This does not grant ownership - /// or permit reuse of the retired identity. + /// or permit reuse of the retired identity. A tombstone can retain a recovery + /// claim or unresolved node-log authority; this alone is not clean withdrawal + /// or fleet role-settlement evidence. pub async fn is_retired(&self, session: SessionId) -> Result { let path = self.layout.node_path(session.as_bytes()); let Some((record, _)) = self.load_record_at(&path).await? else { @@ -211,6 +227,20 @@ impl NodeDirectory { Ok(matches!(record, NodeRecord::Tombstone(_))) } + /// Confirms an exact session's permanent withdrawal without retained log + /// authority or a recovery claimant. Missing or advertised sessions return + /// false. Callers still prove local shutdown and complete foreign roles; + /// this query cannot establish fleet maintenance completion by itself. + pub async fn is_withdrawn(&self, session: SessionId) -> Result { + let path = self.layout.node_path(session.as_bytes()); + let Some((record, _)) = self.load_record_at(&path).await? else { + return Ok(false); + }; + validate_record_path(&self.layout, record.session(), &path)?; + Ok(matches!(record, NodeRecord::Tombstone(current) + if current.claimant.is_none() && current.log.is_none())) + } + pub(super) fn validate(&self, advertisement: &NodeAdvertisement, now_ms: i64) -> Result<()> { advertisement.validate_at(now_ms)?; advertisement.verify_signature()?; @@ -385,13 +415,14 @@ impl RecoveryCandidateWindow { } } -pub(super) fn recovery_executor_eligible(advertisement: &NodeAdvertisement) -> bool { +pub(super) fn recovery_executor_eligible(advertisement: &NodeAdvertisement, now_ms: i64) -> bool { let capacity = advertisement.capacity(); let placement_has_headroom = advertisement.placement_capacity().is_none_or(|placement| { placement.active_cells < placement.max_active_cells && placement.running_jobs < placement.job_capacity }); - capacity.log_protocol == NODE_LOG_PROTOCOL_VERSION + advertisement.accepts_new_roles(now_ms) + && capacity.log_protocol == NODE_LOG_PROTOCOL_VERSION && capacity.free_memory_bytes != 0 && capacity.free_disk_bytes != 0 && capacity.job_credits != 0 diff --git a/crates/cellule-runtime/src/node/directory/recovered/mod.rs b/crates/cellule-runtime/src/node/directory/recovered/mod.rs new file mode 100644 index 00000000..88778219 --- /dev/null +++ b/crates/cellule-runtime/src/node/directory/recovered/mod.rs @@ -0,0 +1,166 @@ +//! Canonical authorization and complete member closure after recovered tail pinning. +use super::*; +use crate::node::log_recovery::retirement::RecoveredNodeLogRetirementProof; + +/// Fresh authorization of one original receiver's canonically recovered lane. +/// Applications authenticate the live claimant and bind `member` to the local +/// receiver before requesting this proof. It supplies no lane deletion permission. +pub struct RecoveredLogRetirementAuthorization { + member: NodeId, + leader_node: NodeId, + sealed: SealedNodeLog, +} +impl RecoveredLogRetirementAuthorization { + /// Exact physical receiver authorized by the original ensemble. + #[must_use] + pub const fn member(&self) -> NodeId { + self.member + } + /// Original physical leader from the canonical failed-session tombstone. + #[must_use] + pub const fn leader_node(&self) -> NodeId { + self.leader_node + } + /// Canonical sealed/retired epoch and pinned recovery manifest identity. + #[must_use] + pub const fn sealed(&self) -> &SealedNodeLog { + &self.sealed + } +} + +impl NodeDirectory { + /// Adopts an exact committed recovered retirement after a lost result or + /// controller restart, including after local grace collection. `None` means + /// the matching canonical epoch is still Sealed and its members must be + /// confirmed. Missing, changed or unrecovered authority is an error. + pub async fn retired_recovered_log( + &self, + sealed: &SealedNodeLog, + claimant: SessionId, + now_ms: i64, + ) -> Result> { + if now_ms < 0 || claimant == sealed.session() { + return Err(Error::Fenced); + } + self.load(claimant, now_ms).await?.ok_or(Error::Fenced)?; + let path = self.layout.node_path(sealed.session().as_bytes()); + let Some((NodeRecord::Tombstone(current), _)) = self.load_record_at(&path).await? else { + return Err(Error::Fenced); + }; + let log = current.log.as_ref().ok_or(Error::Fenced)?; + if current.session != sealed.session() + || log.retire_recovered(current.node)? != sealed.log().retire_recovered(current.node)? + { + return Err(Error::Fenced); + } + Ok( + (log.phase() == NodeLogPhase::Retired).then(|| SealedNodeLog { + session: current.session, + log: log.clone(), + }), + ) + } + + /// Rechecks the receiver's exact recovery retirement request against authority. + /// Authenticate `claimant` using application enrollment; never trust a claimed + /// sender ID. Expiry, a recovery claim or a native seal alone cannot authorize + /// retirement. Only canonical Sealed/Retired authority after pinning does so. + pub async fn authorize_recovered_log_retire( + &self, + claimant: SessionId, + member: NodeId, + leader: SessionId, + epoch: u64, + manifest: Option, + now_ms: i64, + ) -> Result { + if now_ms < 0 || epoch == 0 || claimant == leader { + return Err(Error::Fenced); + } + self.load(claimant, now_ms) + .await? + .ok_or(Error::PeerAuthorization( + "recovered log retirement requester is not live", + ))?; + let path = self.layout.node_path(leader.as_bytes()); + let Some((NodeRecord::Tombstone(current), _)) = self.load_record_at(&path).await? else { + return Err(Error::PeerAuthorization( + "recovered log leader is not fenced", + )); + }; + let log = current.log.as_ref().ok_or(Error::Fenced)?; + if current.session != leader + || log.epoch() != epoch + || log.recovery_manifest() != manifest + || !log.members().contains(&member) + || !matches!(log.phase(), NodeLogPhase::Sealed | NodeLogPhase::Retired) + { + return Err(Error::Fenced); + } + Ok(RecoveredLogRetirementAuthorization { + member, + leader_node: current.node, + sealed: SealedNodeLog { + session: leader, + log: log.clone(), + }, + }) + } + + /// CAS-closes a recovered epoch only after every original member confirms its fence. + /// Lost replies adopt the exact Retired record. The tombstone and manifest + /// remain permanent fencing/history; grace-aged local collection is separate. + pub async fn retire_recovered_log( + &self, + proof: &RecoveredNodeLogRetirementProof, + claimant: SessionId, + now_ms: i64, + ) -> Result { + if now_ms < 0 || claimant == proof.sealed().session() { + return Err(Error::Fenced); + } + self.load(claimant, now_ms).await?.ok_or(Error::Fenced)?; + let path = self.layout.node_path(proof.sealed().session().as_bytes()); + let Some((NodeRecord::Tombstone(mut current), token)) = self.load_record_at(&path).await? + else { + return Err(Error::Fenced); + }; + if current.session != proof.sealed().session() { + return Err(Error::Fenced); + } + let log = current.log.as_ref().ok_or(Error::Fenced)?; + let retired = log.retire_recovered(current.node)?; + if retired != proof.sealed().log().retire_recovered(current.node)? { + return Err(Error::Fenced); + } + if log.phase() == NodeLogPhase::Retired { + return Ok(SealedNodeLog { + session: current.session, + log: retired, + }); + } + current.log = Some(retired.clone()); + current.validate()?; + let expected = SealedNodeLog { + session: current.session, + log: retired, + }; + match self + .layout + .store() + .update(&path, Bytes::from(current.encode()?), token) + .await + { + Ok(_) => Ok(expected), + Err(source) => match self.load_record_at(&path).await? { + Some((NodeRecord::Tombstone(current), _)) + if current.session == expected.session + && current.log.as_ref() == Some(&expected.log) => + { + Ok(expected) + } + _ => Err(source.into()), + }, + } + } +} diff --git a/crates/cellule-runtime/src/node/directory/recovery.rs b/crates/cellule-runtime/src/node/directory/recovery.rs index 1960b546..458ff71b 100644 --- a/crates/cellule-runtime/src/node/directory/recovery.rs +++ b/crates/cellule-runtime/src/node/directory/recovery.rs @@ -103,7 +103,7 @@ impl NodeDirectory { .await? .ok_or(Error::Node("node recovery claimant is not live"))?; if require_recovery_eligibility - && !recovery_executor_eligible(claimant_advertisement.advertisement()) + && !recovery_executor_eligible(claimant_advertisement.advertisement(), now_ms) { return Err(Error::Capacity("node recovery claimant is not eligible")); } @@ -252,7 +252,8 @@ impl NodeDirectory { .iter() .filter_map(|member| { live.iter().find(|advertisement| { - advertisement.node() == *member && recovery_executor_eligible(advertisement) + advertisement.node() == *member + && recovery_executor_eligible(advertisement, now_ms) }) }) .min_by(|left, right| left.node().as_bytes().cmp(right.node().as_bytes())) @@ -311,7 +312,7 @@ impl NodeDirectory { .load(claimant, now_ms) .await? .ok_or(Error::Node("node recovery claimant is not live"))?; - if !recovery_executor_eligible(claimant_advertisement.advertisement()) { + if !recovery_executor_eligible(claimant_advertisement.advertisement(), now_ms) { return Ok(Vec::new()); } if claimant_node.is_some_and(|node| claimant_advertisement.advertisement.node() != node) { @@ -459,7 +460,7 @@ impl NodeDirectory { if !live_sessions.insert(node) { return Err(Error::Node("multiple live sessions advertise one node")); } - if recovery_executor_eligible(&advertisement) { + if recovery_executor_eligible(&advertisement, now_ms) { live_nodes.insert(node); } live.push(*advertisement); @@ -519,7 +520,7 @@ impl NodeDirectory { return Err(Error::Fenced); }; if let Some(log) = ¤t.log - && log.phase() == NodeLogPhase::Sealed + && matches!(log.phase(), NodeLogPhase::Sealed | NodeLogPhase::Retired) && log.recovery_manifest() == recovery_manifest { return Ok(SealedNodeLog { @@ -546,8 +547,10 @@ impl NodeDirectory { Some((NodeRecord::Tombstone(current), _)) if current.session == fenced.session && current.log.as_ref().is_some_and(|current| { - current.phase() == NodeLogPhase::Sealed - && current.recovery_manifest() == recovery_manifest + matches!( + current.phase(), + NodeLogPhase::Sealed | NodeLogPhase::Retired + ) && current.recovery_manifest() == recovery_manifest }) => { Ok(SealedNodeLog { diff --git a/crates/cellule-runtime/src/node/durability/mod.rs b/crates/cellule-runtime/src/node/durability/mod.rs index ccb89ecf..78c0c558 100644 --- a/crates/cellule-runtime/src/node/durability/mod.rs +++ b/crates/cellule-runtime/src/node/durability/mod.rs @@ -8,7 +8,8 @@ use crate::identity::NodeId; use crate::identity::SessionId; use crate::node::lease::NodeLeaseGuard; use crate::node::log::{ - CommitTicket, DurabilityGate, DurabilityProof, DurabilitySource, NodeLogRotationBarrier, + CommitTicket, DurabilityGate, DurabilityProof, DurabilitySource, NodeLogRetirementObservation, + NodeLogRetirementProof, }; use crate::node::log_shipper::{NodeLogShipper, NodeLogSubmission}; use crate::node::log_transport::NodeLogTransport; @@ -19,6 +20,23 @@ use crate::{Error, Result}; /// Implementations must serialize these mutations with heartbeat refreshes and /// reconcile an ambiguous CAS only when the exact session and log epoch match. pub trait NodeLogAuthority: Send + Sync { + /// Requires complete native member fences for every closure when fleet + /// enrollment retirement depends on them. Ordinary authorities retain + /// their best-effort rotation contract. + fn requires_confirmed_retirement(&self) -> bool { + false + } + /// Observes joined member responses before confirmation and closure. + /// This callback may retain diagnostics; it grants no retirement authority. + fn observe_retirement(&self, _observation: Arc) -> Result<()> { + Ok(()) + } + /// Retains an original shutdown failure while a managed drain keeps retrying. + /// The error includes failures before member observation, such as pending + /// object coverage. This callback grants no closure authority. + fn observe_shutdown_failure(&self, _error: Arc) -> Result<()> { + Ok(()) + } /// Activates one log epoch for this session. fn activate<'a>(&'a self, log_epoch: u64) -> BoxFuture<'a, Result<()>>; @@ -30,7 +48,10 @@ pub trait NodeLogAuthority: Send + Sync { ) -> BoxFuture<'a, Result<()>>; /// Closes one log epoch at its rotation barrier. - fn close<'a>(&'a self, barrier: &'a NodeLogRotationBarrier) -> BoxFuture<'a, Result<()>>; + fn close<'a>( + &'a self, + retirement: &'a NodeLogRetirementObservation, + ) -> BoxFuture<'a, Result<()>>; } /// Provider-neutral inputs for constructing one node-log durability epoch. @@ -87,6 +108,13 @@ impl NodeDurabilityConfig { }) } + /// Returns the configured boot, physical node and epoch before construction. + /// The provider still owns authenticated enrollment and authority validation. + #[must_use] + pub const fn identity(&self) -> (SessionId, NodeId, u64) { + (self.session, self.node, self.log_epoch) + } + /// Constructs the runtime-owned durability object for this epoch. pub fn build(self) -> Result> { let gate = DurabilityGate::new(self.session, self.node, self.log_epoch, self.members)?; @@ -104,6 +132,19 @@ impl NodeDurabilityConfig { self.node_lease, ))) } + + /// Checks construction bounds without spawning a shipper or enrolling any + /// follower. Managed producers call this before accepting journal obligations. + pub fn validate(&self) -> Result<()> { + DurabilityGate::new( + self.session, + self.node, + self.log_epoch, + self.members.iter().copied(), + )?; + NodeLogShipper::validate_limits(self.limits)?; + Ok(()) + } } /// One enrolled node-log epoch and its non-forgeable durability proof boundary. @@ -120,6 +161,8 @@ pub struct NodeDurability { activated: OnceCell<()>, object_coverage: Mutex<()>, shutdown: Mutex<()>, + retirement: std::sync::Mutex>>, + retirement_proof: OnceCell>, closed: std::sync::atomic::AtomicBool, } @@ -143,6 +186,8 @@ impl NodeDurability { activated: OnceCell::new(), object_coverage: Mutex::new(()), shutdown: Mutex::new(()), + retirement: std::sync::Mutex::new(None), + retirement_proof: OnceCell::new(), closed: std::sync::atomic::AtomicBool::new(false), } } @@ -251,19 +296,119 @@ impl NodeDurability { Ok(proof) } - /// Drains accepted frames and permanently closes fleet issuance for this epoch. + /// Returns this binding's exact enrolled log epoch. + pub fn log_epoch(&self) -> Result { + self.gate.log_epoch() + } + + /// Returns immutable configured scope, including after retirement. This + /// metadata does not establish current authority or fleet readiness. + pub fn identity(&self) -> Result<(SessionId, NodeId, u64)> { + self.gate.identity() + } + + /// Returns the latest joined member results, preserving individual errors. + /// This observation does not assert complete retirement or current authority. + pub fn retirement_observation(&self) -> Result>> { + Ok(self + .retirement + .lock() + .map_err(|_| Error::Node("node-log retirement lock poisoned"))? + .clone()) + } + + /// Drains accepted frames and closes this epoch. Ordinary authorities use + /// best-effort member retirement; managed fleet authorities retain strict + /// retries through caller deadlines. Inspect evidence before finalization. pub async fn shutdown(&self) -> Result<()> { + if !self.requires_confirmed_retirement() { + return self.shutdown_epoch(false).await.map(|_| ()); + } + // Runtime drain is a retained task. Returning a transient member error + // here would cache a terminal shutdown failure and strand this epoch. + // Keep its canonical barrier and observations through caller deadlines. + loop { + match self.shutdown_epoch(true).await { + Ok(_) => return Ok(()), + Err(error) => self.authority.observe_shutdown_failure(Arc::new(error))?, + } + tokio::time::sleep(std::time::Duration::from_millis(50)).await; + } + } + + /// Returns the authority's closure policy for retained supervisor retries. + #[must_use] + pub fn requires_confirmed_retirement(&self) -> bool { + self.authority.requires_confirmed_retirement() + } + + /// Closes this epoch only after every member confirms its exact append fence. + /// Object coverage still precedes retirement. A failed member blocks the + /// authority close and remains retryable; healthy siblings are joined first. + /// Ordinary closure without complete receipts cannot later manufacture proof. + pub async fn shutdown_for_maintenance(&self) -> Result> { + self.shutdown_epoch(true).await?.ok_or(Error::Node( + "node-log member retirement remains unconfirmed", + )) + } + + async fn shutdown_epoch( + &self, + require_confirmation: bool, + ) -> Result>> { let _shutdown = self.shutdown.lock().await; if self.closed.load(std::sync::atomic::Ordering::Acquire) { - return Ok(()); + let proof = self.retirement_proof.get().cloned(); + if require_confirmation && proof.is_none() { + return Err(Error::Node( + "node-log member retirement remains unconfirmed", + )); + } + return Ok(proof); } self.shipper.shutdown().await?; let barrier = self.gate.begin_rotation()?; - crate::node::log::retire_node_log(Arc::clone(&self.transport), &barrier).await?; - self.authority.close(&barrier).await?; + // A complete member fence precedes authority closure. Retain it before + // awaiting that CAS: an accepted close with a lost reply makes further + // retire RPCs unauthorized, although the original fences remain valid. + let retained = self.retirement_observation()?; + let observation = match retained { + Some(observation) if observation.confirmed().is_ok() => { + if observation.barrier() != &barrier { + return Err(Error::Node("retained node-log retirement barrier differs")); + } + observation + } + _ => { + let observation = Arc::new( + crate::node::log::retire_node_log(Arc::clone(&self.transport), &barrier) + .await?, + ); + *self + .retirement + .lock() + .map_err(|_| Error::Node("node-log retirement lock poisoned"))? = + Some(Arc::clone(&observation)); + observation + } + }; + self.authority + .observe_retirement(Arc::clone(&observation))?; + let confirmation = observation.confirmed(); + let proof = if require_confirmation { + Some(Arc::new(confirmation?)) + } else { + confirmation.ok().map(Arc::new) + }; + self.authority.close(&observation).await?; + if let Some(proof) = &proof { + self.retirement_proof + .set(Arc::clone(proof)) + .map_err(|_| Error::Node("node-log retirement proof already installed"))?; + } self.closed .store(true, std::sync::atomic::Ordering::Release); - Ok(()) + Ok(proof) } } diff --git a/crates/cellule-runtime/src/node/durability/tests.rs b/crates/cellule-runtime/src/node/durability/tests.rs index 8447dc82..3e13dff8 100644 --- a/crates/cellule-runtime/src/node/durability/tests.rs +++ b/crates/cellule-runtime/src/node/durability/tests.rs @@ -49,7 +49,10 @@ impl NodeLogAuthority for RecordingAuthority { }) } - fn close<'a>(&'a self, _barrier: &'a NodeLogRotationBarrier) -> BoxFuture<'a, Result<()>> { + fn close<'a>( + &'a self, + _retirement: &'a crate::node::log::NodeLogRetirementObservation, + ) -> BoxFuture<'a, Result<()>> { Box::pin(async { Ok(()) }) } } @@ -86,7 +89,10 @@ impl NodeLogAuthority for BlockingAuthority { }) } - fn close<'a>(&'a self, _barrier: &'a NodeLogRotationBarrier) -> BoxFuture<'a, Result<()>> { + fn close<'a>( + &'a self, + _retirement: &'a crate::node::log::NodeLogRetirementObservation, + ) -> BoxFuture<'a, Result<()>> { Box::pin(async { Ok(()) }) } } diff --git a/crates/cellule-runtime/src/node/log/mod.rs b/crates/cellule-runtime/src/node/log/mod.rs index d6cfabbc..e5819dc4 100644 --- a/crates/cellule-runtime/src/node/log/mod.rs +++ b/crates/cellule-runtime/src/node/log/mod.rs @@ -3,16 +3,20 @@ use std::collections::{BTreeMap, BTreeSet, HashMap, HashSet}; use std::path::Path; use std::sync::{Arc, Mutex}; -use futures_util::future::join_all; use tokio::sync::Notify; use crate::identity::NodeId; use crate::identity::SessionId; -use crate::node::log_transport::{NodeLogTransport, RetireRequest}; +use crate::node::log_transport::NodeLogTransport; use crate::node::{NodeDirectory, VersionedNodeAdvertisement}; use crate::{Error, Result}; mod recovery; +mod retirement; +pub(crate) use retirement::retire_node_log; +pub use retirement::{ + NodeLogMemberRetirement, NodeLogRetirementObservation, NodeLogRetirementProof, +}; pub use recovery::*; @@ -139,6 +143,7 @@ pub struct DurabilityGate { struct GateState { leader_session: SessionId, + leader_node: NodeId, log_epoch: u64, members: HashSet, follower_through: HashMap, @@ -174,6 +179,7 @@ impl DurabilityGate { Ok(Self { inner: Arc::new(Mutex::new(GateState { leader_session, + leader_node, log_epoch, members, follower_through, @@ -262,6 +268,18 @@ impl DurabilityGate { Ok(()) } + /// Returns this gate's immutable enrolled epoch, including after rotation. + pub fn log_epoch(&self) -> Result { + Ok(self.lock()?.log_epoch) + } + + /// Returns the immutable configured boot, physical node and epoch. This + /// metadata is not a signed authority observation or durability proof. + pub fn identity(&self) -> Result<(SessionId, NodeId, u64)> { + let state = self.lock()?; + Ok((state.leader_session, state.leader_node, state.log_epoch)) + } + pub(crate) fn shipping_scope(&self) -> Result<(SessionId, u64, Vec)> { let state = self.lock()?; if state.fenced || state.rotating { @@ -537,32 +555,6 @@ pub async fn close_node_log( directory.close_log(observed, &barrier, now_ms).await } -pub(crate) async fn retire_node_log( - transport: Arc, - barrier: &NodeLogRotationBarrier, -) -> Result<()> { - let retirements = join_all(barrier.members().iter().map(|member| { - let transport = Arc::clone(&transport); - let member = *member; - let request = RetireRequest { - leader_session: barrier.leader_session(), - log_epoch: barrier.log_epoch(), - covered_through: barrier.covered_through(), - }; - async move { transport.retire(member, request).await } - })) - .await; - let expected_base = barrier.covered_through().saturating_add(1); - for receipt in retirements.into_iter().flatten() { - if receipt.base_sequence != expected_base - || receipt.durable_through != barrier.covered_through() - { - return Err(Error::Node("follower retire receipt differs")); - } - } - Ok(()) -} - fn validate_ticket(state: &GateState, ticket: CommitTicket) -> Result<()> { if ticket.leader_session != state.leader_session || ticket.log_epoch != state.log_epoch diff --git a/crates/cellule-runtime/src/node/log/retirement.rs b/crates/cellule-runtime/src/node/log/retirement.rs new file mode 100644 index 00000000..0791a0df --- /dev/null +++ b/crates/cellule-runtime/src/node/log/retirement.rs @@ -0,0 +1,130 @@ +//! Complete member retirement evidence for maintenance, preserving ordinary rotation. +use super::*; +use crate::follower::FollowerReceipt; +use crate::node::log_transport::RetireRequest; +use futures_util::future::join_all; + +/// One authenticated member's original retirement response or transport error. +#[derive(Clone, Debug)] +pub struct NodeLogMemberRetirement { + member: NodeId, + result: std::result::Result>, +} +impl NodeLogMemberRetirement { + pub(crate) fn new(member: NodeId, result: Result) -> Self { + Self { + member, + result: result.map_err(Arc::new), + } + } + /// Returns the exact member addressed by the retirement request. + #[must_use] + pub const fn member(&self) -> NodeId { + self.member + } + /// Returns the original checked receipt or error, independently of siblings. + pub fn result(&self) -> std::result::Result> { + self.result.clone() + } +} + +/// Joined retirement responses for one object-covered, issuance-stopped epoch. +/// Failed members remain obligations; ordinary epoch closure is a separate fact. +#[derive(Clone, Debug)] +pub struct NodeLogRetirementObservation { + barrier: NodeLogRotationBarrier, + members: Vec, +} +impl NodeLogRetirementObservation { + /// Returns the exact original epoch and contiguous object-coverage barrier. + #[must_use] + pub const fn barrier(&self) -> &NodeLogRotationBarrier { + &self.barrier + } + /// Returns every member in barrier order, including every original failure. + #[must_use] + pub fn members(&self) -> &[NodeLogMemberRetirement] { + &self.members + } + /// Issues a proof only after every addressed member confirmed its exact fence. + /// The proof does not by itself assert leader-directory closure or lane deletion. + pub fn confirmed(&self) -> Result { + for member in &self.members { + if let Err(source) = &member.result { + return Err(Error::Facility { + name: "node-log-member-retirement", + source: Box::new(RetainedRetirementError(Arc::clone(source))), + }); + } + } + Ok(NodeLogRetirementProof { + barrier: self.barrier.clone(), + }) + } +} + +/// Opaque confirmation that every old member fsynced its exact append fence. +/// Produced only from joined, scope-bound transport responses after object coverage. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct NodeLogRetirementProof { + barrier: NodeLogRotationBarrier, +} +impl NodeLogRetirementProof { + /// Returns the exact original leader, epoch, complete member set and watermark. + #[must_use] + pub const fn barrier(&self) -> &NodeLogRotationBarrier { + &self.barrier + } +} + +#[derive(Debug)] +struct RetainedRetirementError(Arc); +impl std::fmt::Display for RetainedRetirementError { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + self.0.fmt(formatter) + } +} +impl std::error::Error for RetainedRetirementError { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + Some(self.0.as_ref()) + } +} + +pub(crate) async fn retire_node_log( + transport: Arc, + barrier: &NodeLogRotationBarrier, +) -> Result { + // Join healthy siblings even after another member fails. Cancellation may + // lose a reply, but never creates a proof or changes the complete member set. + let retirements = join_all(barrier.members().iter().map(|member| { + let transport = Arc::clone(&transport); + let member = *member; + let request = RetireRequest { + leader_session: barrier.leader_session(), + log_epoch: barrier.log_epoch(), + covered_through: barrier.covered_through(), + }; + async move { (member, transport.retire(member, request).await) } + })) + .await; + let expected_base = barrier.covered_through().saturating_add(1); + let mut members = Vec::with_capacity(retirements.len()); + for (member, result) in retirements { + if let Ok(receipt) = &result + && (receipt.base_sequence != expected_base + || receipt.durable_through != barrier.covered_through()) + { + // Preserve the ordinary protocol-error contract: even best-effort + // rotation cannot advance authority after a contradictory receipt. + return Err(Error::Node("follower retire receipt differs")); + } + members.push(NodeLogMemberRetirement { + member, + result: result.map_err(Arc::new), + }); + } + Ok(NodeLogRetirementObservation { + barrier: barrier.clone(), + members, + }) +} diff --git a/crates/cellule-runtime/src/node/log_recovery/mod.rs b/crates/cellule-runtime/src/node/log_recovery/mod.rs index febf001b..ebf4bbf2 100644 --- a/crates/cellule-runtime/src/node/log_recovery/mod.rs +++ b/crates/cellule-runtime/src/node/log_recovery/mod.rs @@ -27,6 +27,7 @@ const MAX_RECOVERY_CATALOG_HEAD_READS: usize = 32; const MAX_RECOVERY_PAGE_BYTES: u64 = 1 << 20; const MAX_RECOVERY_PAGE_FRAMES: usize = 4_096; +pub mod retirement; mod tail; mod witness; diff --git a/crates/cellule-runtime/src/node/log_recovery/retirement/mod.rs b/crates/cellule-runtime/src/node/log_recovery/retirement/mod.rs new file mode 100644 index 00000000..54040b89 --- /dev/null +++ b/crates/cellule-runtime/src/node/log_recovery/retirement/mod.rs @@ -0,0 +1,108 @@ +//! Complete receiver fence evidence after canonical recovery has pinned every tail. +use super::*; +use crate::node::log::NodeLogMemberRetirement; +use crate::node::log_transport::{RecoveredNodeLogTransport, RecoveredRetireRequest}; + +/// Joined responses from every original member, retaining individual failures. +pub struct RecoveredNodeLogRetirement { + sealed: SealedNodeLog, + members: Vec, +} +impl RecoveredNodeLogRetirement { + /// Original canonical recovery completion used by all requests. + #[must_use] + pub const fn sealed(&self) -> &SealedNodeLog { + &self.sealed + } + /// Every original member's response, including failed and ambiguous results. + #[must_use] + pub fn members(&self) -> &[NodeLogMemberRetirement] { + &self.members + } + /// Confirms only a complete original ensemble of durable native fences. + pub fn confirmed(&self) -> Result { + for member in &self.members { + if let Err(source) = member.result() { + return Err(Error::Facility { + name: "recovered-log-member-retirement", + source: Box::new(RetainedError(source)), + }); + } + } + Ok(RecoveredNodeLogRetirementProof { + sealed: self.sealed.clone(), + members: self.members.clone(), + }) + } +} + +/// Opaque complete native retirement evidence; canonical directory CAS is separate. +pub struct RecoveredNodeLogRetirementProof { + sealed: SealedNodeLog, + members: Vec, +} +impl RecoveredNodeLogRetirementProof { + /// Exact recovered epoch, ensemble and original pinned manifest identity. + #[must_use] + pub const fn sealed(&self) -> &SealedNodeLog { + &self.sealed + } + /// Original authenticated native fence responses retained by confirmation. + #[must_use] + pub fn members(&self) -> &[NodeLogMemberRetirement] { + &self.members + } +} + +#[derive(Debug)] +struct RetainedError(Arc); +impl std::fmt::Display for RetainedError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + self.0.fmt(f) + } +} +impl std::error::Error for RetainedError { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + Some(self.0.as_ref()) + } +} + +/// Retires the complete original ensemble after canonical recovery completion. +/// Transport receivers authenticate the requester and reread canonical authority. +/// Every member is joined even after a sibling fails. Cancelled/lost waiters +/// supply no proof; replay the same recovered epoch to adopt its native fences. +pub async fn retire_recovered_members( + transport: Arc, + sealed: &SealedNodeLog, +) -> Result { + if !matches!( + sealed.log().phase(), + NodeLogPhase::Sealed | NodeLogPhase::Retired + ) { + return Err(Error::Fenced); + } + let responses = join_all(sealed.log().members().iter().map(|member| { + let transport = Arc::clone(&transport); + let request = RecoveredRetireRequest { + sealed: sealed.clone(), + }; + let member = *member; + async move { + let result = transport + .retire_recovered(member, request) + .await + .and_then(|receipt| { + if receipt.base_sequence != receipt.durable_through.saturating_add(1) { + return Err(Error::Node("recovered follower retirement receipt differs")); + } + Ok(receipt) + }); + NodeLogMemberRetirement::new(member, result) + } + })) + .await; + Ok(RecoveredNodeLogRetirement { + sealed: sealed.clone(), + members: responses, + }) +} diff --git a/crates/cellule-runtime/src/node/log_shipper/mod.rs b/crates/cellule-runtime/src/node/log_shipper/mod.rs index 04caa910..9caee644 100644 --- a/crates/cellule-runtime/src/node/log_shipper/mod.rs +++ b/crates/cellule-runtime/src/node/log_shipper/mod.rs @@ -201,14 +201,7 @@ impl NodeLogShipper { interval: Duration, ) -> Result { let (leader, log_epoch, members) = gate.shipping_scope()?; - let batch_bytes = limits - .max_capture_bytes - .checked_add((MAX_BATCH_FRAMES as u64) * NODE_FRAME_HEADER_BYTES) - .ok_or(Error::Capacity("node-log outstanding bytes"))?; - let permits = usize::try_from(batch_bytes) - .ok() - .filter(|bytes| *bytes <= Semaphore::MAX_PERMITS && *bytes <= u32::MAX as usize) - .ok_or(Error::Capacity("node-log outstanding bytes"))?; + let (batch_bytes, permits) = Self::validate_limits(limits)?; let runtime = tokio::runtime::Handle::try_current().map_err(Error::RuntimeStart)?; let (sender, receiver) = mpsc::channel(MAX_QUEUED_SUBMISSIONS); let bytes = Arc::new(Semaphore::new(permits)); @@ -236,6 +229,18 @@ impl NodeLogShipper { }) } + pub(crate) fn validate_limits(limits: cellule_ltx::Limits) -> Result<(u64, usize)> { + let batch_bytes = limits + .max_capture_bytes + .checked_add((MAX_BATCH_FRAMES as u64) * NODE_FRAME_HEADER_BYTES) + .ok_or(Error::Capacity("node-log outstanding bytes"))?; + let permits = usize::try_from(batch_bytes) + .ok() + .filter(|bytes| *bytes <= Semaphore::MAX_PERMITS && *bytes <= u32::MAX as usize) + .ok_or(Error::Capacity("node-log outstanding bytes"))?; + Ok((batch_bytes, permits)) + } + /// Assigns a consecutive ticket and retains the encoded frames for shipping. /// /// Queue, byte admission, disk reads, and canonical encoding happen before diff --git a/crates/cellule-runtime/src/node/log_state.rs b/crates/cellule-runtime/src/node/log_state.rs index 44e6efd8..f276d221 100644 --- a/crates/cellule-runtime/src/node/log_state.rs +++ b/crates/cellule-runtime/src/node/log_state.rs @@ -328,6 +328,23 @@ impl NodeLogStatus { ) } + pub(crate) fn retire_recovered(&self, leader: NodeId) -> Result { + self.validate(leader)?; + if !matches!(self.phase, NodeLogPhase::Sealed | NodeLogPhase::Retired) { + return Err(Error::Fenced); + } + Self::from_parts( + leader, + NodeLogPhase::Retired, + self.epoch, + self.members.clone(), + self.active, + self.tiered_through, + None, + self.recovery_manifest, + ) + } + pub(crate) fn permits_append(&self, leader: NodeId, member: NodeId, epoch: u64) -> Result<()> { self.validate(leader)?; if self.phase != NodeLogPhase::Open diff --git a/crates/cellule-runtime/src/node/log_transport.rs b/crates/cellule-runtime/src/node/log_transport.rs index 06734315..d9534878 100644 --- a/crates/cellule-runtime/src/node/log_transport.rs +++ b/crates/cellule-runtime/src/node/log_transport.rs @@ -6,6 +6,7 @@ use crate::follower::FollowerStore; use crate::follower::{FollowerReceipt, FollowerTailPage}; use crate::identity::NodeId; use crate::identity::SessionId; +use crate::node::{NodeDirectory, SealedNodeLog}; use crate::{Error, Result}; /// One ordered follower append with the leader's safe truncation watermark. @@ -41,6 +42,27 @@ pub struct RetireRequest { pub covered_through: u64, } +/// Exact canonically sealed failed-owner log whose pinned tail may be retired. +/// This carries no caller-selected truncation watermark. Remote receivers must +/// authenticate the live requester and obtain fresh directory authorization. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct RecoveredRetireRequest { + /// Opaque canonical recovery completion, including its original ensemble. + pub sealed: SealedNodeLog, +} + +/// Recovery retirement capability of the same authenticated node-log transport. +/// Implement this explicit request path with the ordinary transport; do not +/// reinterpret a live-owner retirement or infer authorization from expiry. +pub trait RecoveredNodeLogTransport: NodeLogTransport { + /// Retires only after receiver-side canonical sealed-log authorization. + fn retire_recovered<'a>( + &'a self, + member: NodeId, + request: RecoveredRetireRequest, + ) -> BoxFuture<'a, Result>; +} + /// Bounded read of one already sealed follower tail. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub struct TailRequest { @@ -236,3 +258,106 @@ impl NodeLogTransport for LocalFollowerTransport { }) } } + +/// In-process recovery capability with fresh canonical receiver authorization. +/// This wraps the same ordinary store/transport. Construction pins the trusted +/// requester identity; remote applications own authenticated request encoding. +pub struct LocalRecoveredFollowerTransport { + local: LocalFollowerTransport, + directory: NodeDirectory, + claimant: SessionId, + clock: std::sync::Arc Result + Send + Sync>, +} +impl LocalRecoveredFollowerTransport { + /// Binds the existing local transport to its canonical authority and clock. + pub fn new( + local: LocalFollowerTransport, + directory: NodeDirectory, + claimant: SessionId, + clock: impl Fn() -> Result + Send + Sync + 'static, + ) -> Result { + if claimant.as_bytes() == &[0; 16] { + return Err(Error::Fenced); + } + Ok(Self { + local, + directory, + claimant, + clock: std::sync::Arc::new(clock), + }) + } +} +impl NodeLogTransport for LocalRecoveredFollowerTransport { + fn append<'a>( + &'a self, + member: NodeId, + request: AppendRequest, + ) -> BoxFuture<'a, Result> { + self.local.append(member, request) + } + fn seal<'a>( + &'a self, + member: NodeId, + request: SealRequest, + ) -> BoxFuture<'a, Result> { + self.local.seal(member, request) + } + fn retire<'a>( + &'a self, + member: NodeId, + request: RetireRequest, + ) -> BoxFuture<'a, Result> { + self.local.retire(member, request) + } + fn tail<'a>( + &'a self, + member: NodeId, + request: TailRequest, + ) -> BoxFuture<'a, Result>> { + self.local.tail(member, request) + } + fn tail_page<'a>( + &'a self, + member: NodeId, + request: TailRequest, + ) -> BoxFuture<'a, Result> { + self.local.tail_page(member, request) + } +} +impl RecoveredNodeLogTransport for LocalRecoveredFollowerTransport { + fn retire_recovered<'a>( + &'a self, + member: NodeId, + request: RecoveredRetireRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async move { + self.local.validate_member(member)?; + let authorization = self + .directory + .authorize_recovered_log_retire( + self.claimant, + member, + request.sealed.session(), + request.sealed.log().epoch(), + request.sealed.log().recovery_manifest(), + (self.clock)()?, + ) + .await?; + // Match every immutable field, including complete original ensemble, + // before handing the fresh authorization to the native lane owner. + let actual = authorization.sealed(); + if actual.session() != request.sealed.session() + || actual.log().epoch() != request.sealed.log().epoch() + || actual.log().members() != request.sealed.log().members() + || actual.log().active() != request.sealed.log().active() + || actual.log().tiered_through() != request.sealed.log().tiered_through() + { + return Err(Error::Fenced); + } + self.local + .store + .retire_recovered(member, authorization) + .await + }) + } +} diff --git a/crates/cellule-runtime/src/node/mod.rs b/crates/cellule-runtime/src/node/mod.rs index 557ca1e1..9bbf9612 100644 --- a/crates/cellule-runtime/src/node/mod.rs +++ b/crates/cellule-runtime/src/node/mod.rs @@ -27,7 +27,9 @@ use crate::node::log::NodeLogRotationBarrier; use crate::node::log_state::{NodeLogPhase, NodeLogStatus, NodeRecoveryClaim}; use crate::{Error, Result}; -const MAX_NODE_BYTES: u64 = 64 * 1024; +/// Canonical byte bound shared by signed advertisements and session tombstones. +/// Collectors can use it to admit their bounded retained node observations. +pub const MAX_NODE_BYTES: u64 = 64 * 1024; const MAX_ENDPOINT_BYTES: usize = 512; const MAX_FAILURE_DOMAIN_BYTES: usize = 253; const MAX_MODULES: usize = 128; @@ -44,6 +46,7 @@ const PLACEMENT_SIGNING_DOMAIN: &[u8] = b"crab.node-placement.v1\0"; const NODE_LOG_SELECTION_DOMAIN: &[u8] = b"crab.node-log.member.v1\0"; const RECOVERY_CANDIDATE_ROTATION_DOMAIN: &[u8] = b"crab.node-recovery-candidate.v1\0"; const PLACEMENT_SCHEMA_VERSION: u32 = 2; +const OPERATIONAL_PLACEMENT_SCHEMA_VERSION: u32 = 3; /// Current private follower-log wire and persistence protocol. pub const NODE_LOG_PROTOCOL_VERSION: u32 = 1; @@ -51,8 +54,16 @@ pub use advertisement::{ FencedNodeSession, NodeAdvertisement, NodeTakeoverProof, SealedNodeLog, VersionedNodeAdvertisement, }; -pub use capacity::{NodeCapacity, NodeFailureDomain, NodePlacementCapacity}; -pub use directory::{EnrolledPeerVerifier, NodeDirectory}; +pub use capacity::{ + NodeCapacity, NodeFailureDomain, NodeMode, NodeOperationalSample, NodePlacementCapacity, + NodePressure, +}; +pub use directory::{ + EnrolledPeerVerifier, FollowerLogObservation, LogInventoryCursor, LogInventoryPage, + LogLeaderState, NodeDirectory, NodeLogEnrollmentAttempt, NodeLogEnrollmentProof, + NodeLogEnrollmentRefusalProof, NodeSessionClosure, NodeSessionFence, PreparedNodeLogEnrollment, + RecoveredLogRetirementAuthorization, +}; mod advertisement; mod capacity; mod directory; diff --git a/crates/cellule-runtime/src/node/tests/closure.rs b/crates/cellule-runtime/src/node/tests/closure.rs new file mode 100644 index 00000000..2bd4e6d1 --- /dev/null +++ b/crates/cellule-runtime/src/node/tests/closure.rs @@ -0,0 +1,225 @@ +use super::*; + +#[tokio::test] +async fn session_closure_requires_exact_permanent_physical_fence_and_live_claimant() { + let directory = directory(); + let leader = SessionId::from_bytes([11; 16]); + let physical = NodeId::from_bytes([71; 16]); + let claimant = SessionId::from_bytes([12; 16]); + let key = SigningKey::from_bytes(&[9; 32]); + directory + .create( + advertisement_for_node_capacity( + physical, + leader, + &key, + 1, + NOW_MS, + NodeCapacity::default(), + ), + NOW_MS, + ) + .await + .unwrap(); + directory + .create( + advertisement_for(claimant, &key, 1, NOW_MS + 9_000), + NOW_MS + 9_000, + ) + .await + .unwrap(); + for now in [NOW_MS + 9_001, NOW_MS + 10_001] { + assert!( + directory + .closed_session(physical, leader, claimant, now) + .await + .is_err() + ); + } + directory + .claim_expired(leader, claimant, NOW_MS + 10_001) + .await + .unwrap(); + let closure = directory + .closed_session(physical, leader, claimant, NOW_MS + 10_002) + .await + .unwrap(); + assert_eq!(closure.node(), physical); + assert_eq!(closure.session(), leader); + assert_eq!(closure.expires_at_ms(), NOW_MS + 10_000); + assert_eq!(closure.retired_at_ms(), NOW_MS + 10_001); + assert!(closure.log().is_none()); + assert_eq!( + directory + .fenced_session(physical, leader, claimant, NOW_MS + 10_002) + .await + .unwrap(), + closure.fence() + ); + assert!( + directory + .closed_session(node(leader), leader, claimant, NOW_MS + 10_002) + .await + .is_err() + ); + assert!( + directory + .closed_session( + physical, + SessionId::from_bytes([90; 16]), + claimant, + NOW_MS + 10_002 + ) + .await + .is_err() + ); + assert!( + directory + .closed_session(physical, leader, leader, NOW_MS + 10_002) + .await + .is_err() + ); + assert!( + directory + .closed_session(physical, leader, claimant, NOW_MS + 19_001) + .await + .is_err() + ); + // Claim renewal changes no original immutable closure field. + let fenced = directory + .claim_expired(leader, claimant, NOW_MS + 10_003) + .await + .unwrap(); + directory + .refresh_recovery_claim(&fenced, NOW_MS + 10_004) + .await + .unwrap(); + assert_eq!( + directory + .closed_session(physical, leader, claimant, NOW_MS + 10_005) + .await + .unwrap(), + closure + ); +} + +#[tokio::test] +async fn permanent_session_fence_precedes_active_recovery_and_survives_claim_adoption() { + let directory = directory(); + let key = SigningKey::from_bytes(&[7; 32]); + let leader = SessionId::from_bytes([1; 16]); + let claimant = SessionId::from_bytes([2; 16]); + let original = directory + .create(advertisement_for(leader, &key, 1, NOW_MS), NOW_MS) + .await + .unwrap(); + directory + .create( + advertisement_for(claimant, &key, 1, NOW_MS + 9_000), + NOW_MS + 9_000, + ) + .await + .unwrap(); + let enrolled = directory + .recruit_log(&original, 4, 1, 2, NOW_MS + 9_001) + .await + .unwrap(); + directory + .activate_log(&enrolled, NOW_MS + 9_002) + .await + .unwrap(); + for now in [NOW_MS + 9_003, NOW_MS + 10_001] { + assert!(matches!( + directory + .fenced_session(node(leader), leader, claimant, now) + .await, + Err(Error::Control("node session has no permanent fence")) + )); + } + let fenced = directory + .claim_expired(leader, claimant, NOW_MS + 10_001) + .await + .unwrap(); + assert_eq!(fenced.log().unwrap().phase(), NodeLogPhase::Recovering); + assert!(fenced.log().unwrap().active()); + assert!(matches!( + fenced.direct_takeover(), + Err(Error::PendingPublication) + )); + let fence = directory + .fenced_session(node(leader), leader, claimant, NOW_MS + 10_002) + .await + .unwrap(); + assert_eq!(fence.node(), node(leader)); + assert_eq!(fence.session(), leader); + assert_eq!(fence.expires_at_ms(), NOW_MS + 10_000); + assert_eq!(fence.retired_at_ms(), NOW_MS + 10_001); + assert!(matches!( + directory + .closed_session(node(leader), leader, claimant, NOW_MS + 10_002) + .await, + Err(Error::Control("node session log retirement is incomplete")) + )); + for (physical, session, sender, now) in [ + (node(claimant), leader, claimant, NOW_MS + 10_002), + ( + node(leader), + SessionId::from_bytes([90; 16]), + claimant, + NOW_MS + 10_002, + ), + (node(leader), leader, leader, NOW_MS + 10_002), + (node(leader), leader, claimant, NOW_MS + 19_001), + (node(leader), leader, claimant, -1), + ] { + assert!( + directory + .fenced_session(physical, session, sender, now) + .await + .is_err() + ); + } + let replacement = SessionId::from_bytes([3; 16]); + directory + .create( + advertisement_for(replacement, &key, 1, NOW_MS + 40_001), + NOW_MS + 40_001, + ) + .await + .unwrap(); + let adopted = directory + .claim_expired(leader, replacement, NOW_MS + 40_002) + .await + .unwrap(); + assert!(adopted.claim_generation() > fenced.claim_generation()); + assert_eq!( + directory + .fenced_session(node(leader), leader, replacement, NOW_MS + 40_003) + .await + .unwrap(), + fence + ); + // Authority-only stage fixture: real overlay/bundle pinning is qualified in + // the public lifecycle suite. The permanent fence contains no manifest. + directory + .seal_recovery( + &adopted, + Some(Digest::from_bytes([81; 32])), + NOW_MS + 40_003, + ) + .await + .unwrap(); + assert_eq!( + directory + .fenced_session(node(leader), leader, replacement, NOW_MS + 40_004) + .await + .unwrap(), + fence + ); + assert!(matches!( + directory + .closed_session(node(leader), leader, replacement, NOW_MS + 40_004) + .await, + Err(Error::Control("node session log retirement is incomplete")) + )); +} diff --git a/crates/cellule-runtime/src/node/tests/enrollment.rs b/crates/cellule-runtime/src/node/tests/enrollment.rs new file mode 100644 index 00000000..7b80a00e --- /dev/null +++ b/crates/cellule-runtime/src/node/tests/enrollment.rs @@ -0,0 +1,677 @@ +//! Prepared enrollment uses the canonical selector and immutable CAS inputs. +use super::*; + +async fn ensemble() -> (NodeDirectory, VersionedNodeAdvertisement, SigningKey) { + let directory = directory(); + let key = SigningKey::from_bytes(&[7; 32]); + let source = directory + .create(advertisement(&key, 1, NOW_MS), NOW_MS) + .await + .unwrap(); + for member in [ + SessionId::from_bytes([2; 16]), + SessionId::from_bytes([3; 16]), + ] { + directory + .create(advertisement_for(member, &key, 1, NOW_MS), NOW_MS) + .await + .unwrap(); + } + (directory, source, key) +} + +#[tokio::test] +async fn preparation_is_read_only_and_rebases_heartbeats_without_reselecting_members() { + let (directory, source, key) = ensemble().await; + let prepared = directory + .prepare_log_enrollment(&source, 7, 1, 3, NOW_MS + 1) + .await + .unwrap() + .unwrap(); + assert_eq!(prepared.source(), source.advertisement()); + assert_eq!( + prepared.log().members(), + directory + .select_log_members(source.advertisement().session(), 1, NOW_MS + 1, 3) + .await + .unwrap() + ); + assert_eq!(prepared.followers().len(), 2); + for (member, signed) in prepared.log().members().iter().zip(prepared.followers()) { + assert_eq!(*member, signed.node()); + signed.verify_signature().unwrap(); + } + assert!( + directory + .load(source.advertisement().session(), NOW_MS + 1) + .await + .unwrap() + .unwrap() + .advertisement() + .log() + .is_none() + ); + directory + .refresh(&source, advertisement(&key, 2, NOW_MS + 100), NOW_MS + 100) + .await + .unwrap(); + let attempt = directory + .prepare_log_enrollment_attempt(&prepared, NOW_MS + 101) + .await + .unwrap(); + assert_eq!(attempt.observed().advertisement().progress(), 2); + assert_eq!(attempt.observed().advertisement().generation(), 2); + assert!( + directory + .inspect_log_enrollment(&attempt, NOW_MS + 101) + .await + .unwrap() + .is_none() + ); + let proof = directory + .clone() + .commit_log_enrollment(&attempt, NOW_MS + 102) + .await + .unwrap(); + assert_eq!(proof.enrollment().advertisement().generation(), 3); + assert_eq!( + proof.enrollment().advertisement().log(), + Some(prepared.log()) + ); + assert_eq!(proof.prepared().followers(), prepared.followers()); +} + +#[tokio::test] +async fn an_attempt_keeps_its_original_cas_version_when_heartbeat_races_dispatch() { + let (directory, source, key) = ensemble().await; + let prepared = directory + .prepare_log_enrollment(&source, 7, 1, 3, NOW_MS + 1) + .await + .unwrap() + .unwrap(); + let attempt = directory + .prepare_log_enrollment_attempt(&prepared, NOW_MS + 2) + .await + .unwrap(); + directory + .refresh(&source, advertisement(&key, 2, NOW_MS + 100), NOW_MS + 100) + .await + .unwrap(); + assert!( + directory + .commit_log_enrollment(&attempt, NOW_MS + 101) + .await + .is_err() + ); + assert!( + directory + .inspect_log_enrollment(&attempt, NOW_MS + 101) + .await + .unwrap() + .is_none() + ); + assert!( + directory + .load(source.advertisement().session(), NOW_MS + 101) + .await + .unwrap() + .unwrap() + .advertisement() + .log() + .is_none() + ); +} + +#[tokio::test] +async fn enrollment_inspection_reconciles_activation_coverage_and_heartbeat_then_keeps_absence_unknown() + { + let (directory, source, key) = ensemble().await; + let prepared = directory + .prepare_log_enrollment(&source, 7, 1, 3, NOW_MS + 1) + .await + .unwrap() + .unwrap(); + let attempt = directory + .prepare_log_enrollment_attempt(&prepared, NOW_MS + 2) + .await + .unwrap(); + let enrolled = directory + .commit_log_enrollment(&attempt, NOW_MS + 3) + .await + .unwrap(); + let active = directory + .activate_log(enrolled.enrollment(), NOW_MS + 4) + .await + .unwrap(); + let covered = directory + .advance_log_coverage(&active, 1, NOW_MS + 5) + .await + .unwrap(); + let refreshed = directory + .refresh(&covered, advertisement(&key, 2, NOW_MS + 100), NOW_MS + 100) + .await + .unwrap(); + let proof = directory + .inspect_log_enrollment(&attempt, NOW_MS + 101) + .await + .unwrap() + .unwrap(); + assert_eq!( + proof.enrollment().advertisement(), + refreshed.advertisement() + ); + let gate = crate::node::log::DurabilityGate::new( + source.advertisement().session(), + source.advertisement().node(), + 7, + prepared.log().members().iter().copied(), + ) + .unwrap(); + let ticket = gate.issue(1).unwrap(); + gate.prove_object(ticket).unwrap(); + let closed = directory + .close_log(&refreshed, &gate.begin_rotation().unwrap(), NOW_MS + 102) + .await + .unwrap(); + assert!( + directory + .inspect_log_enrollment(&attempt, NOW_MS + 103) + .await + .unwrap() + .is_none() + ); + directory.withdraw(&closed, NOW_MS + 104).await.unwrap(); + assert!( + directory + .inspect_log_enrollment(&attempt, NOW_MS + 105) + .await + .unwrap() + .is_none() + ); +} + +#[tokio::test] +async fn prepared_enrollment_cannot_cross_directory_instances_with_identical_fleet_scope() { + let (directory, source, _) = ensemble().await; + let foreign = NodeDirectory::new( + directory.layout.clone(), + directory.fleet, + directory.image, + directory.release, + ); + let prepared = directory + .prepare_log_enrollment(&source, 7, 1, 3, NOW_MS + 1) + .await + .unwrap() + .unwrap(); + let attempt = directory + .prepare_log_enrollment_attempt(&prepared, NOW_MS + 2) + .await + .unwrap(); + assert!( + foreign + .prepare_log_enrollment_attempt(&prepared, NOW_MS + 2) + .await + .is_err() + ); + assert!( + foreign + .commit_log_enrollment(&attempt, NOW_MS + 2) + .await + .is_err() + ); + assert!( + foreign + .inspect_log_enrollment(&attempt, NOW_MS + 2) + .await + .is_err() + ); + assert!( + directory + .inspect_log_enrollment(&attempt, NOW_MS + 2) + .await + .unwrap() + .is_none() + ); +} + +#[tokio::test] +async fn prepared_enrollment_rejects_expired_followers_even_when_new_boots_share_physical_ids() { + let (directory, source, key) = ensemble().await; + let prepared = directory + .prepare_log_enrollment(&source, 7, 1, 3, NOW_MS + 1) + .await + .unwrap() + .unwrap(); + let later = NOW_MS + 10_001; + directory + .refresh(&source, advertisement(&key, 2, later), later) + .await + .unwrap(); + for (index, original) in prepared.followers().iter().enumerate() { + directory + .create( + advertisement_for_node_capacity( + original.node(), + SessionId::from_bytes([20 + index as u8; 16]), + &key, + 1, + later, + original.capacity(), + ), + later, + ) + .await + .unwrap(); + } + assert!(matches!( + directory + .prepare_log_enrollment_attempt(&prepared, later) + .await, + Err(Error::Node("prepared node-log follower is not live")) + )); + assert!( + directory + .load(source.advertisement().session(), later) + .await + .unwrap() + .unwrap() + .advertisement() + .log() + .is_none() + ); +} + +#[tokio::test] +async fn prepared_enrollment_revalidates_selected_capacity_without_substituting_other_members() { + let (directory, source, key) = ensemble().await; + let prepared = directory + .prepare_log_enrollment(&source, 7, 1, 3, NOW_MS + 1) + .await + .unwrap() + .unwrap(); + let original = &prepared.followers()[0]; + let observed = directory + .load(original.session(), NOW_MS + 2) + .await + .unwrap() + .unwrap(); + let mut capacity = original.capacity(); + capacity.follower_free_bytes = 0; + directory + .refresh( + &observed, + advertisement_for_node_capacity( + original.node(), + original.session(), + &key, + 2, + NOW_MS + 100, + capacity, + ), + NOW_MS + 100, + ) + .await + .unwrap(); + assert!(matches!( + directory + .prepare_log_enrollment_attempt(&prepared, NOW_MS + 101) + .await, + Err(Error::Node( + "prepared node-log follower cannot receive enrollment" + )) + )); + assert_eq!(prepared.followers()[0], *original); +} + +#[tokio::test] +async fn selected_receivers_revalidate_cordon_and_pressure_while_source_can_evacuate() { + for (mode, pressure) in [ + (NodeMode::Cordoned, NodePressure::Normal), + (NodeMode::Draining, NodePressure::Normal), + (NodeMode::Active, NodePressure::Critical), + ] { + let (directory, source, key) = ensemble().await; + let prepared = directory + .prepare_log_enrollment(&source, 7, 1, 3, NOW_MS + 1) + .await + .unwrap() + .unwrap(); + let signed = |node, session, capacity, mode, pressure| { + advertisement_for_node_capacity(node, session, &key, 2, NOW_MS + 100, capacity) + .with_operational_placement( + NodePlacementCapacity { + memory_capacity_bytes: 10_000, + disk_capacity_bytes: 20_000, + max_active_cells: 10, + job_capacity: 10, + ..NodePlacementCapacity::default() + }, + NodeOperationalSample { + mode, + pressure, + sequence: 2, + observed_at_ms: NOW_MS + 100, + }, + &key, + ) + .unwrap() + }; + directory + .refresh( + &source, + signed( + source.advertisement().node(), + source.advertisement().session(), + source.advertisement().capacity(), + NodeMode::Cordoned, + NodePressure::Normal, + ), + NOW_MS + 100, + ) + .await + .unwrap(); + // Outbound replacement of existing obligations must remain available. + let attempt = directory + .prepare_log_enrollment_attempt(&prepared, NOW_MS + 101) + .await + .unwrap(); + let receiver = &prepared.followers()[0]; + let observed = directory + .load(receiver.session(), NOW_MS + 101) + .await + .unwrap() + .unwrap(); + directory + .refresh( + &observed, + signed( + receiver.node(), + receiver.session(), + receiver.capacity(), + mode, + pressure, + ), + NOW_MS + 101, + ) + .await + .unwrap(); + assert!(matches!( + directory + .prepare_log_enrollment_attempt(&prepared, NOW_MS + 102) + .await, + Err(Error::Node( + "prepared node-log follower cannot receive enrollment" + )) + )); + assert!( + directory + .inspect_log_enrollment(&attempt, NOW_MS + 102) + .await + .unwrap() + .is_none() + ); + } +} + +#[tokio::test] +async fn exact_boot_revalidation_rejects_replaced_signers_even_with_equal_node_and_session_ids() { + for replace_source in [false, true] { + let (directory, source, _) = ensemble().await; + let prepared = directory + .prepare_log_enrollment(&source, 7, 1, 3, NOW_MS + 1) + .await + .unwrap() + .unwrap(); + let original = if replace_source { + prepared.source() + } else { + &prepared.followers()[0] + }; + let observed = directory + .load(original.session(), NOW_MS + 2) + .await + .unwrap() + .unwrap(); + let replacement = advertisement_for_node_capacity( + original.node(), + original.session(), + &SigningKey::from_bytes(&[88; 32]), + 2, + NOW_MS + 100, + original.capacity(), + ); + // Deliberately bypass the canonical refresh contract to simulate a + // backing-store replacement. The replacement is validly signed itself. + directory + .layout + .store() + .update( + &directory.layout.node_path(original.session().as_bytes()), + Bytes::from(replacement.encode().unwrap()), + observed.token, + ) + .await + .unwrap(); + let result = directory + .prepare_log_enrollment_attempt(&prepared, NOW_MS + 101) + .await; + let expected = if replace_source { + "node-log leader boot differs from preparation" + } else { + "node-log follower boot differs from preparation" + }; + assert!(matches!(result, Err(Error::Node(message)) if message == expected)); + } +} + +#[tokio::test] +async fn refusal_fence_and_enrollment_compete_on_the_same_exact_source_version() { + for enrollment_wins in [false, true] { + let (directory, source, _) = ensemble().await; + let prepared = directory + .prepare_log_enrollment(&source, 7, 1, 3, NOW_MS + 1) + .await + .unwrap() + .unwrap(); + let attempt = directory + .prepare_log_enrollment_attempt(&prepared, NOW_MS + 2) + .await + .unwrap(); + if enrollment_wins { + directory + .commit_log_enrollment(&attempt, NOW_MS + 3) + .await + .unwrap(); + assert!( + directory + .fence_log_enrollment(&attempt, NOW_MS + 4) + .await + .is_err() + ); + assert!( + directory + .inspect_log_enrollment(&attempt, NOW_MS + 4) + .await + .unwrap() + .is_some() + ); + } else { + let refusal = directory + .fence_log_enrollment(&attempt, NOW_MS + 3) + .await + .unwrap(); + assert_eq!(refusal.prepared().log(), prepared.log()); + assert_eq!(refusal.refusal().advertisement().generation(), 2); + assert!(refusal.refusal().advertisement().log().is_none()); + // Delayed dispatch and duplicated fence requests keep the same token. + assert!( + directory + .commit_log_enrollment(&attempt, NOW_MS + 4) + .await + .is_err() + ); + let duplicate = directory + .fence_log_enrollment(&attempt, NOW_MS + 4) + .await + .unwrap(); + assert_eq!( + duplicate.refusal().advertisement(), + refusal.refusal().advertisement() + ); + assert!( + directory + .inspect_log_enrollment(&attempt, NOW_MS + 4) + .await + .unwrap() + .is_none() + ); + } + } +} + +#[tokio::test] +async fn newer_empty_source_does_not_authorize_a_rebased_refusal_fence() { + let (directory, source, key) = ensemble().await; + let prepared = directory + .prepare_log_enrollment(&source, 7, 1, 3, NOW_MS + 1) + .await + .unwrap() + .unwrap(); + let attempt = directory + .prepare_log_enrollment_attempt(&prepared, NOW_MS + 2) + .await + .unwrap(); + let refreshed = directory + .refresh(&source, advertisement(&key, 2, NOW_MS + 100), NOW_MS + 100) + .await + .unwrap(); + assert!( + directory + .fence_log_enrollment(&attempt, NOW_MS + 101) + .await + .is_err() + ); + let current = directory + .load(source.advertisement().session(), NOW_MS + 101) + .await + .unwrap() + .unwrap(); + assert_eq!(current.advertisement(), refreshed.advertisement()); + assert!( + directory + .inspect_log_enrollment(&attempt, NOW_MS + 101) + .await + .unwrap() + .is_none() + ); +} + +#[tokio::test(flavor = "multi_thread")] +async fn racing_refusal_and_enrollment_have_exactly_one_conditional_winner() { + let (directory, source, _) = ensemble().await; + let prepared = directory + .prepare_log_enrollment(&source, 7, 1, 3, NOW_MS + 1) + .await + .unwrap() + .unwrap(); + let attempt = directory + .prepare_log_enrollment_attempt(&prepared, NOW_MS + 2) + .await + .unwrap(); + let barrier = Arc::new(tokio::sync::Barrier::new(3)); + let enroll = { + let directory = directory.clone(); + let attempt = attempt.clone(); + let barrier = barrier.clone(); + tokio::spawn(async move { + barrier.wait().await; + directory.commit_log_enrollment(&attempt, NOW_MS + 3).await + }) + }; + let refuse = { + let directory = directory.clone(); + let attempt = attempt.clone(); + let barrier = barrier.clone(); + tokio::spawn(async move { + barrier.wait().await; + directory.fence_log_enrollment(&attempt, NOW_MS + 3).await + }) + }; + barrier.wait().await; + let enrolled = enroll.await.unwrap(); + let refused = refuse.await.unwrap(); + assert_ne!(enrolled.is_ok(), refused.is_ok()); + assert_eq!( + directory + .inspect_log_enrollment(&attempt, NOW_MS + 4) + .await + .unwrap() + .is_some(), + enrolled.is_ok() + ); + let current = directory + .load(source.advertisement().session(), NOW_MS + 4) + .await + .unwrap() + .unwrap(); + assert_eq!(current.advertisement().generation(), 2); +} + +#[tokio::test] +async fn enrollment_evidence_is_replay_stable_and_separates_native_outcomes() { + let (directory, source, _) = ensemble().await; + let prepared = directory + .prepare_log_enrollment(&source, 7, 1, 3, NOW_MS + 1) + .await + .unwrap() + .unwrap(); + let attempt = directory + .prepare_log_enrollment_attempt(&prepared, NOW_MS + 2) + .await + .unwrap(); + let original = attempt.evidence_digest().unwrap(); + assert_eq!(original, attempt.clone().evidence_digest().unwrap()); + let enrolled = directory + .commit_log_enrollment(&attempt, NOW_MS + 3) + .await + .unwrap(); + assert_ne!(original, enrolled.evidence_digest().unwrap()); + assert_eq!( + enrolled.evidence_digest().unwrap(), + directory + .inspect_log_enrollment(&attempt, NOW_MS + 4) + .await + .unwrap() + .unwrap() + .evidence_digest() + .unwrap() + ); + let (other, source, _) = ensemble().await; + let prepared = other + .prepare_log_enrollment(&source, 8, 1, 3, NOW_MS + 1) + .await + .unwrap() + .unwrap(); + let refused_attempt = other + .prepare_log_enrollment_attempt(&prepared, NOW_MS + 2) + .await + .unwrap(); + assert_ne!(original, refused_attempt.evidence_digest().unwrap()); + let refused = other + .fence_log_enrollment(&refused_attempt, NOW_MS + 3) + .await + .unwrap(); + assert_ne!( + refused_attempt.evidence_digest().unwrap(), + refused.evidence_digest().unwrap() + ); + assert_ne!( + enrolled.evidence_digest().unwrap(), + refused.evidence_digest().unwrap() + ); + assert_eq!( + refused.evidence_digest().unwrap(), + refused.clone().evidence_digest().unwrap() + ); +} diff --git a/crates/cellule-runtime/src/node/tests/log.rs b/crates/cellule-runtime/src/node/tests/log.rs index 1776e374..f571247d 100644 --- a/crates/cellule-runtime/src/node/tests/log.rs +++ b/crates/cellule-runtime/src/node/tests/log.rs @@ -2,6 +2,47 @@ use super::*; +#[tokio::test] +async fn stale_collection_with_an_unclosed_log_cannot_settle_drained_withdrawal() { + let key = SigningKey::from_bytes(&[7; 32]); + let directory = directory(); + let leader = SessionId::from_bytes([1; 16]); + let member = SessionId::from_bytes([2; 16]); + let original = directory + .create(advertisement_for(leader, &key, 1, NOW_MS), NOW_MS) + .await + .unwrap(); + directory + .create(advertisement_for(member, &key, 1, NOW_MS), NOW_MS) + .await + .unwrap(); + let enrolled = directory + .recruit_log(&original, 4, 1, 2, NOW_MS + 1) + .await + .unwrap(); + directory.activate_log(&enrolled, NOW_MS + 2).await.unwrap(); + let now = original.advertisement().expires_at_ms() + STALE_ADVERTISEMENT_RETENTION_MS; + assert_eq!(directory.collect_stale(now, 2).await.unwrap(), 2); + assert!(directory.is_retired(leader).await.unwrap()); + assert!(!directory.is_withdrawn(leader).await.unwrap()); + assert!(directory.log_epoch_referenced(leader, 4).await.unwrap()); + // The original token precedes recruitment. Collection preserves the log + // in a permanent tombstone, but supplies no planned-close or coverage proof. + assert!(matches!( + directory.withdraw(&original, now).await, + Err(Error::Node( + "node log must be sealed before session withdrawal" + )) + )); + assert!(matches!( + directory.withdraw_after_drain(&original, now).await, + Err(Error::Node( + "node log must be sealed before session withdrawal" + )) + )); + assert!(directory.log_epoch_referenced(leader, 4).await.unwrap()); +} + #[tokio::test] async fn node_log_enrollment_activation_and_coverage_are_authoritative() { let key = SigningKey::from_bytes(&[7; 32]); @@ -168,6 +209,7 @@ async fn clean_node_log_close_clears_authority_before_session_withdrawal() { .is_err() ); directory.withdraw(&closed, NOW_MS + 5).await.unwrap(); + assert!(directory.is_withdrawn(leader).await.unwrap()); assert!(directory.load(leader, NOW_MS + 6).await.unwrap().is_none()); assert!(!directory.log_epoch_referenced(leader, 4).await.unwrap()); } diff --git a/crates/cellule-runtime/src/node/tests/mod.rs b/crates/cellule-runtime/src/node/tests/mod.rs index df72ebdf..545fc09f 100644 --- a/crates/cellule-runtime/src/node/tests/mod.rs +++ b/crates/cellule-runtime/src/node/tests/mod.rs @@ -21,11 +21,29 @@ use crate::peer::{PeerOperation, PeerPrincipal, PeerSigner, wire as peer_wire}; // fixtures stay here. mod append_authorization; mod candidates; +mod closure; +mod enrollment; mod log; +mod operational; mod placement; mod records; +mod recovered_retirement; mod sessions; +// Thread-local operation counts let canonical codec tests assert a verification +// budget without timing thresholds or interference from parallel tests. +std::thread_local! { + static SIGNATURE_PASSES: std::cell::Cell = const { std::cell::Cell::new(0) }; +} + +pub(super) fn record_signature_pass() { + SIGNATURE_PASSES.with(|count| count.set(count.get() + 1)); +} + +fn signature_passes() -> usize { + SIGNATURE_PASSES.with(std::cell::Cell::get) +} + const NOW_MS: i64 = 1_000_000; fn node(session: SessionId) -> NodeId { diff --git a/crates/cellule-runtime/src/node/tests/operational.rs b/crates/cellule-runtime/src/node/tests/operational.rs new file mode 100644 index 00000000..cad4569b --- /dev/null +++ b/crates/cellule-runtime/src/node/tests/operational.rs @@ -0,0 +1,274 @@ +//! Operational placement and the schema 2 bridge signing contract. + +use super::*; +use crate::fleet::placement::{PlacementEligibility, PlacementPressure}; +use crate::node::advertisement::{ + RawCapacity, RawIdentity, RawLease, RawNodeLog, RawPlacementCapacity, validate_successor, +}; + +// The top-level schema 2 shape and field ordering from the baseline. Keeping +// this independent of RawAdvertisement catches fields accidentally emitted by +// bridge writers and demonstrates the strict old reader's rollout barrier. +#[derive(Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +struct Schema2Advertisement { + version: u8, + identity: RawIdentity, + lease: RawLease, + log: Option, + capacity: RawCapacity, + #[serde(default, skip_serializing_if = "Option::is_none")] + placement: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + placement_version: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + placement_signature: Option, +} + +fn placement() -> NodePlacementCapacity { + NodePlacementCapacity { + memory_capacity_bytes: 8_192, + disk_capacity_bytes: 16_384, + active_cells: 3, + max_active_cells: 16, + running_jobs: 2, + job_capacity: 8, + publication_backlog: 4, + hydration_backlog: 5, + primitive_backlog: 6, + } +} + +fn sample(mode: NodeMode, pressure: NodePressure) -> NodeOperationalSample { + NodeOperationalSample { + mode, + pressure, + sequence: 7, + observed_at_ms: NOW_MS, + } +} + +fn operational(mode: NodeMode, pressure: NodePressure) -> NodeAdvertisement { + let key = SigningKey::from_bytes(&[7; 32]); + advertisement(&key, 1, NOW_MS) + .with_operational_placement(placement(), sample(mode, pressure), &key) + .unwrap() +} + +#[test] +fn bridge_writer_preserves_the_schema2_serializer_and_signing_payload() { + let key = SigningKey::from_bytes(&[7; 32]); + let current = advertisement(&key, 1, NOW_MS) + .with_placement_capacity(placement(), &key) + .unwrap(); + let bytes = current.encode().unwrap(); + let mut legacy: Schema2Advertisement = serde_json::from_slice(&bytes).unwrap(); + assert_eq!(serde_json::to_vec(&legacy).unwrap(), bytes); + assert_eq!(legacy.placement_version, Some(2)); + assert_eq!(current.operational_sample(), None); + legacy.placement_signature = None; + legacy.log = None; + legacy.lease.generation.clear(); + let mut expected = PLACEMENT_SIGNING_DOMAIN.to_vec(); + expected.extend(serde_json::to_vec(&legacy).unwrap()); + assert_eq!(current.placement_signing_bytes().unwrap(), expected); + key.verifying_key() + .verify( + &expected, + &Signature::from_bytes(¤t.placement_signature), + ) + .unwrap(); + let schema3 = operational(NodeMode::Active, NodePressure::Normal) + .encode() + .unwrap(); + assert!(serde_json::from_slice::(&schema3).is_err()); + assert!(NodeAdvertisement::decode_canonical(&schema3).is_ok()); +} + +#[test] +fn signed_mode_and_every_stable_pressure_tier_drive_proactive_eligibility() { + for mode in [NodeMode::Active, NodeMode::Cordoned, NodeMode::Draining] { + for pressure in [ + NodePressure::Normal, + NodePressure::Constrained, + NodePressure::Shedding, + NodePressure::Critical, + ] { + let advertisement = operational(mode, pressure); + let decoded = + NodeAdvertisement::decode_canonical(&advertisement.encode().unwrap()).unwrap(); + assert_eq!(decoded.operational_sample(), Some(sample(mode, pressure))); + let observation = + PlacementObservation::from_signed_advertisement(&decoded, NOW_MS + 1, false) + .unwrap(); + assert_eq!(observation.draining, mode != NodeMode::Active); + assert_eq!( + observation.free_memory_bytes, + advertisement.capacity.free_memory_bytes + ); + let score = PlacementPlanner::default() + .rank( + crate::CellId::from_bytes([9; 32]), + NOW_MS + 1, + &[observation], + ) + .unwrap()[0]; + let eligible = mode == NodeMode::Active && pressure == NodePressure::Normal; + assert_eq!( + score.eligibility == PlacementEligibility::Eligible, + eligible + ); + assert_eq!(decoded.accepts_new_roles(NOW_MS + 1), eligible); + if pressure == NodePressure::Shedding { + assert_eq!(observation.pressure, PlacementPressure::Shedding); + } + } + } + assert!(NodePressure::try_from(crate::fleet::pressure::PressureState::Recovering).is_err()); +} + +#[test] +fn operational_fields_are_signed_and_partial_or_unknown_tiers_are_rejected() { + let original = operational(NodeMode::Active, NodePressure::Normal) + .encode() + .unwrap(); + for (field, value) in [ + ("mode", serde_json::json!("cordoned")), + ("pressure", serde_json::json!("shedding")), + ("sequence", serde_json::json!("8")), + ( + "observed_at_ms", + serde_json::json!((NOW_MS - 1).to_string()), + ), + ] { + let mut changed: serde_json::Value = serde_json::from_slice(&original).unwrap(); + changed["operational_sample"][field] = value; + assert!( + NodeAdvertisement::decode_canonical(&serde_json::to_vec(&changed).unwrap()).is_err() + ); + } + for field in ["mode", "pressure", "sequence", "observed_at_ms"] { + let mut changed: serde_json::Value = serde_json::from_slice(&original).unwrap(); + changed["operational_sample"] + .as_object_mut() + .unwrap() + .remove(field); + assert!( + NodeAdvertisement::decode_canonical(&serde_json::to_vec(&changed).unwrap()).is_err() + ); + } + let mut invalid = operational(NodeMode::Active, NodePressure::Normal); + invalid.operational_sample = None; + assert!(invalid.encode().is_err()); + invalid = operational(NodeMode::Active, NodePressure::Normal); + invalid.placement_version = 2; + assert!(invalid.encode().is_err()); + let wrong_key = SigningKey::from_bytes(&[99; 32]); + assert!( + operational(NodeMode::Active, NodePressure::Normal) + .with_placement_capacity(placement(), &wrong_key) + .is_err() + ); + let mut forged = operational(NodeMode::Active, NodePressure::Normal); + forged.operational_sample.as_mut().unwrap().pressure = NodePressure::Shedding; + assert!(PlacementObservation::from_signed_advertisement(&forged, NOW_MS + 1, false).is_err()); +} + +#[test] +fn heartbeat_republication_does_not_advance_the_measurement_time() { + let previous = operational(NodeMode::Active, NodePressure::Normal); + let key = SigningKey::from_bytes(&[7; 32]); + let next_at = NOW_MS + 1; + let mut next = advertisement(&key, 2, next_at) + .with_operational_placement(placement(), previous.operational_sample.unwrap(), &key) + .unwrap(); + next.generation = previous.generation + 1; + validate_successor(&previous, &next).unwrap(); + assert_eq!( + PlacementObservation::from_signed_advertisement(&next, next_at, false) + .unwrap() + .observed_at_ms, + NOW_MS + ); + let mut changed = next.clone(); + changed.operational_sample.as_mut().unwrap().pressure = NodePressure::Shedding; + assert!(validate_successor(&previous, &changed).is_err()); + changed.operational_sample.as_mut().unwrap().sequence += 1; + validate_successor(&previous, &changed).unwrap(); + changed.operational_sample.as_mut().unwrap().observed_at_ms = NOW_MS - 1; + assert!(validate_successor(&previous, &changed).is_err()); + let mut stale = next; + stale.operational_sample.as_mut().unwrap().observed_at_ms = NOW_MS - 30_001; + assert!(!stale.accepts_new_roles(next_at)); + let mut current = previous; + current.operational_sample.as_mut().unwrap().mode = NodeMode::Cordoned; + let legacy = advertisement(&key, 2, next_at) + .with_placement_capacity(placement(), &key) + .unwrap(); + assert!(validate_successor(¤t, &legacy).is_err()); +} + +#[tokio::test] +async fn closed_operational_mode_excludes_new_reader_and_follower_selection() { + let key = SigningKey::from_bytes(&[7; 32]); + let directory = directory(); + let sessions = [1, 2, 3, 4].map(|byte| SessionId::from_bytes([byte; 16])); + for (session, mode) in sessions.into_iter().zip([ + NodeMode::Active, + NodeMode::Cordoned, + NodeMode::Active, + NodeMode::Active, + ]) { + directory + .create( + advertisement_for_capacity( + session, + &key, + 1, + NOW_MS, + NodeCapacity { + free_memory_bytes: 64 << 20, + free_disk_bytes: 64 << 20, + follower_free_bytes: 32 << 20, + follower_retained_bytes: 0, + job_credits: 3, + log_protocol: NODE_LOG_PROTOCOL_VERSION, + }, + ) + .with_operational_placement(placement(), sample(mode, NodePressure::Normal), &key) + .unwrap(), + NOW_MS, + ) + .await + .unwrap(); + } + let followers = directory + .select_log_members(sessions[0], 1_000, NOW_MS + 1, 4) + .await + .unwrap(); + assert_eq!(followers.len(), 2); + assert!(!followers.contains(&node(sessions[1]))); + let readers = directory + .select_readers( + crate::CellId::from_bytes([9; 32]), + sessions[0], + Digest::from_bytes([6; 32]), + 3, + NOW_MS + 1, + 4, + ) + .await + .unwrap(); + assert_eq!(readers.len(), 2); + assert!( + !readers + .iter() + .any(|reader| reader.node() == node(sessions[1])) + ); + // Recovery executor admission uses the same explicit mode; an existing + // follower tail remains accessible through its separately fenced path. + assert!(!crate::node::directory::recovery_executor_eligible( + &operational(NodeMode::Cordoned, NodePressure::Normal), + NOW_MS + 1 + )); +} diff --git a/crates/cellule-runtime/src/node/tests/placement.rs b/crates/cellule-runtime/src/node/tests/placement.rs index 7c1deabd..5917b732 100644 --- a/crates/cellule-runtime/src/node/tests/placement.rs +++ b/crates/cellule-runtime/src/node/tests/placement.rs @@ -303,7 +303,7 @@ fn placement_schema_is_mixed_version_safe_and_fail_closed() { ); let mut future = current; - future.placement_version = 3; + future.placement_version = 4; future.placement_signature = [0; 64]; let decoded_future = NodeAdvertisement::decode_canonical(&future.encode().unwrap()).unwrap(); assert!(!decoded_future.has_signed_placement()); diff --git a/crates/cellule-runtime/src/node/tests/records.rs b/crates/cellule-runtime/src/node/tests/records.rs index ca9dc6b2..ad06e473 100644 --- a/crates/cellule-runtime/src/node/tests/records.rs +++ b/crates/cellule-runtime/src/node/tests/records.rs @@ -211,3 +211,98 @@ async fn refresh_accepts_new_signed_capacity_for_same_boot_session() { assert_eq!(refreshed.advertisement().capacity(), next_capacity); assert!(refreshed.advertisement().verify_signature().is_ok()); } + +#[test] +fn canonical_advertisement_decode_verifies_each_immutable_signature_set_once() { + for original in canonical_advertisements() { + let bytes = original.encode().unwrap(); + let before = signature_passes(); + let decoded = NodeAdvertisement::decode_canonical(&bytes).unwrap(); + assert_eq!(decoded, original); + assert_eq!(signature_passes() - before, 1); + assert_eq!(decoded.encode().unwrap(), bytes); + } +} + +fn canonical_advertisements() -> [NodeAdvertisement; 3] { + let key = SigningKey::from_bytes(&[7; 32]); + let legacy = advertisement(&key, 1, NOW_MS); + let placement = NodePlacementCapacity { + memory_capacity_bytes: 2_000, + disk_capacity_bytes: 4_000, + active_cells: 1, + max_active_cells: 8, + running_jobs: 0, + job_capacity: 3, + publication_backlog: 0, + hydration_backlog: 0, + primitive_backlog: 0, + }; + let bridge = legacy + .clone() + .with_placement_capacity(placement, &key) + .unwrap(); + let operational = legacy + .clone() + .with_operational_placement( + placement, + NodeOperationalSample { + mode: NodeMode::Draining, + pressure: NodePressure::Critical, + sequence: 1, + observed_at_ms: NOW_MS, + }, + &key, + ) + .unwrap(); + [legacy, bridge, operational] +} + +#[test] +fn canonical_advertisement_decode_keeps_both_signature_gates_and_original_errors() { + for original in canonical_advertisements() { + let bytes = original.encode().unwrap(); + let mut base: serde_json::Value = serde_json::from_slice(&bytes).unwrap(); + base["identity"]["signature"] = serde_json::Value::String("00".repeat(64)); + let before = signature_passes(); + assert!(matches!( + NodeAdvertisement::decode_canonical(&serde_json::to_vec(&base).unwrap()), + Err(Error::PeerSignature(_)), + )); + assert_eq!(signature_passes() - before, 1); + if original.has_signed_placement() { + let mut placement: serde_json::Value = serde_json::from_slice(&bytes).unwrap(); + let signature = placement["placement_signature"].as_str().unwrap(); + let mut corrupt = signature.as_bytes().to_vec(); + corrupt[0] = if corrupt[0] == b'0' { b'1' } else { b'0' }; + placement["placement_signature"] = + serde_json::Value::String(String::from_utf8(corrupt).unwrap()); + let before = signature_passes(); + assert!(matches!( + NodeAdvertisement::decode_canonical(&serde_json::to_vec(&placement).unwrap()), + Err(Error::PeerSignature(_)), + )); + assert_eq!(signature_passes() - before, 1); + } + } +} + +#[test] +fn canonical_advertisement_decode_keeps_exact_byte_and_size_gates() { + for original in canonical_advertisements() { + let mut bytes = original.encode().unwrap(); + bytes.push(b' '); + let before = signature_passes(); + assert!(matches!( + NodeAdvertisement::decode_canonical(&bytes), + Err(Error::Node("advertisement JSON is not canonical")), + )); + assert_eq!(signature_passes() - before, 1); + } + let before = signature_passes(); + assert!(matches!( + NodeAdvertisement::decode_canonical(&vec![b' '; MAX_NODE_BYTES as usize + 1]), + Err(Error::Node("advertisement exceeds 64 KiB")), + )); + assert_eq!(signature_passes(), before); +} diff --git a/crates/cellule-runtime/src/node/tests/recovered_retirement.rs b/crates/cellule-runtime/src/node/tests/recovered_retirement.rs new file mode 100644 index 00000000..bdb7faf4 --- /dev/null +++ b/crates/cellule-runtime/src/node/tests/recovered_retirement.rs @@ -0,0 +1,431 @@ +//! Canonical authorization and native fences; full tail pinning is covered by runtime recovery. +use super::*; +use crate::follower::FollowerStore; +use crate::node::log_recovery::retirement::retire_recovered_members; +use crate::node::log_transport::{ + LocalFollowerTransport, LocalRecoveredFollowerTransport, RecoveredNodeLogTransport, + RecoveredRetireRequest, +}; +use std::sync::atomic::{AtomicBool, Ordering}; + +async fn enrolled(activate: bool) -> (NodeDirectory, VersionedNodeAdvertisement, SessionId) { + let directory = directory(); + let key = SigningKey::from_bytes(&[7; 32]); + let leader = SessionId::from_bytes([1; 16]); + let original = directory + .create(advertisement_for(leader, &key, 1, NOW_MS), NOW_MS) + .await + .unwrap(); + for byte in [2, 3] { + let member = SessionId::from_bytes([byte; 16]); + directory + .create( + advertisement_for(member, &key, 1, NOW_MS + 9_000), + NOW_MS + 9_000, + ) + .await + .unwrap(); + } + let enrolled = directory + .recruit_log(&original, 4, 1, 3, NOW_MS + 9_001) + .await + .unwrap(); + let active = if activate { + directory + .activate_log(&enrolled, NOW_MS + 9_002) + .await + .unwrap() + } else { + enrolled + }; + (directory, active, SessionId::from_bytes([2; 16])) +} + +fn frames() -> Vec { + let root = tempfile::TempDir::new().unwrap(); + let limits = cellule_ltx::Limits::default(); + let mut db = cellule_ltx::Db::open(&root.path().join("cell.sqlite"), limits).unwrap(); + let mut frames = Vec::new(); + for sequence in 1..=2 { + db.transaction(|tx| { + tx.execute_batch(if sequence == 1 { + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES (1)" + } else { + "UPDATE counter SET value = 2" + }) + }) + .unwrap(); + let capture = db.capture().unwrap(); + let segment = &capture.segments[0]; + frames.push( + cellule_ltx::encode_node_frame( + cellule_ltx::NodeFrameScope { + leader_session: [1; 16], + log_epoch: 4, + node_sequence: sequence, + application: [9; 16], + cell: [6; 32], + incarnation: [7; 16], + cell_epoch: 1, + commit_sequence: sequence, + }, + segment.info().clone(), + Bytes::from(std::fs::read(segment.path()).unwrap()), + limits, + ) + .unwrap() + .encoded() + .clone(), + ); + } + frames +} + +#[tokio::test] +async fn recovered_retirement_rejects_live_recovering_foreign_and_expired_requests() { + let (directory, active, claimant) = enrolled(true).await; + let leader = active.advertisement().session(); + let member = node(claimant); + assert!( + directory + .authorize_recovered_log_retire(claimant, member, leader, 4, None, NOW_MS + 9_003) + .await + .is_err() + ); + let fenced = directory + .claim_expired(leader, claimant, NOW_MS + 10_001) + .await + .unwrap(); + assert!( + directory + .authorize_recovered_log_retire(claimant, member, leader, 4, None, NOW_MS + 10_002) + .await + .is_err() + ); + // Authority fixture only: the runtime integration case pins real overlays. + let manifest = Some(Digest::from_bytes([20; 32])); + directory + .seal_recovery(&fenced, manifest, NOW_MS + 10_002) + .await + .unwrap(); + for (sender, receiver, epoch, pointer, now) in [ + (leader, member, 4, manifest, NOW_MS + 10_003), + (claimant, node(leader), 4, manifest, NOW_MS + 10_003), + (claimant, member, 5, manifest, NOW_MS + 10_003), + (claimant, member, 4, None, NOW_MS + 10_003), + (claimant, member, 4, manifest, NOW_MS + 19_000), + ] { + assert!( + directory + .authorize_recovered_log_retire(sender, receiver, leader, epoch, pointer, now) + .await + .is_err() + ); + } + let authorization = directory + .authorize_recovered_log_retire(claimant, member, leader, 4, manifest, NOW_MS + 10_003) + .await + .unwrap(); + assert_eq!(authorization.member(), member); + assert_eq!(authorization.sealed().log().recovery_manifest(), manifest); + assert!(directory.log_epoch_referenced(leader, 4).await.unwrap()); +} + +struct Members { + peers: Vec<(NodeId, LocalRecoveredFollowerTransport)>, + lose_first: AtomicBool, + contradict_first: AtomicBool, +} +impl Members { + fn peer(&self, member: NodeId) -> &LocalRecoveredFollowerTransport { + &self + .peers + .iter() + .find(|(node, _)| *node == member) + .unwrap() + .1 + } +} +impl NodeLogTransport for Members { + fn append<'a>( + &'a self, + member: NodeId, + request: AppendRequest, + ) -> BoxFuture<'a, Result> { + self.peer(member).append(member, request) + } + fn seal<'a>( + &'a self, + member: NodeId, + request: SealRequest, + ) -> BoxFuture<'a, Result> { + self.peer(member).seal(member, request) + } + fn retire<'a>( + &'a self, + member: NodeId, + request: RetireRequest, + ) -> BoxFuture<'a, Result> { + self.peer(member).retire(member, request) + } + fn tail<'a>( + &'a self, + member: NodeId, + request: TailRequest, + ) -> BoxFuture<'a, Result>> { + self.peer(member).tail(member, request) + } +} +impl RecoveredNodeLogTransport for Members { + fn retire_recovered<'a>( + &'a self, + member: NodeId, + request: RecoveredRetireRequest, + ) -> BoxFuture<'a, Result> { + Box::pin(async move { + let mut receipt = self.peer(member).retire_recovered(member, request).await?; + if member == self.peers[0].0 { + if self.lose_first.swap(false, Ordering::AcqRel) { + return Err(Error::Node("lost recovered retirement reply")); + } + if self.contradict_first.swap(false, Ordering::AcqRel) { + receipt.base_sequence = 0; + } + } + Ok(receipt) + }) + } +} + +#[tokio::test] +async fn recovered_retirement_joins_original_members_retains_fences_and_replays_lost_results() { + let (directory, active, claimant) = enrolled(true).await; + let leader = active.advertisement().session(); + let fenced = directory + .claim_expired(leader, claimant, NOW_MS + 10_001) + .await + .unwrap(); + let sealed = directory + .seal_recovery(&fenced, Some(Digest::from_bytes([20; 32])), NOW_MS + 10_002) + .await + .unwrap(); + let first = tempfile::TempDir::new().unwrap(); + let second = tempfile::TempDir::new().unwrap(); + let roots = [first, second]; + let mut stores = Vec::new(); + let mut peers = Vec::new(); + let frames = frames(); + for (index, member) in sealed.log().members().iter().copied().enumerate() { + let store = FollowerStore::open( + roots[index].path().to_owned(), + cellule_ltx::Limits::default(), + cellule_ltx::DiskBudget::new(1 << 30), + ) + .unwrap(); + store + .append(leader, 4, frames[..2 - index].to_vec(), 0) + .await + .unwrap(); + let peer = LocalRecoveredFollowerTransport::new( + LocalFollowerTransport::new(member, store.clone()), + directory.clone(), + claimant, + || Ok(NOW_MS + 10_003), + ) + .unwrap(); + // Canonical completion cannot replace the original receiver's native seal. + assert!( + peer.retire_recovered( + member, + RecoveredRetireRequest { + sealed: sealed.clone() + } + ) + .await + .is_err() + ); + assert!(store.retained_bytes() > 8); + store.seal(leader, 4).await.unwrap(); + if index == 0 { + let marker = roots[index] + .path() + .join("followers") + .join(crate::identity::encode_hex(leader.as_bytes())) + .join("4/sealed"); + std::fs::write(&marker, 3u64.to_le_bytes()).unwrap(); + assert!(matches!( + peer.retire_recovered( + member, + RecoveredRetireRequest { + sealed: sealed.clone() + } + ) + .await, + Err(Error::Node("follower seal watermark differs")) + )); + assert!(store.retained_bytes() > 8); + std::fs::write(marker, 2u64.to_le_bytes()).unwrap(); + } + stores.push(store); + peers.push((member, peer)); + } + let transport = Arc::new(Members { + peers, + lose_first: AtomicBool::new(true), + contradict_first: AtomicBool::new(false), + }); + let observed = retire_recovered_members(transport.clone(), &sealed) + .await + .unwrap(); + assert_eq!(observed.members().len(), 2); + assert!(observed.members()[0].result().is_err()); + assert!(observed.members()[1].result().is_ok()); + let source = observed.members()[0].result().unwrap_err(); + let failure = observed.confirmed().err().unwrap(); + assert!(std::error::Error::source(&failure).is_some()); + assert!(Arc::ptr_eq( + &source, + &observed.members()[0].result().unwrap_err() + )); + for store in &stores { + assert_eq!(store.retained_bytes(), 8); + } + assert!(directory.log_epoch_referenced(leader, 4).await.unwrap()); + transport.contradict_first.store(true, Ordering::Release); + let contradicted = retire_recovered_members(transport.clone(), &sealed) + .await + .unwrap(); + assert!(contradicted.confirmed().is_err()); + assert!(contradicted.members()[1].result().is_ok()); + let replay = retire_recovered_members(transport, &sealed) + .await + .unwrap() + .confirmed() + .unwrap(); + assert_eq!(replay.members()[0].result().unwrap().durable_through, 2); + assert_eq!(replay.members()[1].result().unwrap().durable_through, 1); + assert!( + directory + .retired_recovered_log(&sealed, claimant, NOW_MS + 10_003) + .await + .unwrap() + .is_none() + ); + let retired = directory + .retire_recovered_log(&replay, claimant, NOW_MS + 10_004) + .await + .unwrap(); + assert_eq!(retired.log().phase(), NodeLogPhase::Retired); + assert_eq!( + retired.log().recovery_manifest(), + sealed.log().recovery_manifest() + ); + assert_eq!( + directory + .retire_recovered_log(&replay, claimant, NOW_MS + 10_005) + .await + .unwrap(), + retired + ); + assert!(!directory.log_epoch_referenced(leader, 4).await.unwrap()); + assert!( + directory + .takeover_proof(leader, claimant, NOW_MS + 10_005) + .await + .unwrap() + .is_some() + ); + let resumed = directory + .seal_recovery(&fenced, sealed.log().recovery_manifest(), NOW_MS + 10_005) + .await + .unwrap(); + assert!(NodeTakeoverProof::after_recovery(&fenced, &resumed).is_ok()); + for store in stores { + assert!(store.append(leader, 4, frames.clone(), 0).await.is_err()); + let candidates = store.retired_lanes(i64::MAX, 2).await.unwrap(); + assert_eq!(candidates.len(), 1); + // Exact native candidate and canonical closure are both present. This + // unit supplies the candidate's actual mtime as its chosen grace bound. + assert!( + store + .remove_retired(candidates[0], candidates[0].retired_at_ms()) + .await + .unwrap() + ); + assert_eq!(store.retained_bytes(), 0); + } + assert_eq!( + directory + .retired_recovered_log(&sealed, claimant, NOW_MS + 10_005) + .await + .unwrap(), + Some(retired) + ); +} + +#[tokio::test] +async fn recovered_inactive_enrollment_fences_empty_lanes_but_refuses_unexpected_records() { + let (directory, enrolled, claimant) = enrolled(false).await; + let leader = enrolled.advertisement().session(); + let fenced = directory + .claim_expired(leader, claimant, NOW_MS + 10_001) + .await + .unwrap(); + let sealed = directory + .seal_recovery(&fenced, None, NOW_MS + 10_002) + .await + .unwrap(); + assert!(!sealed.log().active()); + let root = tempfile::TempDir::new().unwrap(); + let store = FollowerStore::open( + root.path().to_owned(), + cellule_ltx::Limits::default(), + cellule_ltx::DiskBudget::new(1 << 30), + ) + .unwrap(); + let member = sealed.log().members()[0]; + let authorization = directory + .authorize_recovered_log_retire(claimant, member, leader, 4, None, NOW_MS + 10_003) + .await + .unwrap(); + assert_eq!( + store + .retire_recovered(member, authorization) + .await + .unwrap() + .durable_through, + 0 + ); + assert_eq!(store.retained_bytes(), 8); + assert!(store.append(leader, 4, frames(), 0).await.is_err()); + let other = tempfile::TempDir::new().unwrap(); + let unexpected = FollowerStore::open( + other.path().to_owned(), + cellule_ltx::Limits::default(), + cellule_ltx::DiskBudget::new(1 << 30), + ) + .unwrap(); + unexpected.append(leader, 4, frames(), 0).await.unwrap(); + let authorization = directory + .authorize_recovered_log_retire(claimant, member, leader, 4, None, NOW_MS + 10_003) + .await + .unwrap(); + assert!(matches!( + unexpected.retire_recovered(member, authorization).await, + Err(Error::Node("follower lane has uncovered records")) + )); + assert!(unexpected.retained_bytes() > 8); + // A native seal cannot turn unexpected records from an inactive canonical + // enrollment into a recoverable, acknowledged tail. + unexpected.seal(leader, 4).await.unwrap(); + let authorization = directory + .authorize_recovered_log_retire(claimant, member, leader, 4, None, NOW_MS + 10_003) + .await + .unwrap(); + assert!(matches!( + unexpected.retire_recovered(member, authorization).await, + Err(Error::Node("follower lane has uncovered records")) + )); + assert!(unexpected.retained_bytes() > 8); + assert!(directory.log_epoch_referenced(leader, 4).await.unwrap()); +} diff --git a/crates/cellule-runtime/src/node/tests/sessions.rs b/crates/cellule-runtime/src/node/tests/sessions.rs index 2f31f842..e64ee3a5 100644 --- a/crates/cellule-runtime/src/node/tests/sessions.rs +++ b/crates/cellule-runtime/src/node/tests/sessions.rs @@ -303,6 +303,8 @@ async fn expired_session_claim_is_atomic_idempotent_and_blocks_refresh() { .claim_expired(session, claimant, NOW_MS + 10_000) .await .unwrap(); + assert!(directory.is_retired(session).await.unwrap()); + assert!(!directory.is_withdrawn(session).await.unwrap()); assert_eq!(fenced.session(), session); assert_eq!( directory diff --git a/crates/cellule-runtime/src/peer/dispatch/convert.rs b/crates/cellule-runtime/src/peer/dispatch/convert.rs index c0246313..5ff8560b 100644 --- a/crates/cellule-runtime/src/peer/dispatch/convert.rs +++ b/crates/cellule-runtime/src/peer/dispatch/convert.rs @@ -203,7 +203,26 @@ pub(crate) fn error_reply(error: Error) -> wire::PeerReply { "Cell read replica is unavailable", 100, ), + Error::OwnerHistoryIncomplete { .. } => ( + wire::error::Code::Unavailable, + wire::error::Outcome::NotStarted, + "original Cell owner history is incomplete", + 100, + ), + Error::AcquisitionHistoryIncomplete { .. } => ( + wire::error::Code::Unavailable, + wire::error::Outcome::NotStarted, + "Cell acquisition history is incomplete", + 100, + ), + Error::RootLineageIncomplete { .. } | Error::RootPrefixUnproven { .. } => ( + wire::error::Code::Unavailable, + wire::error::Outcome::NotStarted, + "Cell root prefix is unproven", + 100, + ), Error::Fenced + | Error::CellReleaseRefused { .. } | Error::CellNotActive | Error::CellDraining | Error::RuntimeClosed @@ -247,6 +266,7 @@ pub(crate) fn error_reply(error: Error) -> wire::PeerReply { Error::Control(_) | Error::Catalog(_) | Error::Node(_) + | Error::FleetOperation(_) | Error::Release(_) | Error::Backup(_) | Error::Retention(_) diff --git a/crates/cellule-runtime/src/peer/tests.rs b/crates/cellule-runtime/src/peer/tests.rs index bcbb1504..e01207fa 100644 --- a/crates/cellule-runtime/src/peer/tests.rs +++ b/crates/cellule-runtime/src/peer/tests.rs @@ -534,3 +534,55 @@ fn replica_behind_error_preserves_both_positions_over_peer_wire() { } )); } + +#[test] +fn incomplete_owner_history_cannot_be_reported_as_success_or_empty_inventory() { + let reply = dispatch::error_reply(Error::OwnerHistoryIncomplete { + cell: crate::CellId::from_bytes([1; 32]), + incarnation: IncarnationId::from_bytes([2; 16]), + epoch: 3, + }); + let decoded = decode_peer_reply(&encode_peer_reply(&reply).unwrap()).unwrap(); + let Some(wire::peer_reply::Outcome::Error(error)) = decoded.outcome else { + panic!("expected history blocker"); + }; + assert_eq!(error.code, wire::error::Code::Unavailable as i32); + assert_eq!(error.outcome, wire::error::Outcome::NotStarted as i32); + assert_eq!(error.message, "original Cell owner history is incomplete"); + assert!(error.application_details.is_empty()); +} + +#[test] +fn missing_or_unproven_root_lineage_cannot_be_reported_as_success() { + let root = cellule_ltx::RootRef { + cell: [1; 32], + incarnation: [2; 16], + digest: [3; 32], + position: cellule_ltx::Position { + txid: 1, + checksum: cellule_ltx::types::CHECKSUM_FLAG, + }, + commit_sequence: 1, + }; + for source in [ + Error::AcquisitionHistoryIncomplete { + cell: crate::identity::CellId::from_bytes(root.cell), + incarnation: crate::identity::IncarnationId::from_bytes(root.incarnation), + epoch: 2, + }, + Error::RootLineageIncomplete { root }, + Error::RootPrefixUnproven { + prefix: Box::new(root), + root: Box::new(root), + }, + ] { + let reply = dispatch::error_reply(source); + let decoded = decode_peer_reply(&encode_peer_reply(&reply).unwrap()).unwrap(); + let Some(wire::peer_reply::Outcome::Error(error)) = decoded.outcome else { + panic!("expected prefix blocker"); + }; + assert_eq!(error.code, wire::error::Code::Unavailable as i32); + assert_eq!(error.outcome, wire::error::Outcome::NotStarted as i32); + assert!(error.application_details.is_empty()); + } +} diff --git a/crates/cellule-runtime/src/primitives/effects/api.rs b/crates/cellule-runtime/src/primitives/effects/api.rs index 15727b77..0a8d0371 100644 --- a/crates/cellule-runtime/src/primitives/effects/api.rs +++ b/crates/cellule-runtime/src/primitives/effects/api.rs @@ -42,8 +42,8 @@ pub fn register_effect_delivery( ) -> crate::Result<()> { registry.bind_effect_runner::()?; registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_query::>()?; + registry.bind_lease_command::>()?; + registry.bind_lease_query::>()?; registry.bind_query::>() } diff --git a/crates/cellule-runtime/src/primitives/maintenance.rs b/crates/cellule-runtime/src/primitives/maintenance.rs index 27304930..9df29c93 100644 --- a/crates/cellule-runtime/src/primitives/maintenance.rs +++ b/crates/cellule-runtime/src/primitives/maintenance.rs @@ -105,6 +105,19 @@ impl PersistedWorkInventory { } } +/// Durable work class preventing an ordinary idle owner transfer. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum TransferWorkClass { + /// A source Effect is due or already leased. + Effects, + /// A Queue message is ready or leased. + Queue, + /// A Workflow run, Activity, or timer remains executable. + Workflow, + /// A Cron schedule has a due tick. + Cron, +} + /// Transfer-only view of durable work that still needs the current owner. /// /// This deliberately has a separate contract from [`PersistedWorkInventory`]: @@ -120,6 +133,20 @@ impl TransferWorkInventory { self.bits == 0 } + pub(crate) const fn first_blocker(self) -> Option { + if self.bits & TRANSFER_EFFECTS != 0 { + Some(TransferWorkClass::Effects) + } else if self.bits & TRANSFER_QUEUE != 0 { + Some(TransferWorkClass::Queue) + } else if self.bits & TRANSFER_WORKFLOW != 0 { + Some(TransferWorkClass::Workflow) + } else if self.bits & TRANSFER_CRON != 0 { + Some(TransferWorkClass::Cron) + } else { + None + } + } + #[cfg(test)] const fn bits(self) -> u8 { self.bits @@ -557,6 +584,14 @@ mod tests { let inventory = inspect_transfer_work(&connection, CatalogRole::Queue, 10).unwrap(); assert_eq!(inventory.bits(), TRANSFER_EFFECTS | TRANSFER_QUEUE); assert!(!inventory.is_settled()); + assert_eq!(inventory.first_blocker(), Some(TransferWorkClass::Effects)); + connection.execute("DELETE FROM sys_effects", []).unwrap(); + assert_eq!( + inspect_transfer_work(&connection, CatalogRole::Queue, 10) + .unwrap() + .first_blocker(), + Some(TransferWorkClass::Queue) + ); } #[test] diff --git a/crates/cellule-runtime/src/primitives/maintenance_readiness/mod.rs b/crates/cellule-runtime/src/primitives/maintenance_readiness/mod.rs new file mode 100644 index 00000000..f29e2956 --- /dev/null +++ b/crates/cellule-runtime/src/primitives/maintenance_readiness/mod.rs @@ -0,0 +1,108 @@ +//! Primitive obligations at a foreground-closed maintenance barrier. +//! +//! Pending durable work can travel in an exact root after source claims close. +//! A live or malformed lease retains its current owner. This snapshot does not +//! prove actor quiescence, object coverage, or release, and never changes rows. + +use rusqlite::Connection; + +use crate::cell::catalog::CatalogRole; +use crate::{Error, Result}; + +/// An obligation that cannot yet move under planned maintenance. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum MaintenanceWorkBlocker { + /// An outstanding external Effect lease needs completion or normal expiry. + EffectLease, + /// A Queue delivery lease needs completion or normal expiry. + QueueLease, + /// An Activity lease needs completion or normal expiry. + ActivityLease, + /// Complete Blob stream/upload/pin coverage has not been established. + BlobInventory, +} + +impl MaintenanceWorkBlocker { + const fn bit(self) -> u8 { + match self { + Self::EffectLease => 1, + Self::QueueLease => 2, + Self::ActivityLease => 4, + Self::BlobInventory => 8, + } + } +} + +/// Bounded read-only inventory; it grants no execution or movement authority. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub struct MaintenanceWorkInventory { + bits: u8, +} + +impl MaintenanceWorkInventory { + /// Checks one obligation, without collapsing multiple simultaneous leases. + #[must_use] + pub const fn has_blocker(self, blocker: MaintenanceWorkBlocker) -> bool { + self.bits & blocker.bit() != 0 + } + + /// True only for this primitive snapshot. Source foreground closure and all + /// actor/publication/role barriers are additionally required for release. + #[must_use] + pub const fn is_transferable(self) -> bool { + self.bits == 0 + } +} + +pub(crate) fn inspect( + connection: &Connection, + role: CatalogRole, + now_ms: i64, +) -> Result { + if now_ms < 0 { + return Err(Error::Command("invalid maintenance inspection time")); + } + let mut inventory = MaintenanceWorkInventory::default(); + if exists( + connection, + "SELECT EXISTS(SELECT 1 FROM sys_effects INDEXED BY sys_effects_leases WHERE state = 1 AND (lease_until_ms IS NULL OR lease_until_ms < 0 OR token IS NULL OR length(token) != 16 OR lease_until_ms > ?1) LIMIT 1)", + now_ms, + )? { + inventory.bits |= MaintenanceWorkBlocker::EffectLease.bit(); + } + if role == CatalogRole::Queue + && exists( + connection, + "SELECT EXISTS(SELECT 1 FROM queue_messages INDEXED BY queue_leases WHERE state = 1 AND (lease_until_ms IS NULL OR lease_until_ms < 0 OR token IS NULL OR length(token) != 16 OR lease_until_ms > ?1) LIMIT 1)", + now_ms, + )? + { + inventory.bits |= MaintenanceWorkBlocker::QueueLease.bit(); + } + if role == CatalogRole::Workflow + && exists( + connection, + "SELECT EXISTS(SELECT 1 FROM workflow_activities INDEXED BY activities_leases WHERE state = 1 AND (lease_until_ms IS NULL OR lease_until_ms < 0 OR token IS NULL OR length(token) != 16 OR lease_until_ms > ?1) LIMIT 1)", + now_ms, + )? + { + inventory.bits |= MaintenanceWorkBlocker::ActivityLease.bit(); + } + if role == CatalogRole::Blob { + // Actor queue closure alone cannot inventory external upload/stream and + // pin owners. Keep this class explicit until those owners join the scan. + inventory.bits |= MaintenanceWorkBlocker::BlobInventory.bit(); + } + Ok(inventory) +} + +fn exists(connection: &Connection, sql: &str, now_ms: i64) -> Result { + match connection.query_row(sql, [now_ms], |row| row.get::<_, i64>(0))? { + 0 => Ok(false), + 1 => Ok(true), + _ => Err(Error::Command("invalid maintenance existence result")), + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/cellule-runtime/src/primitives/maintenance_readiness/tests.rs b/crates/cellule-runtime/src/primitives/maintenance_readiness/tests.rs new file mode 100644 index 00000000..d328d3e2 --- /dev/null +++ b/crates/cellule-runtime/src/primitives/maintenance_readiness/tests.rs @@ -0,0 +1,143 @@ +use super::*; +use crate::cell::schema::install_runtime_schema; +use crate::identity::{CellId, IncarnationId}; +use crate::primitives::maintenance::inspect_transfer_work; +use crate::primitives::{ + cron::install_cron_schema, queue::install_queue_schema, workflow::install_workflow_schema, +}; + +fn database() -> Connection { + let mut connection = Connection::open_in_memory().unwrap(); + install_runtime_schema( + &mut connection, + CellId::from_bytes([1; 32]), + IncarnationId::from_bytes([2; 16]), + 1, + ) + .unwrap(); + let transaction = connection.transaction().unwrap(); + install_queue_schema(&transaction).unwrap(); + install_workflow_schema(&transaction).unwrap(); + install_cron_schema(&transaction).unwrap(); + transaction.commit().unwrap(); + connection +} + +fn pending_effect(connection: &Connection) { + connection.execute("INSERT INTO sys_effects(effect_id, destination, operation, state, attempt, due_at_ms, expires_at_ms, token, lease_until_ms, created_sequence, result) VALUES (?1, ?2, X'01', 0, 0, 0, 1000, NULL, NULL, 1, NULL)", ([3_u8; 32].as_slice(), [4_u8; 32].as_slice())).unwrap(); +} + +fn pending_queue(connection: &Connection) { + connection.execute("INSERT INTO queue_messages(message_id, payload, state, attempt, due_at_ms, expires_at_ms, token, lease_until_ms, result_code) VALUES (?1, X'01', 0, 0, 0, 1000, NULL, NULL, NULL)", [[5_u8; 16].as_slice()]).unwrap(); +} + +fn pending_workflow(connection: &Connection) { + connection + .execute( + "INSERT INTO workflow_runs VALUES (X'01', ?1, ?2, 0, X'', 0, NULL, NULL)", + ([6_u8; 16].as_slice(), [7_u8; 32].as_slice()), + ) + .unwrap(); + connection.execute("INSERT INTO workflow_activities(run_id, activity_id, activity_type, input, state, attempt, due_at_ms, expires_at_ms, token, lease_until_ms, completion_token, completion_digest, result) VALUES (?1, ?2, 'test', X'', 0, 0, 0, 1000, NULL, NULL, NULL, NULL, NULL)", ([6_u8;16].as_slice(),[8_u8;16].as_slice())).unwrap(); + connection + .execute( + "INSERT INTO workflow_timers(run_id, timer_id, due_at_ms, state) VALUES (?1, ?2, 0, 0)", + ([6_u8; 16].as_slice(), [9_u8; 16].as_slice()), + ) + .unwrap(); +} + +#[test] +fn pending_durable_work_can_move_at_a_maintenance_barrier_but_idle_transfer_stays_closed() { + let connection = database(); + pending_effect(&connection); + pending_queue(&connection); + pending_workflow(&connection); + connection + .execute( + "INSERT INTO cron_schedules VALUES (?1, 0, X'', X'', 1000, 0, 0, 1, 1, 0)", + [[10_u8; 16].as_slice()], + ) + .unwrap(); + for role in [ + CatalogRole::Sql, + CatalogRole::Queue, + CatalogRole::Workflow, + CatalogRole::Cron, + ] { + assert!(inspect(&connection, role, 10).unwrap().is_transferable()); + assert!( + !inspect_transfer_work(&connection, role, 10) + .unwrap() + .is_settled() + ); + } +} + +#[test] +fn every_live_lease_remains_visible_and_expiry_inspection_does_not_rewrite_it() { + let connection = database(); + pending_effect(&connection); + pending_queue(&connection); + pending_workflow(&connection); + let token = [11_u8; 16]; + for table in ["sys_effects", "queue_messages", "workflow_activities"] { + connection + .execute( + &format!("UPDATE {table} SET state = 1, token = ?1, lease_until_ms = 100"), + [token.as_slice()], + ) + .unwrap(); + } + assert!( + inspect(&connection, CatalogRole::Sql, 99) + .unwrap() + .has_blocker(MaintenanceWorkBlocker::EffectLease) + ); + let queue = inspect(&connection, CatalogRole::Queue, 99).unwrap(); + assert!(queue.has_blocker(MaintenanceWorkBlocker::EffectLease)); + assert!(queue.has_blocker(MaintenanceWorkBlocker::QueueLease)); + let workflow = inspect(&connection, CatalogRole::Workflow, 99).unwrap(); + assert!(workflow.has_blocker(MaintenanceWorkBlocker::EffectLease)); + assert!(workflow.has_blocker(MaintenanceWorkBlocker::ActivityLease)); + for role in [CatalogRole::Sql, CatalogRole::Queue, CatalogRole::Workflow] { + assert!(inspect(&connection, role, 100).unwrap().is_transferable()); + } + for table in ["sys_effects", "queue_messages", "workflow_activities"] { + let retained: (i64, Vec, i64) = connection + .query_row( + &format!("SELECT state, token, lease_until_ms FROM {table}"), + [], + |row| Ok((row.get(0)?, row.get(1)?, row.get(2)?)), + ) + .unwrap(); + assert_eq!(retained, (1, token.to_vec(), 100)); + } +} + +#[test] +fn malformed_time_schema_and_unproven_blob_inventory_never_report_ready() { + let connection = database(); + assert!(inspect(&connection, CatalogRole::Sql, -1).is_err()); + pending_queue(&connection); + connection + .execute( + "UPDATE queue_messages SET state = 1, token = ?1, lease_until_ms = -1", + [[12_u8; 16].as_slice()], + ) + .unwrap(); + assert!( + inspect(&connection, CatalogRole::Queue, 100) + .unwrap() + .has_blocker(MaintenanceWorkBlocker::QueueLease) + ); + assert!( + inspect(&connection, CatalogRole::Blob, 100) + .unwrap() + .has_blocker(MaintenanceWorkBlocker::BlobInventory) + ); + connection + .execute_batch("DROP TABLE workflow_activities") + .unwrap(); + assert!(inspect(&connection, CatalogRole::Workflow, 100).is_err()); +} diff --git a/crates/cellule-runtime/src/primitives/mod.rs b/crates/cellule-runtime/src/primitives/mod.rs index 1af1faa4..38569d36 100644 --- a/crates/cellule-runtime/src/primitives/mod.rs +++ b/crates/cellule-runtime/src/primitives/mod.rs @@ -7,6 +7,7 @@ pub mod cron; pub mod effects; pub mod kv; pub mod maintenance; +pub mod maintenance_readiness; pub mod queue; pub mod sql; pub mod workflow; diff --git a/crates/cellule-runtime/src/primitives/queue/api/mod.rs b/crates/cellule-runtime/src/primitives/queue/api/mod.rs index a5c239fe..054fa3c3 100644 --- a/crates/cellule-runtime/src/primitives/queue/api/mod.rs +++ b/crates/cellule-runtime/src/primitives/queue/api/mod.rs @@ -100,9 +100,9 @@ pub fn register_queue(registry: &mut RegistryBuilder) -> crate:: )?; registry.bind_command::>()?; registry.bind_command::>()?; - registry.bind_command::>()?; + registry.bind_lease_command::>()?; registry.bind_command::>()?; - registry.bind_query::>()?; + registry.bind_lease_query::>()?; registry.bind_query::>()?; register_maintenance::(registry) } diff --git a/crates/cellule-runtime/src/primitives/workflow/activity_api/mod.rs b/crates/cellule-runtime/src/primitives/workflow/activity_api/mod.rs index 6e3e70fc..a315135d 100644 --- a/crates/cellule-runtime/src/primitives/workflow/activity_api/mod.rs +++ b/crates/cellule-runtime/src/primitives/workflow/activity_api/mod.rs @@ -56,9 +56,9 @@ pub fn register_workflow_activities( } registry.bind_activity_runner::()?; registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_query::>() + registry.bind_lease_command::>()?; + registry.bind_lease_command::>()?; + registry.bind_lease_query::>() } /// Registers one native handler for a module definition and activity type. diff --git a/crates/cellule-runtime/src/publication/lineage.rs b/crates/cellule-runtime/src/publication/lineage.rs new file mode 100644 index 00000000..0c045f5c --- /dev/null +++ b/crates/cellule-runtime/src/publication/lineage.rs @@ -0,0 +1,76 @@ +//! Native lineage retention joined with the existing immutable preparation owner. +use super::*; +use cellule_ltx::{RootPreparation, RootPreparationFuture, RootPreparationMetadata}; +use std::sync::{Arc, Mutex}; + +pub(super) type LineageConfirmation = Arc>>; + +struct RootLineageRecorder { + authority: CellAuthority, + confirmed: LineageConfirmation, +} +impl RootPreparationMetadata for RootLineageRecorder { + fn retain(&self, preparation: RootPreparation) -> RootPreparationFuture<'_> { + Box::pin(async move { + self.authority + .retain_root_preparation(preparation) + .await + .map_err(|source| Box::new(source) as Box)?; + *self.confirmed.lock().map_err(|_| { + Box::new(Error::Peer("root lineage confirmation lock poisoned")) + as Box + })? = Some(preparation); + Ok(()) + }) + } +} + +pub(super) fn replica( + replica: cellule_ltx::CellReplica, + authority: &CellAuthority, +) -> (cellule_ltx::CellReplica, LineageConfirmation) { + let confirmed = Arc::new(Mutex::new(None)); + let replica = replica.with_root_metadata(Arc::new(RootLineageRecorder { + authority: authority.clone(), + confirmed: confirmed.clone(), + })); + (replica, confirmed) +} + +pub(super) fn error(source: cellule_ltx::LtxError) -> Error { + match source { + cellule_ltx::LtxError::RootPreparation { source } => match source.downcast::() { + Ok(source) => *source, + Err(source) => cellule_ltx::LtxError::RootPreparation { source }.into(), + }, + source => source.into(), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn preparation_preserves_native_retry_policy_and_original_typed_source() { + let error = super::error(cellule_ltx::LtxError::Deadline); + assert!(retryable_publication_error(&error)); + assert_eq!(runtime_retry_hint(&error), None); + let error = super::error(cellule_ltx::LtxError::RootPreparation { + source: Box::new(Error::Fenced), + }); + assert!(matches!(error, Error::Fenced)); + assert!(!retryable_publication_error(&error)); + let error = super::error(cellule_ltx::LtxError::RootPreparation { + source: Box::new(std::io::Error::new( + std::io::ErrorKind::PermissionDenied, + "foreign source", + )), + }); + assert!(matches!( + error, + Error::Ltx(cellule_ltx::LtxError::RootPreparation { .. }) + )); + assert!(!retryable_publication_error(&error)); + } +} diff --git a/crates/cellule-runtime/src/publication/mod.rs b/crates/cellule-runtime/src/publication/mod.rs index 5cfb331b..c05bf6a0 100644 --- a/crates/cellule-runtime/src/publication/mod.rs +++ b/crates/cellule-runtime/src/publication/mod.rs @@ -10,6 +10,8 @@ use crate::node::log_shipper::NodeLogSubmission; use crate::retry::{Backoff, retry_hint, retryable_storage_error}; use crate::{Error, Result}; +mod lineage; + const COMPACTION_CHECK_INTERVAL: u8 = 8; const COMPACTION_DEBT_SEGMENTS: usize = 32; const MAX_COMPACTION_CASCADE: usize = 9; @@ -46,6 +48,7 @@ pub struct CellPublisher { node_lease: Option, node_durability: Option, telemetry: crate::fleet::telemetry::CellTelemetryHandle, + lineage_confirmed: Option, } enum AppendBase { @@ -84,6 +87,7 @@ impl CellPublisher { node_lease: None, node_durability: None, telemetry: crate::fleet::telemetry::CellTelemetryHandle::default(), + lineage_confirmed: None, } } @@ -142,6 +146,10 @@ impl CellPublisher { &self.observed } + pub(crate) fn resource_limits(&self) -> cellule_ltx::Limits { + self.replica.limits() + } + /// Returns the Cell storage layout this publisher writes through. #[must_use] pub(crate) fn layout(&self) -> &cellule_ltx::CellStorageLayout { @@ -559,7 +567,7 @@ impl CellPublisher { ) -> Result { let mut backoff = Backoff::default(); loop { - let replica = self.replica.clone(); + let (replica, confirmation) = lineage::replica(self.replica.clone(), &self.authority); let attempt = async { match base { AppendBase::Published(root) => { @@ -584,11 +592,20 @@ impl CellPublisher { } }; match result { - Ok(prepared) => return Ok(prepared), - Err(error) if retryable_ltx_error(&error) => { - backoff.wait(ltx_retry_hint(&error)).await; + Ok(prepared) => { + self.lineage_confirmed = *confirmation + .lock() + .map_err(|_| Error::Peer("root lineage confirmation lock poisoned"))?; + return Ok(prepared); + } + Err(source) => { + let error = lineage::error(source); + if retryable_publication_error(&error) { + backoff.wait(runtime_retry_hint(&error)).await; + } else { + return Err(error); + } } - Err(error) => return Err(error.into()), } } } @@ -637,11 +654,22 @@ impl CellPublisher { Transition::Publish, ), }; - match self - .authority - .transition(&self.observed, successor.clone(), transition) - .await - { + // Retain verified preparation inputs before the authority can + // select this root. A failed CAS leaves only a private proposal; + // the same retry/adoption owner handles metadata ambiguity. + let publication = async { + if self.lineage_confirmed != Some(prepared.preparation()) { + // External preparations and rebased compaction inputs use + // the same canonical metadata path before authority CAS. + self.authority.retain_root_lineage(prepared).await?; + self.lineage_confirmed = Some(prepared.preparation()); + } + self.authority + .transition(&self.observed, successor.clone(), transition) + .await + } + .await; + match publication { Ok(published) => { self.check_node_lease()?; self.observed = published; @@ -904,10 +932,11 @@ impl CellDurabilitySubmitter { } fn retryable_publication_error(error: &Error) -> bool { - let Error::Storage(error) = error else { - return false; - }; - retryable_storage_error(error) + match error { + Error::Storage(error) => retryable_storage_error(error), + Error::Ltx(error) => retryable_ltx_error(error), + _ => false, + } } fn retryable_ltx_error(error: &cellule_ltx::LtxError) -> bool { @@ -917,10 +946,11 @@ fn retryable_ltx_error(error: &cellule_ltx::LtxError) -> bool { } fn runtime_retry_hint(error: &Error) -> Option { - let Error::Storage(error) = error else { - return None; - }; - retry_hint(error) + match error { + Error::Storage(error) => retry_hint(error), + Error::Ltx(error) => ltx_retry_hint(error), + _ => None, + } } fn ltx_retry_hint(error: &cellule_ltx::LtxError) -> Option { diff --git a/crates/cellule-runtime/src/publication/tests.rs b/crates/cellule-runtime/src/publication/tests.rs index 8855b1f3..7053f34e 100644 --- a/crates/cellule-runtime/src/publication/tests.rs +++ b/crates/cellule-runtime/src/publication/tests.rs @@ -68,6 +68,7 @@ async fn verify_compaction_append(schema: u32) { observed, directory.path().to_owned(), ); + let mut prefix = None; for sequence in 1..=8_u64 { database .transaction(|transaction| { @@ -82,6 +83,9 @@ async fn verify_compaction_append(schema: u32) { let cuts = database.capture_deferred().unwrap(); let prepared = publisher.prepare_append(&cuts, sequence, 1).await.unwrap(); publisher.publish_prepared(&prepared, None).await.unwrap(); + if sequence == 1 { + prefix = Some(prepared.root()); + } } assert!(publisher.compaction_due()); let before = publisher.control().value().ltx_root().unwrap(); @@ -91,6 +95,15 @@ async fn verify_compaction_append(schema: u32) { assert_eq!(after.position, before.position); assert_eq!(after.commit_sequence, before.commit_sequence); assert_eq!(replica.open_root(&after).await.unwrap().segment_count(), 1); + let prefix = prefix.unwrap(); + let proof = publisher + .authority + .verify_root_prefix(prefix, after, &replica, 64) + .await + .unwrap(); + assert_eq!(proof.prefix(), prefix); + assert_eq!(proof.root(), after); + assert!(proof.inspected_roots() >= 8 && proof.dependency_count() > 0); assert_eq!(publisher.compact_one_quiet().await.unwrap(), Some(false)); assert!(!publisher.compaction_due()); @@ -192,6 +205,20 @@ async fn verify_compaction_append(schema: u32) { assert_eq!(publisher.control().value().revision, before_revision + 1); let forced = publisher.control().value().ltx_root().unwrap(); assert!(replica.open_root(&forced).await.unwrap().segment_count() < 32); + let proof = publisher + .authority + .verify_root_prefix(prefix, forced, &replica, 64) + .await + .unwrap(); + assert_eq!(proof.root(), forced); + assert!(proof.inspected_roots() > 30); + assert!(matches!( + publisher + .authority + .verify_root_prefix(prefix, forced, &replica, 2) + .await, + Err(Error::Capacity(_)) + )); database.close().unwrap(); } @@ -302,7 +329,7 @@ impl crate::node::durability::NodeLogAuthority for RefusingAuthority { fn close<'a>( &'a self, - _barrier: &'a crate::node::log::NodeLogRotationBarrier, + _retirement: &'a crate::node::log::NodeLogRetirementObservation, ) -> futures_util::future::BoxFuture<'a, crate::Result<()>> { Box::pin(async { Err(Error::Node("test authority refuses closing")) }) } diff --git a/crates/cellule-runtime/src/recovery/manifest/inventory.rs b/crates/cellule-runtime/src/recovery/manifest/inventory.rs new file mode 100644 index 00000000..109fd80e --- /dev/null +++ b/crates/cellule-runtime/src/recovery/manifest/inventory.rs @@ -0,0 +1,101 @@ +use super::{MAX_MANIFEST_BYTES, PinnedRecoveryCell, RecoveryManifest, RecoveryManifestStore}; +use crate::identity::{Digest, SessionId}; +use crate::{Error, Result}; + +/// Complete metadata from one immutable, digest-verified recovery manifest. +/// +/// This inventories recovered suffixes across every application in the manifest. +/// It does not include object-covered Cells without a recovered suffix, verify +/// bundle availability, or establish current ownership, serving, or node closure. +/// The caller must obtain the manifest identity from canonical sealed-log authority. +pub struct RecoveryManifestInventory { + leader_session: SessionId, + log_epoch: u64, + manifest_digest: Digest, + cells: Vec, +} + +impl RecoveryManifestInventory { + /// Returns the original failed leader session. + #[must_use] + pub const fn leader_session(&self) -> SessionId { + self.leader_session + } + + /// Returns the original node-log epoch. + #[must_use] + pub const fn log_epoch(&self) -> u64 { + self.log_epoch + } + + /// Returns the digest of the complete canonical manifest bytes. + #[must_use] + pub const fn manifest_digest(&self) -> Digest { + self.manifest_digest + } + + /// Returns every original pinned scope, including other applications. + #[must_use] + pub fn cells(&self) -> &[PinnedRecoveryCell] { + &self.cells + } +} + +impl RecoveryManifestStore { + /// Reads the complete original recovered-suffix set without starting recovery. + /// + /// Uses the same bounded, canonical, digest- and scope-checked read as + /// [`Self::load_overlay`]. The store's application does not filter this set. + /// Each row can subsequently be verified through the matching application's + /// `load_overlay`; metadata alone does not verify the referenced bundle. + pub async fn load_manifest( + &self, + leader_session: SessionId, + log_epoch: u64, + manifest_digest: Digest, + ) -> Result { + let manifest = self + .load_manifest_body(leader_session, log_epoch, manifest_digest) + .await?; + let cells = manifest + .cells + .iter() + .map(|cell| cell.pinned(leader_session, log_epoch, manifest_digest)) + .collect(); + Ok(RecoveryManifestInventory { + leader_session, + log_epoch, + manifest_digest, + cells, + }) + } + + pub(super) async fn load_manifest_body( + &self, + leader_session: SessionId, + log_epoch: u64, + manifest_digest: Digest, + ) -> Result { + if leader_session.as_bytes().iter().all(|byte| *byte == 0) || log_epoch == 0 { + return Err(Error::Node("invalid recovery manifest scope")); + } + let path = self.layout.node_log_recovery_path( + leader_session.as_bytes(), + log_epoch, + manifest_digest.as_bytes(), + ); + let (body, _) = self + .layout + .store() + .get_with_etag_bounded(&path, MAX_MANIFEST_BYTES) + .await?; + if *blake3::hash(&body).as_bytes() != *manifest_digest.as_bytes() { + return Err(Error::Node("recovery manifest digest differs")); + } + let manifest = RecoveryManifest::decode(&body)?; + if manifest.leader_session != leader_session || manifest.log_epoch != log_epoch { + return Err(Error::Node("recovery manifest path scope differs")); + } + Ok(manifest) + } +} diff --git a/crates/cellule-runtime/src/recovery/manifest/mod.rs b/crates/cellule-runtime/src/recovery/manifest/mod.rs index c7ad79cd..9cced140 100644 --- a/crates/cellule-runtime/src/recovery/manifest/mod.rs +++ b/crates/cellule-runtime/src/recovery/manifest/mod.rs @@ -15,6 +15,9 @@ use crate::{Error, Result}; const MAX_MANIFEST_BYTES: u64 = 2 << 20; const MULTIPART_BYTES: usize = 8 << 20; +mod inventory; +pub use inventory::RecoveryManifestInventory; + /// One control-ready pointer returned after bundle and manifest publication. pub struct PinnedRecoveryCell { /// Application the recovered Cell belongs to. @@ -365,23 +368,13 @@ impl RecoveryManifestStore { Ok(PinnedRecoveryCells { cells: manifest .cells - .into_iter() - .map(|cell| PinnedRecoveryCell { - application: ApplicationId::from_bytes(cell.application), - cell: CellId::from_bytes(cell.cell), - incarnation: IncarnationId::from_bytes(cell.incarnation), - cell_epoch: cell.cell_epoch, - recovery: RecoveryOverlayRef { + .iter() + .map(|cell| { + cell.pinned( leader_session, log_epoch, - manifest_digest: Digest::from_bytes(manifest_digest), - first_node_sequence: cell.first_node_sequence, - last_node_sequence: cell.last_node_sequence, - predecessor: runtime_root(cell.predecessor), - final_txid: cell.final_position.txid, - final_checksum: cell.final_position.checksum, - final_commit_sequence: cell.final_commit_sequence, - }, + Digest::from_bytes(manifest_digest), + ) }) .collect(), summary, @@ -395,25 +388,13 @@ impl RecoveryManifestStore { incarnation: IncarnationId, recovery: &RecoveryOverlayRef, ) -> Result { - let path = self.layout.node_log_recovery_path( - recovery.leader_session.as_bytes(), - recovery.log_epoch, - recovery.manifest_digest.as_bytes(), - ); - let (body, _) = self - .layout - .store() - .get_with_etag_bounded(&path, MAX_MANIFEST_BYTES) + let manifest = self + .load_manifest_body( + recovery.leader_session, + recovery.log_epoch, + recovery.manifest_digest, + ) .await?; - if *blake3::hash(&body).as_bytes() != *recovery.manifest_digest.as_bytes() { - return Err(Error::Node("recovery manifest digest differs")); - } - let manifest = RecoveryManifest::decode(&body)?; - if manifest.leader_session != recovery.leader_session - || manifest.log_epoch != recovery.log_epoch - { - return Err(Error::Node("recovery manifest path scope differs")); - } let row = manifest .cells .into_iter() @@ -423,17 +404,13 @@ impl RecoveryManifestStore { && row.incarnation == *incarnation.as_bytes() }) .ok_or(Error::Node("recovery manifest does not contain Cell"))?; - let expected = RecoveryOverlayRef { - leader_session: recovery.leader_session, - log_epoch: recovery.log_epoch, - manifest_digest: recovery.manifest_digest, - first_node_sequence: row.first_node_sequence, - last_node_sequence: row.last_node_sequence, - predecessor: runtime_root(row.predecessor), - final_txid: row.final_position.txid, - final_checksum: row.final_position.checksum, - final_commit_sequence: row.final_commit_sequence, - }; + let expected = row + .pinned( + recovery.leader_session, + recovery.log_epoch, + recovery.manifest_digest, + ) + .recovery; if &expected != recovery { return Err(Error::Node( "recovery control pointer differs from manifest", @@ -537,6 +514,42 @@ struct ManifestCell { bundle_digest: [u8; 32], } +impl ManifestCell { + fn scope(&self) -> ([u8; 16], [u8; 32], [u8; 16], u64) { + ( + self.application, + self.cell, + self.incarnation, + self.cell_epoch, + ) + } + + fn pinned( + &self, + leader_session: SessionId, + log_epoch: u64, + manifest_digest: Digest, + ) -> PinnedRecoveryCell { + PinnedRecoveryCell { + application: ApplicationId::from_bytes(self.application), + cell: CellId::from_bytes(self.cell), + incarnation: IncarnationId::from_bytes(self.incarnation), + cell_epoch: self.cell_epoch, + recovery: RecoveryOverlayRef { + leader_session, + log_epoch, + manifest_digest, + first_node_sequence: self.first_node_sequence, + last_node_sequence: self.last_node_sequence, + predecessor: runtime_root(self.predecessor), + final_txid: self.final_position.txid, + final_checksum: self.final_position.checksum, + final_commit_sequence: self.final_commit_sequence, + }, + } + } +} + impl RecoveryManifest { fn encode(&self) -> Result> { let body = serde_json::to_vec(&RawManifest::from(self))?; @@ -662,6 +675,17 @@ impl TryFrom for RecoveryManifest { log_epoch, cells, }; + // Publication sorts the complete scope and rejects duplicates. Preserve + // that invariant on reads so no controller sees an ambiguous subset. + if manifest + .cells + .windows(2) + .any(|pair| pair[0].scope() >= pair[1].scope()) + { + return Err(Error::Node( + "recovery manifest Cell scopes are not strictly ordered", + )); + } if leader_session.as_bytes().iter().all(|byte| *byte == 0) || log_epoch == 0 || manifest.cells.iter().any(|cell| { diff --git a/crates/cellule-runtime/src/recovery/manifest/tests.rs b/crates/cellule-runtime/src/recovery/manifest/tests.rs index b09ea330..3a8cbe15 100644 --- a/crates/cellule-runtime/src/recovery/manifest/tests.rs +++ b/crates/cellule-runtime/src/recovery/manifest/tests.rs @@ -71,6 +71,13 @@ impl RecoveryArtifactStore for MemoryArtifactStore { async fn recovery_fixture_with_store( artifacts: Option>, +) -> RecoveryFixture { + recovery_fixture_with_scopes(artifacts, false).await +} + +async fn recovery_fixture_with_scopes( + artifacts: Option>, + multiple: bool, ) -> RecoveryFixture { let limits = cellule_ltx::Limits::default(); let directory = tempfile::TempDir::new().unwrap(); @@ -112,7 +119,7 @@ async fn recovery_fixture_with_store( limits, ) .unwrap(); - let recovered = build_recovery_overlays( + let mut recovered = build_recovery_overlays( vec![frame], &[RecoveryBase { application, @@ -122,6 +129,45 @@ async fn recovery_fixture_with_store( limits, ) .unwrap(); + if multiple { + for (application, cell, incarnation, sequence) in [ + ([3; 16], [7; 32], [8; 16], 2), + ([9; 16], [10; 32], [11; 16], 3), + ] { + let layout = + CellStorageLayout::new(layout.store().clone(), Path::from("root"), application); + let replica = cellule_ltx::CellReplica::new(layout, cell, incarnation, limits).unwrap(); + let root = replica.prepare(None, &first, 1, 1).await.unwrap().root(); + let frame = cellule_ltx::encode_node_frame( + cellule_ltx::NodeFrameScope { + leader_session: [1; 16], + log_epoch: 2, + node_sequence: sequence, + application, + cell, + incarnation, + cell_epoch: 6, + commit_sequence: 2, + }, + segment.info().clone(), + Bytes::from(std::fs::read(segment.path()).unwrap()), + limits, + ) + .unwrap(); + recovered.extend( + build_recovery_overlays( + vec![frame], + &[RecoveryBase { + application, + cell_epoch: 6, + root, + }], + limits, + ) + .unwrap(), + ); + } + } let manifests = RecoveryManifestStore::new(layout.clone(), limits); let manifests = artifacts.map_or(manifests.clone(), |store| { manifests.with_recovery_artifacts(store) @@ -138,7 +184,7 @@ async fn recovery_fixture_with_store( layout, replica, manifests, - pinned: pinned.pop().unwrap(), + pinned: pinned.remove(0), publication, base, final_position: tail.position, @@ -405,3 +451,269 @@ async fn load_overlay_rejects_corrupt_bundle_bytes() { Error::Node("recovery bundle digest differs") )); } + +async fn raw_manifest(fixture: &RecoveryFixture) -> RawManifest { + let recovery = &fixture.pinned.recovery; + let path = fixture.layout.node_log_recovery_path( + recovery.leader_session.as_bytes(), + recovery.log_epoch, + recovery.manifest_digest.as_bytes(), + ); + let (body, _) = fixture + .layout + .store() + .get_with_etag_bounded(&path, MAX_MANIFEST_BYTES) + .await + .unwrap(); + serde_json::from_slice(&body).unwrap() +} + +async fn install_manifest_bytes(fixture: &RecoveryFixture, body: Vec) -> Digest { + let digest = Digest::from_bytes(*blake3::hash(&body).as_bytes()); + let recovery = &fixture.pinned.recovery; + let path = fixture.layout.node_log_recovery_path( + recovery.leader_session.as_bytes(), + recovery.log_epoch, + digest.as_bytes(), + ); + fixture + .layout + .store() + .put(&path, Bytes::from(body)) + .await + .unwrap(); + digest +} + +#[tokio::test] +async fn manifest_inventory_reconstructs_every_scope_across_applications() { + let fixture = recovery_fixture_with_scopes(None, true).await; + // A newly constructed adapter has no local pinning result or artifact cache. + let manifests = + RecoveryManifestStore::new(fixture.layout.clone(), cellule_ltx::Limits::default()); + let original = &fixture.pinned.recovery; + let inventory = manifests + .load_manifest( + original.leader_session, + original.log_epoch, + original.manifest_digest, + ) + .await + .unwrap(); + assert_eq!(inventory.leader_session(), original.leader_session); + assert_eq!(inventory.log_epoch(), original.log_epoch); + assert_eq!(inventory.manifest_digest(), original.manifest_digest); + assert_eq!(inventory.cells().len(), 3); + assert_eq!(inventory.cells()[0].cell, fixture.pinned.cell); + assert_eq!(inventory.cells()[0].recovery, fixture.pinned.recovery); + assert_eq!(inventory.cells()[1].cell, CellId::from_bytes([7; 32])); + assert_eq!( + inventory.cells()[2].application, + ApplicationId::from_bytes([9; 16]) + ); + for row in inventory.cells() { + let layout = CellStorageLayout::new( + fixture.layout.store().clone(), + Path::from("root"), + *row.application.as_bytes(), + ); + let store = RecoveryManifestStore::new(layout, cellule_ltx::Limits::default()); + let overlay = store + .load_overlay(row.cell, row.incarnation, &row.recovery) + .await + .unwrap(); + assert_eq!(overlay.final_position(), fixture.final_position); + assert_eq!(row.cell_epoch, 6); + assert_eq!(row.recovery.manifest_digest, inventory.manifest_digest()); + } +} + +#[tokio::test] +async fn manifest_inventory_is_metadata_and_does_not_certify_bundle_availability() { + let fixture = recovery_fixture().await; + let recovery = &fixture.pinned.recovery; + let raw = raw_manifest(&fixture).await; + let bundle_digest: [u8; 32] = unhex(&raw.cells[0].bundle_digest).unwrap(); + let path = fixture.layout.node_log_bundle_path( + recovery.leader_session.as_bytes(), + recovery.log_epoch, + &bundle_digest, + ); + fixture.inner.delete(&path).await.unwrap(); + let inventory = fixture + .manifests + .load_manifest( + recovery.leader_session, + recovery.log_epoch, + recovery.manifest_digest, + ) + .await + .unwrap(); + assert_eq!(inventory.cells().len(), 1); + assert!(matches!( + load_error(&fixture, recovery).await, + Error::Storage(StorageError::NotFound { .. }) + )); +} + +#[tokio::test] +async fn manifest_inventory_rejects_invalid_scope_and_preserves_missing_object_error() { + let fixture = recovery_fixture().await; + let recovery = &fixture.pinned.recovery; + for (leader, epoch) in [ + (SessionId::from_bytes([0; 16]), 2), + (recovery.leader_session, 0), + ] { + assert!(matches!( + fixture + .manifests + .load_manifest(leader, epoch, recovery.manifest_digest) + .await, + Err(Error::Node("invalid recovery manifest scope")) + )); + } + assert!(matches!( + fixture + .manifests + .load_manifest( + recovery.leader_session, + recovery.log_epoch, + Digest::from_bytes([99; 32]) + ) + .await, + Err(Error::Storage(StorageError::NotFound { .. })) + )); +} + +#[tokio::test] +async fn manifest_inventory_rejects_corrupt_digest_and_wrong_path_scope() { + let fixture = recovery_fixture().await; + let recovery = &fixture.pinned.recovery; + let path = fixture.layout.node_log_recovery_path( + recovery.leader_session.as_bytes(), + recovery.log_epoch, + recovery.manifest_digest.as_bytes(), + ); + let raw = raw_manifest(&fixture).await; + fixture + .inner + .put(&path, Bytes::from_static(b"corrupt").into()) + .await + .unwrap(); + assert!(matches!( + fixture + .manifests + .load_manifest( + recovery.leader_session, + recovery.log_epoch, + recovery.manifest_digest + ) + .await, + Err(Error::Node("recovery manifest digest differs")) + )); + for (leader, epoch) in [([88; 16], "2"), ([1; 16], "3")] { + let mut changed: RawManifest = + serde_json::from_slice(&serde_json::to_vec(&raw).unwrap()).unwrap(); + changed.leader_session = encode_hex(&leader); + changed.log_epoch = epoch.into(); + let digest = install_manifest_bytes(&fixture, serde_json::to_vec(&changed).unwrap()).await; + assert!(matches!( + fixture + .manifests + .load_manifest(recovery.leader_session, recovery.log_epoch, digest) + .await, + Err(Error::Node("recovery manifest path scope differs")) + )); + } +} + +#[tokio::test] +async fn manifest_inventory_and_overlay_reject_duplicate_and_reordered_scopes() { + let fixture = recovery_fixture_with_scopes(None, true).await; + for duplicate in [true, false] { + let mut raw = raw_manifest(&fixture).await; + if duplicate { + let copy = serde_json::from_slice(&serde_json::to_vec(&raw.cells[0]).unwrap()).unwrap(); + raw.cells.insert(1, copy); + } else { + raw.cells.swap(0, 1); + } + let digest = install_manifest_bytes(&fixture, serde_json::to_vec(&raw).unwrap()).await; + let mut recovery = fixture.pinned.recovery.clone(); + recovery.manifest_digest = digest; + assert!(matches!( + fixture + .manifests + .load_manifest(recovery.leader_session, recovery.log_epoch, digest) + .await, + Err(Error::Node( + "recovery manifest Cell scopes are not strictly ordered" + )) + )); + assert!(matches!( + load_error(&fixture, &recovery).await, + Error::Node("recovery manifest Cell scopes are not strictly ordered") + )); + } +} + +#[tokio::test] +async fn manifest_inventory_rejects_self_consistent_noncanonical_or_invalid_shapes() { + let fixture = recovery_fixture().await; + let raw = raw_manifest(&fixture).await; + let valid = serde_json::to_value(&raw).unwrap(); + let mut bodies = vec![serde_json::to_vec_pretty(&raw).unwrap()]; + for variant in 0..7 { + let mut changed = valid.clone(); + match variant { + 0 => { + changed["unknown"] = serde_json::json!(true); + } + 1 => { + changed["version"] = serde_json::json!(2); + } + 2 => { + changed["cells"] = serde_json::json!([]); + } + 3 => { + changed["cells"] = serde_json::json!(vec![valid["cells"][0].clone(); 1_025]); + } + 4 => { + changed["cells"][0]["first_node_sequence"] = serde_json::json!("0"); + } + 5 => { + changed["cells"][0]["cell_epoch"] = serde_json::json!("06"); + } + _ => { + changed["cells"][0]["final_txid"] = changed["cells"][0]["predecessor_txid"].clone(); + } + } + bodies.push(serde_json::to_vec(&changed).unwrap()); + } + for body in bodies { + let digest = install_manifest_bytes(&fixture, body).await; + let recovery = &fixture.pinned.recovery; + assert!( + fixture + .manifests + .load_manifest(recovery.leader_session, recovery.log_epoch, digest) + .await + .is_err() + ); + } +} + +#[tokio::test] +async fn manifest_inventory_rejects_oversized_object_before_decode() { + let fixture = recovery_fixture().await; + let digest = + install_manifest_bytes(&fixture, vec![b' '; MAX_MANIFEST_BYTES as usize + 1]).await; + let recovery = &fixture.pinned.recovery; + assert!(matches!( + fixture + .manifests + .load_manifest(recovery.leader_session, recovery.log_epoch, digest) + .await, + Err(Error::Storage(_)) + )); +} diff --git a/crates/cellule-runtime/src/registry/builder/mod.rs b/crates/cellule-runtime/src/registry/builder/mod.rs index 6e1003ae..bc728ae7 100644 --- a/crates/cellule-runtime/src/registry/builder/mod.rs +++ b/crates/cellule-runtime/src/registry/builder/mod.rs @@ -22,7 +22,9 @@ pub struct RegistryBuilder { pub(super) build: BuildDescriptor, pub(super) modules: Vec<&'static ModuleDescriptor>, pub(super) commands: BTreeMap, + pub(super) lease_commands: BTreeSet, pub(super) queries: BTreeMap, + pub(super) lease_queries: BTreeSet, pub(super) workflow_definitions: HashMap<(String, [u8; 32]), Vec>, pub(super) activities: BTreeMap, pub(super) activity_claims: BTreeSet, @@ -47,7 +49,9 @@ impl RegistryBuilder { build, modules: Vec::new(), commands: BTreeMap::new(), + lease_commands: BTreeSet::new(), queries: BTreeMap::new(), + lease_queries: BTreeSet::new(), workflow_definitions: HashMap::new(), activities: BTreeMap::new(), activity_claims: BTreeSet::new(), @@ -80,18 +84,39 @@ impl RegistryBuilder { /// Binds one descriptor command key to its monomorphized typed function. pub fn bind_command(&mut self) -> std::result::Result<(), RegistryError> { let key = BindingKey::new(C::MODULE, C::ID, C::CODEC_VERSION)?; - if self.commands.insert(key, typed_command::).is_some() { - return Err(Error::Registry("duplicate command binding")); + if let std::collections::btree_map::Entry::Vacant(entry) = self.commands.entry(key) { + entry.insert(typed_command::); + Ok(()) + } else { + Err(Error::Registry("duplicate command binding")) } + } + + // Only native primitives can register this capability. Each such handler + // validates an existing exact lease; application commands cannot declare + // themselves completions or supply a management/wire admission flag. + pub(crate) fn bind_lease_command(&mut self) -> Result<()> { + self.bind_command::()?; + self.lease_commands + .insert(BindingKey::new(C::MODULE, C::ID, C::CODEC_VERSION)?); Ok(()) } /// Binds one descriptor query key to its monomorphized typed function. pub fn bind_query(&mut self) -> std::result::Result<(), RegistryError> { let key = BindingKey::new(Q::MODULE, Q::ID, Q::CODEC_VERSION)?; - if self.queries.insert(key, typed_query::).is_some() { - return Err(Error::Registry("duplicate query binding")); + if let std::collections::btree_map::Entry::Vacant(entry) = self.queries.entry(key) { + entry.insert(typed_query::); + Ok(()) + } else { + Err(Error::Registry("duplicate query binding")) } + } + + pub(crate) fn bind_lease_query(&mut self) -> Result<()> { + self.bind_query::()?; + self.lease_queries + .insert(BindingKey::new(Q::MODULE, Q::ID, Q::CODEC_VERSION)?); Ok(()) } @@ -496,8 +521,10 @@ impl RegistryBuilder { module_retained_codes, module_names, commands: self.commands, + lease_commands: self.lease_commands, command_descriptors, queries: self.queries, + lease_queries: self.lease_queries, query_descriptors, namespace_modules: namespace_owners, activities: self.activities, @@ -545,8 +572,10 @@ pub struct Registry { pub(super) module_migrations: BTreeMap<&'static str, &'static [MigrationDescriptor]>, pub(super) module_retained_codes: BTreeMap<&'static str, &'static [RetainedCodeDescriptor]>, pub(super) commands: BTreeMap, + pub(super) lease_commands: BTreeSet, pub(super) command_descriptors: BTreeMap, pub(super) queries: BTreeMap, + pub(super) lease_queries: BTreeSet, pub(super) query_descriptors: BTreeMap, pub(super) namespace_modules: HashMap, pub(super) activities: BTreeMap, @@ -562,6 +591,27 @@ pub struct Registry { mod run; impl Registry { + pub(crate) fn command_is_lease_completion( + &self, + module: &str, + id: u32, + codec: u32, + ) -> Result { + Ok(self + .lease_commands + .contains(&BindingKey::new(module, id, codec)?)) + } + + pub(crate) fn query_is_lease_validation( + &self, + module: &str, + id: u32, + codec: u32, + ) -> Result { + Ok(self + .lease_queries + .contains(&BindingKey::new(module, id, codec)?)) + } /// Returns the canonical release descriptor bytes. #[must_use] pub fn release_bytes(&self) -> &[u8] { diff --git a/crates/cellule-runtime/tests/fleet.rs b/crates/cellule-runtime/tests/fleet.rs index b9c40140..717962dc 100644 --- a/crates/cellule-runtime/tests/fleet.rs +++ b/crates/cellule-runtime/tests/fleet.rs @@ -1,6 +1,7 @@ //! Fleet-level integration tests: node leases, placement, pressure, cluster receipts. mod fleet { + pub mod follower_inventory; pub mod node_log_transport; pub mod placement_properties; pub mod pressure; diff --git a/crates/cellule-runtime/tests/fleet/follower_inventory.rs b/crates/cellule-runtime/tests/fleet/follower_inventory.rs new file mode 100644 index 00000000..d9c41f1f --- /dev/null +++ b/crates/cellule-runtime/tests/fleet/follower_inventory.rs @@ -0,0 +1,163 @@ +//! Public persisted-tail inventory and expired-owner authority discovery. + +use std::sync::Arc; + +use bytes::Bytes; +use cellule_ltx::{Db, NodeFrameScope, encode_node_frame}; +use cellule_runtime::{ + Error, + fleet::admission::NodeAdmission, + follower::{FollowerLaneState, FollowerStore}, + identity::{Digest, NodeId, SessionId}, + ltx::{CellStorageLayout, DiskBudget, Limits}, + node::{ + LogLeaderState, NodeAdvertisement, NodeCapacity, NodeDirectory, NodeFailureDomain, NodeMode, + }, +}; +use cellule_store::Store; +use ed25519_dalek::SigningKey; +use object_store::{memory::InMemory, path::Path}; + +const NOW: i64 = 1_000_000; + +fn advertisement(id: u8, at: i64) -> NodeAdvertisement { + NodeAdvertisement::sign( + NodeId::from_bytes([id; 16]), + SessionId::from_bytes([id; 16]), + "https://node.internal:8789".into(), + Digest::from_bytes([2; 32]), + Digest::from_bytes([3; 32]), + Digest::from_bytes([4; 32]), + Digest::from_bytes([5; 32]), + &SigningKey::from_bytes(&[7; 32]), + 1, + at, + at + 10_000, + vec![Digest::from_bytes([6; 32])], + vec![1], + NodeFailureDomain::default(), + NodeCapacity { + free_memory_bytes: 1 << 30, + free_disk_bytes: 1 << 30, + follower_free_bytes: if id == 1 { 0 } else { 1 << 30 }, + follower_retained_bytes: 0, + job_credits: 3, + log_protocol: 1, + }, + ) + .unwrap() +} + +#[tokio::test] +async fn cold_foreign_tail_and_dead_owner_remain_visible_through_cordon_and_recovery_claim() { + let limits = Limits::default(); + let source = tempfile::tempdir().unwrap(); + let mut database = Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); + database + .transaction(|transaction| { + transaction + .execute_batch("CREATE TABLE events(value INTEGER); INSERT INTO events VALUES (17)") + }) + .unwrap(); + let capture = database.capture().unwrap(); + let segment = capture.segments.first().unwrap(); + let frame = encode_node_frame( + NodeFrameScope { + leader_session: [1; 16], + log_epoch: 4, + node_sequence: 1, + application: [3; 16], + cell: [4; 32], + incarnation: [5; 16], + cell_epoch: 6, + commit_sequence: 1, + }, + segment.info().clone(), + Bytes::from(std::fs::read(segment.path()).unwrap()), + limits, + ) + .unwrap() + .encoded() + .clone(); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("root"), + [9; 16], + ); + let directory = NodeDirectory::new( + layout, + Digest::from_bytes([2; 32]), + Digest::from_bytes([4; 32]), + Digest::from_bytes([5; 32]), + ); + let leader = SessionId::from_bytes([1; 16]); + let member = NodeId::from_bytes([2; 16]); + let created = directory.create(advertisement(1, NOW), NOW).await.unwrap(); + directory.create(advertisement(2, NOW), NOW).await.unwrap(); + directory + .recruit_log(&created, 4, 1, 10, NOW + 1) + .await + .unwrap(); + let local_root = tempfile::tempdir().unwrap(); + let disk = DiskBudget::new(1 << 30); + let gate = NodeAdmission::default(); + let store = FollowerStore::open(local_root.path().to_owned(), limits, disk.clone()) + .unwrap() + .with_node_admission(gate.clone()); + assert_eq!( + store + .append(leader, 4, vec![frame.clone()], 0) + .await + .unwrap() + .durable_through, + 1 + ); + gate.cordon().unwrap(); + assert!(matches!( + store.append(leader, 5, vec![frame.clone()], 0).await, + Err(Error::CellDraining) + )); + drop(store); + let reopened = FollowerStore::open(local_root.path().to_owned(), limits, disk) + .unwrap() + .with_node_admission(gate); + let local = reopened + .fleet_lanes_page(None, 128, NOW + 20_000) + .await + .unwrap(); + assert_eq!(local.mode(), NodeMode::Cordoned); + assert_eq!(local.total_lanes(), 1); + assert_eq!(local.unretired_lanes(), 1); + assert_eq!(local.entries()[0].leader, leader); + assert_eq!(local.entries()[0].epoch, 4); + assert_eq!(local.entries()[0].state, FollowerLaneState::Open); + let remote = directory + .follower_logs_page(member, None, 128, NOW + 20_000) + .await + .unwrap(); + assert_eq!(remote.total_logs(), 1); + assert_eq!(remote.entries()[0].leader_state, LogLeaderState::Expired); + directory + .create(advertisement(3, NOW + 20_000), NOW + 20_000) + .await + .unwrap(); + directory + .claim_expired(leader, SessionId::from_bytes([3; 16]), NOW + 20_001) + .await + .unwrap(); + let recovering = directory + .follower_logs_page(member, None, 128, NOW + 20_002) + .await + .unwrap(); + assert_eq!(recovering.total_logs(), 1); + assert_eq!(recovering.entries()[0].leader_state, LogLeaderState::Fenced); + assert!(recovering.entries()[0].log.recovery().is_some()); + // Discovery does not seal, retire, or delete the tail. Ordinary recovery + // still obtains the same acknowledged frame under its own authority. + assert_eq!(reopened.seal(leader, 4).await.unwrap().durable_through, 1); + let retained = reopened.read_tail_page(leader, 4, 1).await.unwrap(); + assert_eq!(retained.frames, vec![frame]); + assert!(retained.next_sequence.is_none()); + assert!(reopened.retained_bytes() > 0); + database.close().unwrap(); +} diff --git a/crates/cellule-runtime/tests/fleet/placement_properties.rs b/crates/cellule-runtime/tests/fleet/placement_properties.rs index ffdf4d05..9869230e 100644 --- a/crates/cellule-runtime/tests/fleet/placement_properties.rs +++ b/crates/cellule-runtime/tests/fleet/placement_properties.rs @@ -107,9 +107,11 @@ fn a_dense_member_hands_over_to_an_empty_one() { disk_bytes: 1 << 20, job_credits: 1, resident_since_ms: NOW_MS - 120_000, + last_used_ms: NOW_MS - 1, last_moved_at_ms: None, stable_observations: 3, settled: true, + maintenance: false, }; let intents = planner .plan_transfers(NOW_MS, &observations, &[demand], Some(&balance)) @@ -157,9 +159,11 @@ fn small_fleets_converge_after_owner_loss_with_fresh_settled_views() { disk_bytes: 1 << 20, job_credits: 1, resident_since_ms: now - 120_000, + last_used_ms: now - 1, last_moved_at_ms: None, stable_observations: 3, settled: true, + maintenance: false, }) .collect::>(); let intents = planner @@ -216,9 +220,11 @@ fn donations_do_not_overfill_a_preferred_receivers_weighted_share() { disk_bytes: 1, job_credits: 1, resident_since_ms: NOW_MS - 120_000, + last_used_ms: NOW_MS - 1, last_moved_at_ms: None, stable_observations: 3, settled: true, + maintenance: false, }) .collect::>(); // Its whole-Cell margin is zero at target two; the existing Cell leaves @@ -237,6 +243,52 @@ fn donations_do_not_overfill_a_preferred_receivers_weighted_share() { assert_eq!(destinations, vec![session(1), session(2)]); } +/// Pressure movement and local idle eviction consume the same actor recency, +/// in opposite orders. Recency ties still have an identity-based total order. +#[test] +fn pressure_recency_order_is_permutation_invariant() { + let planner = PlacementPlanner::default(); + let donor = observation(1, 6, 8, PlacementPressure::Shedding, false); + let receiver = observation(2, 0, 8, PlacementPressure::Normal, false); + let demands = (0..6) + .map(|index| CellTransferDemand { + cell: CellId::from_bytes([index; 32]), + source: donor.session, + generation: 1, + memory_bytes: 1 << 20, + disk_bytes: 1 << 20, + job_credits: 1, + resident_since_ms: NOW_MS - 1_000, + last_used_ms: NOW_MS - i64::from(5 - index) * 10, + last_moved_at_ms: None, + stable_observations: 2, + settled: true, + maintenance: false, + }) + .collect::>(); + let expected = planner + .plan_transfers(NOW_MS, &[donor, receiver], &demands, None) + .unwrap(); + assert_eq!( + expected + .iter() + .map(|intent| intent.cell) + .collect::>(), + vec![CellId::from_bytes([5; 32]), CellId::from_bytes([4; 32])] + ); + for offset in 0..demands.len() { + let mut shuffled = demands.clone(); + shuffled.rotate_left(offset); + shuffled.reverse(); + assert_eq!( + planner + .plan_transfers(NOW_MS, &[receiver, donor], &shuffled, None) + .unwrap(), + expected + ); + } +} + proptest! { #![proptest_config(ProptestConfig { cases: 32, ..ProptestConfig::default() })] @@ -322,9 +374,11 @@ proptest! { disk_bytes: bytes, job_credits: 1, resident_since_ms: NOW_MS - 1_000, + last_used_ms: NOW_MS - 1, last_moved_at_ms: None, stable_observations: 2, settled: true, + maintenance: false, } }) .collect::>(); diff --git a/crates/cellule-runtime/tests/primitives/blob_cron/maintenance/mod.rs b/crates/cellule-runtime/tests/primitives/blob_cron/maintenance/mod.rs new file mode 100644 index 00000000..bc624469 --- /dev/null +++ b/crates/cellule-runtime/tests/primitives/blob_cron/maintenance/mod.rs @@ -0,0 +1,470 @@ +//! Accepted Cron dispatch and unclaimed occurrences across native maintenance. + +use super::*; +use std::{sync::mpsc, time::Duration}; + +use crate::support::fixtures::mutation_identity; +use cellule_runtime::Error; +use cellule_runtime::cell::actor::{CellInventoryEntry, MaintenanceCellRelease}; +use cellule_runtime::cell::executor::{HandlerOutcome, Resolution}; +use cellule_runtime::client::InvocationError; +use cellule_runtime::codec::{BoundedDecoder, BoundedEncoder, WireValue}; +use cellule_runtime::control::ControlState; +use cellule_runtime::fleet::scheduler::scheduler_tick; +use cellule_runtime::primitives::effects::{EffectClaimRequest, EffectLeaseOutcome, EffectSource}; + +mod peer; + +fn node_runtime(session: SessionId) -> CellRuntime { + // These are independent nodes. The convenience default Host shares a + // process-wide disk budget, including reservations held by other runtimes. + CellRuntime::new_with_replica_host( + SqlWorkerPool::new(1, 10).unwrap(), + 16 << 20, + session, + cellule_ltx::Host::default().with_local_disk_budget(cellule_ltx::DiskBudget::new(1 << 30)), + ) + .unwrap() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn accepted_tick_and_due_occurrences_survive_quiescence_and_exact_root_handoff() { + let registry = registry(); + let tenant = TenantId::from_bytes([50; 16]); + let application = ApplicationId::from_bytes([51; 16]); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("cron-maintenance"), + *application.as_bytes(), + ); + let authority = CellAuthority::new(layout.clone()); + let catalog = CellCatalog::new(layout.clone(), tenant); + let directory = tempfile::TempDir::new().unwrap(); + let target = + CellTarget::new(tenant, application, CRON_NAMESPACE, &0_u32.to_be_bytes()).unwrap(); + let incarnation = IncarnationId::from_bytes([52; 16]); + let proof = catalog + .provision( + CatalogEntry::new( + &target, + CatalogRole::Cron, + registry.module_code(CRON_MODULE).unwrap(), + 1, + ) + .unwrap(), + ) + .await + .unwrap(); + let replica = CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + Limits::default(), + ) + .unwrap(); + let session = SessionId::from_bytes([53; 16]); + let control = authority + .create_initial( + &proof, + incarnation, + Owner { + session, + endpoint: "https://cron-source.internal:8081".into(), + }, + ) + .await + .unwrap(); + let runtime = node_runtime(session); + let handle = runtime + .bootstrap( + proof.clone(), + replica.clone(), + authority.clone(), + control, + directory.path().join("source.sqlite"), + cellule_runtime::primitives::cron::install_cron_schema, + ) + .await + .unwrap(); + let client = CellClient::local(registry.clone(), handle.clone()); + let cron = CronNamespace::::new(client.clone(), tenant, application).unwrap(); + let first_due = now_ms(); + let second_due = first_due + 1_000; + for (id, due, identity) in [([54; 16], first_due, 55), ([56; 16], second_due, 57)] { + cron.mutate( + mutation_identity_window(identity, first_due, first_due + 60_000), + CronMutation::Upsert { + schedule_id: id, + target_index: 0, + target_partition: b"destination".to_vec(), + payload: id.to_vec(), + interval_ms: 60_000, + next_due_ms: due, + }, + ) + .await + .unwrap(); + } + let page = runtime.fleet_cells_page(None, 128).await.unwrap(); + let CellInventoryEntry::Owned(owner) = &page.entries()[0] else { + panic!("missing Cron owner") + }; + let generation = owner.generation; + let epoch = owner.position.as_ref().unwrap().epoch; + drop(page); + + // Gate the actual worker transaction, rather than inferring admission from + // a spawned future or a sleep. Use the public scheduler with pinned logical + // time so only the first schedule fires in this accepted Tick. + let identity = mutation_identity(58); + let digest = Digest::from_bytes([59; 32]); + let started = Arc::new(tokio::sync::Notify::new()); + let (release_tx, release_rx) = mpsc::channel(); + let executing = { + let handle = handle.clone(); + let target = target.clone(); + let started = started.clone(); + tokio::spawn(async move { + handle + .execute(identity, digest, first_due, 8, 5, move |transaction| { + started.notify_one(); + release_rx.recv().unwrap(); + let tick = + scheduler_tick(transaction, &target, first_due, &[], None, CRON_TARGETS)?; + let mut encoder = BoundedEncoder::new(5)?; + MaintenanceTickOutcome::Applied { + processed: tick.processed, + } + .encode(&mut encoder)?; + Ok(HandlerOutcome::Success(encoder.finish())) + }) + .await + }) + }; + tokio::time::timeout(Duration::from_secs(3), started.notified()) + .await + .unwrap(); + let quiesced = tokio::time::timeout( + Duration::from_secs(3), + runtime.quiesce_cell_at(target.cell_id(), session, generation, incarnation, epoch), + ) + .await; + let refused = tokio::time::timeout( + Duration::from_secs(3), + registry.run_maintenance_once( + client.clone(), + target.clone(), + mutation_identity(60), + MaintenanceTickRequest { + expected_commit_sequence: 2, + }, + ), + ) + .await; + let denied_claim = tokio::time::timeout( + Duration::from_secs(3), + EffectSource::::new(client, target.clone()).claim( + mutation_identity(69), + EffectClaimRequest { + limit: 1, + lease_ms: 30_000, + }, + ), + ) + .await; + // The second occurrence becomes due while foreground admission is closed. + let remaining = second_due.saturating_sub(now_ms()); + if remaining > 0 { + tokio::time::sleep(Duration::from_millis(remaining as u64)).await; + } + release_tx.send(()).unwrap(); + let accepted = executing.await.unwrap().unwrap(); + quiesced.unwrap().unwrap(); + assert!(matches!( + refused.unwrap(), + Err(InvocationError::NotStarted(Error::CellDraining)) + )); + assert!(matches!( + denied_claim.unwrap(), + Err(InvocationError::NotStarted(Error::CellDraining)) + )); + let cellule_runtime::cell::executor::StoredOutcome::Success { + result, + commit_sequence, + } = &accepted + else { + panic!("accepted Tick did not publish") + }; + let mut decoder = BoundedDecoder::new(result, 5).unwrap(); + assert_eq!( + MaintenanceTickOutcome::decode(&mut decoder).unwrap(), + MaintenanceTickOutcome::Applied { processed: 1 } + ); + decoder.finish().unwrap(); + assert_eq!(*commit_sequence, 3); + assert_eq!( + handle.resolve(identity, digest, now_ms(), 5).await.unwrap(), + Resolution::Committed(accepted.clone()) + ); + let MaintenanceCellRelease::Released(released) = runtime + .release_maintenance_cell_at( + target.cell_id(), + session, + generation, + incarnation, + epoch, + tokio::time::Instant::now() + Duration::from_secs(10), + ) + .await + .unwrap() + else { + panic!("Cron maintenance release refused") + }; + let idle = authority.load(target.cell_id()).await.unwrap().unwrap(); + assert_eq!(idle.value().state, ControlState::Idle); + assert_eq!(idle.value().root.as_ref(), Some(&released.root)); + assert_eq!(released.root.commit_sequence, 3); + assert_eq!(released.epoch, epoch); + assert_eq!(runtime.unreleased_cell_count().await.unwrap(), 0); + + let successor_session = SessionId::from_bytes([61; 16]); + let successor = node_runtime(successor_session); + let restored = successor + .acquire_idle_restored( + proof, + replica, + authority.clone(), + idle, + directory.path().join("successor.sqlite"), + Owner { + session: successor_session, + endpoint: "https://cron-successor.internal:8081".into(), + }, + ) + .await + .unwrap(); + assert_eq!( + restored + .resolve(identity, digest, now_ms(), 5) + .await + .unwrap(), + Resolution::Committed(accepted) + ); + let acquired = authority.load(target.cell_id()).await.unwrap().unwrap(); + assert_eq!(acquired.value().state, ControlState::Serving); + assert_eq!( + acquired.value().owner.as_ref().unwrap().session, + successor_session + ); + assert_eq!(acquired.value().epoch, epoch + 1); + assert_eq!(acquired.value().root.as_ref(), Some(&released.root)); + let client = CellClient::local(registry.clone(), restored.clone()); + let cron = CronNamespace::::new(client.clone(), tenant, application).unwrap(); + for (id, occurrence, due) in [([54; 16], 1, first_due + 60_000), ([56; 16], 0, second_due)] { + let CronQueryResult::Get(Some(schedule)) = cron.get(id, None).await.unwrap().output else { + panic!("schedule missing after exact-root handoff") + }; + assert_eq!(schedule.generation, 1); + assert_eq!(schedule.occurrence, occurrence); + assert_eq!(schedule.next_due_ms, due); + } + + // A stale scheduler observation publishes no occurrence. A current Tick + // generates the remaining due occurrence, then replay returns its receipt. + let stale = registry + .run_maintenance_once( + client.clone(), + target.clone(), + mutation_identity(62), + MaintenanceTickRequest { + expected_commit_sequence: 2, + }, + ) + .await + .unwrap(); + assert_eq!(stale.output, MaintenanceTickOutcome::Stale); + let tick_identity = mutation_identity(63); + let request = MaintenanceTickRequest { + expected_commit_sequence: stale.receipt.commit_sequence, + }; + let tick = registry + .run_maintenance_once(client.clone(), target.clone(), tick_identity, request) + .await + .unwrap(); + assert_eq!( + tick.output, + MaintenanceTickOutcome::Applied { processed: 1 } + ); + let replay = registry + .run_maintenance_once(client.clone(), target.clone(), tick_identity, request) + .await + .unwrap(); + assert_eq!(replay.receipt, tick.receipt); + assert_eq!(replay.output, tick.output); + + let destination_target = + CellTarget::new(tenant, application, TARGET_NAMESPACE, b"destination").unwrap(); + let destination_incarnation = IncarnationId::from_bytes([64; 16]); + let destination_session = SessionId::from_bytes([65; 16]); + let destination_proof = catalog + .provision( + CatalogEntry::new( + &destination_target, + CatalogRole::Application, + registry.module_code(TARGET_MODULE).unwrap(), + 1, + ) + .unwrap(), + ) + .await + .unwrap(); + let destination_control = authority + .create_initial( + &destination_proof, + destination_incarnation, + Owner { + session: destination_session, + endpoint: "https://cron-target.internal:8081".into(), + }, + ) + .await + .unwrap(); + let destination_runtime = node_runtime(destination_session); + let destination = destination_runtime + .bootstrap( + destination_proof, + CellReplica::new( + layout, + *destination_target.cell_id().as_bytes(), + *destination_incarnation.as_bytes(), + Limits::default(), + ) + .unwrap(), + authority, + destination_control, + directory.path().join("target.sqlite"), + |transaction| { + transaction.execute_batch(TARGET_MIGRATION)?; + Ok(()) + }, + ) + .await + .unwrap(); + let peer = peer::client(registry, destination_target, destination.clone()); + let source = EffectSource::::new(client, target); + let claims = source + .claim( + mutation_identity(66), + EffectClaimRequest { + limit: 2, + lease_ms: 30_000, + }, + ) + .await + .unwrap(); + assert_eq!(claims.output.len(), 2); + assert_ne!(claims.output[0].effect_id, claims.output[1].effect_id); + assert!( + source + .validate(claims.output.clone(), claims.receipt) + .await + .unwrap() + .output + ); + let mut delivery_sequences = Vec::new(); + for (index, claim) in claims.output.into_iter().enumerate() { + let delivered = peer.deliver(&claim, now_ms()).await.unwrap(); + assert_eq!(peer.deliver(&claim, now_ms()).await.unwrap(), delivered); + delivery_sequences.push(delivered.commit_sequence()); + assert_eq!( + source + .ack(mutation_identity(67 + index as u8), claim, Vec::new()) + .await + .unwrap() + .output, + EffectLeaseOutcome::Delivered + ); + } + delivery_sequences.sort_unstable(); + assert_eq!(delivery_sequences, vec![1, 2]); + let rows = destination + .query(8, TARGET_INPUT_LIMIT as usize, |connection| { + let mut statement = + connection.prepare("SELECT value FROM received_ticks ORDER BY rowid")?; + let rows = statement + .query_map([], |row| row.get::<_, Vec>(0))? + .collect::, _>>()?; + let mut encoder = BoundedEncoder::new(TARGET_INPUT_LIMIT)?; + encoder.write_count(rows.len())?; + for row in rows { + encoder.write_bytes(&row)?; + } + Ok(encoder.finish()) + }) + .await + .unwrap(); + let mut decoder = BoundedDecoder::new(&rows, TARGET_INPUT_LIMIT).unwrap(); + let count = decoder.read_count().unwrap(); + assert_eq!(count, 2); + let rows = (0..count) + .map(|_| decoder.read_bytes().unwrap().to_vec()) + .collect::>(); + decoder.finish().unwrap(); + let mut invocations = rows + .iter() + .map(|bytes| { + let mut decoder = BoundedDecoder::new(bytes, TARGET_INPUT_LIMIT).unwrap(); + let invocation = CronInvocation::decode(&mut decoder).unwrap(); + decoder.finish().unwrap(); + invocation + }) + .collect::>(); + invocations.sort_by_key(|invocation| invocation.schedule_id); + assert_eq!( + invocations, + vec![ + CronInvocation { + schedule_id: [54; 16], + generation: 1, + occurrence: 1, + scheduled_at_ms: first_due, + payload: vec![54; 16] + }, + CronInvocation { + schedule_id: [56; 16], + generation: 1, + occurrence: 1, + scheduled_at_ms: second_due, + payload: vec![56; 16] + }, + ] + ); + for (id, due) in [ + ([54; 16], first_due + 60_000), + ([56; 16], second_due + 60_000), + ] { + let CronQueryResult::Get(Some(schedule)) = cron.get(id, None).await.unwrap().output else { + panic!("delivered schedule missing") + }; + assert_eq!(schedule.occurrence, 1); + assert_eq!(schedule.next_due_ms, due); + } + for runtime in [&runtime, &successor, &destination_runtime] { + runtime.shutdown().await.unwrap(); + let stats = runtime.stats(); + assert_eq!(stats.active_cells(), 0); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.file_descriptors(), 0); + assert_eq!(stats.retained_bytes(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.primitive_jobs(), 0); + assert_eq!(stats.hydration_jobs(), 0); + assert_eq!(stats.io_slots(), 0); + assert_eq!(stats.blocking_jobs(), 0); + assert_eq!(stats.recovery_jobs(), 0); + assert_eq!(stats.dirty_jobs(), 0); + assert_eq!(stats.scratch_units(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); + assert_eq!(stats.unpublished_node_log_bytes(), 0); + } +} diff --git a/crates/cellule-runtime/tests/primitives/blob_cron/maintenance/peer.rs b/crates/cellule-runtime/tests/primitives/blob_cron/maintenance/peer.rs new file mode 100644 index 00000000..ad10d319 --- /dev/null +++ b/crates/cellule-runtime/tests/primitives/blob_cron/maintenance/peer.rs @@ -0,0 +1,101 @@ +//! Signed in-process transport to the actual destination command and Inbox. + +use super::*; +use cellule_runtime::cell::actor::CellHandle; +use cellule_runtime::peer::{ + EffectPeerClient, PeerAuthorizer, PeerCellResolver, PeerDispatcher, PeerPrincipal, + PeerRoundTrip, PeerSigner, PeerVerifier, VerifiedPeerRequest, +}; +use std::{future::Future, pin::Pin}; + +pub(super) fn client( + registry: Arc, + target: CellTarget, + handle: CellHandle, +) -> EffectPeerClient { + let session = SessionId::from_bytes([70; 16]); + let signer = Arc::new(PeerSigner::new( + session, + registry.release_digest(), + ed25519_dalek::SigningKey::from_bytes(&[71; 32]), + )); + EffectPeerClient::new( + signer.clone(), + PeerPrincipal { + issuer: "cellule:test".into(), + subject: "cron-source".into(), + actions: vec!["cron.invoke".into()], + }, + Arc::new(Loopback { + verifier: Arc::new(PeerVerifier::new( + session, + registry.release_digest(), + signer.verifying_key(), + )), + dispatcher: Arc::new(PeerDispatcher::new( + registry, + Arc::new(Resolver { target, handle }), + Arc::new(Authorizer), + )), + }), + ) +} + +struct Resolver { + target: CellTarget, + handle: CellHandle, +} + +impl PeerCellResolver for Resolver { + fn resolve( + &self, + target: CellTarget, + ) -> Pin> + Send + 'static>> { + let matches = target == self.target; + let handle = self.handle.clone(); + Box::pin(async move { + if matches { + Ok(handle) + } else { + Err(Error::CellNotActive) + } + }) + } +} + +struct Authorizer; + +impl PeerAuthorizer for Authorizer { + fn authorize(&self, request: &VerifiedPeerRequest) -> cellule_runtime::Result<()> { + if request.permits("cron.invoke") { + Ok(()) + } else { + Err(Error::PeerAuthorization("missing Cron invocation action")) + } + } +} + +struct Loopback { + verifier: Arc, + dispatcher: Arc, +} + +impl PeerRoundTrip for Loopback { + fn send( + &self, + target: CellTarget, + request: Vec, + _remaining_ms: u32, + ) -> Pin>> + Send + 'static>> { + let verifier = self.verifier.clone(); + let dispatcher = self.dispatcher.clone(); + Box::pin(async move { + let now = now_ms(); + let verified = verifier.verify(&request, now)?; + if verified.target() != &target { + return Err(Error::Peer("round trip target changed")); + } + dispatcher.dispatch_bytes(&verified, now).await + }) + } +} diff --git a/crates/cellule-runtime/tests/primitives/blob_cron.rs b/crates/cellule-runtime/tests/primitives/blob_cron/mod.rs similarity index 92% rename from crates/cellule-runtime/tests/primitives/blob_cron.rs rename to crates/cellule-runtime/tests/primitives/blob_cron/mod.rs index 69a15dc6..9400dcd8 100644 --- a/crates/cellule-runtime/tests/primitives/blob_cron.rs +++ b/crates/cellule-runtime/tests/primitives/blob_cron/mod.rs @@ -37,13 +37,13 @@ use crate::support::fixtures::{mutation_identity_window, now_ms}; const BLOB_MODULE: &str = "blob-test"; const BLOB_NAMESPACE: NamespaceId = NamespaceId::from_bytes([1; 16]); -const BLOB_MIGRATION: &str = include_str!("../../src/migrations/blob.sql"); +const BLOB_MIGRATION: &str = include_str!("../../../src/migrations/blob.sql"); const CRON_MODULE: &str = "cron-test"; const CRON_NAMESPACE: NamespaceId = NamespaceId::from_bytes([2; 16]); -const CRON_MIGRATION: &str = include_str!("../../src/migrations/cron.sql"); +const CRON_MIGRATION: &str = include_str!("../../../src/migrations/cron.sql"); const TARGET_MODULE: &str = "cron-target-test"; const TARGET_NAMESPACE: NamespaceId = NamespaceId::from_bytes([3; 16]); -const TARGET_MIGRATION: &str = "CREATE TABLE cron_target(value BLOB) STRICT;"; +const TARGET_MIGRATION: &str = "CREATE TABLE received_ticks(value BLOB) STRICT;"; const TARGET_INPUT_LIMIT: u32 = 300 * 1024; const CRON_TARGETS: &[CronTarget] = &[CronTarget::new( TARGET_MODULE, @@ -55,8 +55,17 @@ const CRON_TARGETS: &[CronTarget] = &[CronTarget::new( const BLOB_COMMANDS: &[OperationDescriptor] = &[operation(1, 300 * 1024, 64), operation(2, 8, 5)]; const BLOB_QUERIES: &[OperationDescriptor] = &[operation(1, 4 * 1024, 600 * 1024)]; -const CRON_COMMANDS: &[OperationDescriptor] = &[operation(1, 300 * 1024, 16), operation(2, 8, 5)]; -const CRON_QUERIES: &[OperationDescriptor] = &[operation(1, 64, 600 * 1024)]; +const CRON_COMMANDS: &[OperationDescriptor] = &[ + operation(1, 300 * 1024, 16), + operation(2, 8, 5), + operation(3, 8, 1 << 20), + operation(4, 1 << 20, 16), +]; +const CRON_QUERIES: &[OperationDescriptor] = &[ + operation(1, 64, 600 * 1024), + operation(2, 1 << 20, 1), + operation(3, 32, 1 << 20), +]; const TARGET_COMMANDS: &[OperationDescriptor] = &[operation(9, TARGET_INPUT_LIMIT, 1)]; struct TestBlob; @@ -107,6 +116,14 @@ impl CronModule for TestCron { const QUERY_ID: u32 = 1; } +impl cellule_runtime::primitives::effects::EffectModule for TestCron { + const MODULE: &'static str = CRON_MODULE; + const CLAIM_COMMAND_ID: u32 = 3; + const LEASE_COMMAND_ID: u32 = 4; + const VALIDATE_QUERY_ID: u32 = 2; + const STATUS_QUERY_ID: u32 = 3; +} + impl CellModule for TestCron { const NAME: &'static str = CRON_MODULE; @@ -124,7 +141,8 @@ impl CellModule for TestCron { } fn register(self, registry: &mut RegistryBuilder) -> cellule_runtime::Result<()> { - register_cron::(registry) + register_cron::(registry)?; + cellule_runtime::primitives::effects::register_effect_delivery::(registry) } } @@ -140,9 +158,19 @@ impl Command for ReceiveCron { type Output = (); fn execute( - _: &mut CommandContext<'_, '_>, - _: Self::Input, + context: &mut CommandContext<'_, '_>, + input: Self::Input, ) -> cellule_runtime::Result> { + use cellule_runtime::codec::{BoundedEncoder, WireValue}; + use cellule_runtime::primitives::sql::{SqlBatch, SqlStatement, SqlValue}; + let mut encoder = BoundedEncoder::new(TARGET_INPUT_LIMIT)?; + input.encode(&mut encoder)?; + context.sql(&SqlBatch { + statements: vec![SqlStatement { + sql: "INSERT INTO received_ticks(value) VALUES (?)".into(), + parameters: vec![SqlValue::Blob(encoder.finish())], + }], + })?; Ok(CommandResult::Success(())) } } @@ -593,3 +621,5 @@ const fn operation(id: u32, input_limit: u32, output_limit: u32) -> OperationDes output_limit, } } + +mod maintenance; diff --git a/crates/cellule-runtime/tests/primitives/queue.rs b/crates/cellule-runtime/tests/primitives/queue.rs index d5240da4..e270930b 100644 --- a/crates/cellule-runtime/tests/primitives/queue.rs +++ b/crates/cellule-runtime/tests/primitives/queue.rs @@ -327,5 +327,6 @@ fn send_request(producer: u8, payload: &[u8], available_at_ms: i64) -> QueueSend } mod lease; +mod maintenance; mod namespace; mod send; diff --git a/crates/cellule-runtime/tests/primitives/queue/maintenance.rs b/crates/cellule-runtime/tests/primitives/queue/maintenance.rs new file mode 100644 index 00000000..f72f94bf --- /dev/null +++ b/crates/cellule-runtime/tests/primitives/queue/maintenance.rs @@ -0,0 +1,688 @@ +//! Exact native Queue completion during sticky foreground quiescence. + +use super::*; + +#[tokio::test] +async fn maintenance_quiescence_preserves_native_queue_completion_and_validation() { + let fixture = queue_fixture().await; + let runtime = &fixture.runtime; + let handle = &fixture.handle; + let queue = &fixture.queue; + let cell = fixture.target.cell_id(); + let incarnation = fixture.incarnation; + let first_session = fixture.session; + use cellule_runtime::Error; + use cellule_runtime::cell::actor::CellInventoryEntry; + use cellule_runtime::primitives::maintenance_readiness::MaintenanceWorkBlocker; + + let sent = queue + .send( + mutation_identity(20), + QueueSendRequest { + producer_id: [20; 16], + payload: b"maintenance-job".to_vec(), + available_at_ms: now_ms(), + }, + ) + .await + .unwrap(); + assert!(matches!(sent.output, QueueSendOutcome::Sent { .. })); + let claimed = queue + .claim( + mutation_identity(21), + 0, + QueueClaimRequest { + limit: 1, + lease_ms: 30_000, + }, + ) + .await + .unwrap(); + assert_eq!(claimed.output.len(), 1); + let page = runtime.fleet_cells_page(None, 128).await.unwrap(); + let CellInventoryEntry::Owned(owner) = &page.entries()[0] else { + panic!("owner missing"); + }; + let generation = owner.generation; + let epoch = owner.position.as_ref().unwrap().epoch; + drop(page); + for (source, generation, incarnation, epoch) in [ + ( + SessionId::from_bytes([90; 16]), + generation, + incarnation, + epoch, + ), + (first_session, generation + 1, incarnation, epoch), + ( + first_session, + generation, + IncarnationId::from_bytes([90; 16]), + epoch, + ), + (first_session, generation, incarnation, epoch + 1), + ] { + assert!(matches!( + runtime + .quiesce_cell_at(cell, source, generation, incarnation, epoch) + .await, + Err(Error::Fenced) + )); + assert!(queue.info(0, None).await.is_ok()); + } + runtime + .quiesce_cell_at(cell, first_session, generation, incarnation, epoch) + .await + .unwrap(); + runtime + .quiesce_cell_at(cell, first_session, generation, incarnation, epoch) + .await + .unwrap(); + let page = runtime.fleet_cells_page(None, 128).await.unwrap(); + let CellInventoryEntry::Owned(owner) = &page.entries()[0] else { + panic!("owner missing"); + }; + assert!(owner.quiescing); + drop(page); + assert!(matches!( + queue + .send( + mutation_identity(22), + QueueSendRequest { + producer_id: [22; 16], + payload: vec![1], + available_at_ms: now_ms() + } + ) + .await, + Err(InvocationError::NotStarted(Error::CellDraining)) + )); + assert!(matches!( + queue + .claim( + mutation_identity(23), + 0, + QueueClaimRequest { + limit: 1, + lease_ms: 10_000 + } + ) + .await, + Err(InvocationError::NotStarted(Error::CellDraining)) + )); + assert!(matches!( + queue.info(0, None).await, + Err(InvocationError::NotStarted(Error::CellDraining)) + )); + // Raw callbacks cannot assert the native completion capability. + assert!(matches!( + handle.query(1, 1, |_| panic!("raw query admitted")).await, + Err(Error::CellDraining) + )); + assert!( + queue + .validate_claim(0, claimed.output.clone(), Some(claimed.receipt)) + .await + .unwrap() + .output + ); + let mut incorrect = claimed.output.clone(); + incorrect[0].token = [99; 16]; + assert!( + !queue + .validate_claim(0, incorrect, Some(claimed.receipt)) + .await + .unwrap() + .output + ); + // Inventory uses the same serialized connection and never revokes a lease. + tokio::time::timeout(std::time::Duration::from_secs(5), async { + loop { + let page = runtime.fleet_cells_page(None, 128).await.unwrap(); + if let CellInventoryEntry::Owned(owner) = &page.entries()[0] + && owner + .maintenance_work + .is_some_and(|work| work.has_blocker(MaintenanceWorkBlocker::QueueLease)) + { + break; + } + drop(page); + tokio::time::sleep(std::time::Duration::from_millis(25)).await; + } + }) + .await + .unwrap(); + let acked = queue + .ack( + mutation_identity(24), + 0, + claimed.output[0].message_id, + claimed.output[0].token, + ) + .await + .unwrap(); + assert_eq!( + acked.output, + QueueLeaseOutcome::Applied { + state: QueueState::Acked, + lease_until_ms: None + } + ); + assert!( + !queue + .validate_claim(0, claimed.output, Some(acked.receipt)) + .await + .unwrap() + .output + ); + assert!(matches!( + queue.info(0, None).await, + Err(InvocationError::NotStarted(Error::CellDraining)) + )); + tokio::time::timeout(std::time::Duration::from_secs(5), async { + loop { + let page = runtime.fleet_cells_page(None, 128).await.unwrap(); + if let CellInventoryEntry::Owned(owner) = &page.entries()[0] + && owner + .maintenance_work + .is_some_and(|work| work.is_transferable()) + { + assert!(owner.quiescing); + break; + } + drop(page); + tokio::time::sleep(std::time::Duration::from_millis(25)).await; + } + }) + .await + .unwrap(); + handle.drain().await.unwrap(); + runtime.shutdown().await.unwrap(); + assert_eq!(runtime.stats().retained_bytes(), 0); +} + +// Duplicate application bindings cannot replace native handlers while keeping +// their private completion classification, even if the caller ignores Err. +struct ReplacementLease; +impl cellule_runtime::registry::Command for ReplacementLease { + const MODULE: &'static str = QUEUE_MODULE; + const ID: u32 = 3; + const CODEC_VERSION: u32 = 1; + type Input = cellule_runtime::primitives::queue::QueueLeaseRequest; + type Output = QueueLeaseOutcome; + fn execute( + _: &mut cellule_runtime::registry::CommandContext<'_, '_>, + _: Self::Input, + ) -> cellule_runtime::Result> { + panic!("duplicate lease handler replaced native completion"); + } +} +struct ReplacementValidation; +impl cellule_runtime::registry::Query for ReplacementValidation { + const MODULE: &'static str = QUEUE_MODULE; + const ID: u32 = 1; + const CODEC_VERSION: u32 = 1; + type Input = cellule_runtime::primitives::queue::QueueValidateRequest; + type Output = bool; + fn execute( + _: &mut cellule_runtime::registry::QueryContext<'_>, + _: Self::Input, + ) -> cellule_runtime::Result { + panic!("duplicate validation handler replaced native completion"); + } +} + +struct QueueRuntimeFixture { + registry: Arc, + target: CellTarget, + incarnation: IncarnationId, + session: SessionId, + directory: tempfile::TempDir, + runtime: CellRuntime, + handle: cellule_runtime::cell::actor::CellHandle, + queue: QueueNamespace, + replica: CellReplica, + authority: CellAuthority, +} + +async fn queue_fixture() -> QueueRuntimeFixture { + let baseline = queue_registry(); + let mut builder = RegistryBuilder::new(BuildDescriptor { + source_revision: "queue-api-test".into(), + cargo_lock_digest: Digest::from_bytes([5; 32]), + }); + builder.register(TestQueue).unwrap(); + assert!(builder.bind_command::().is_err()); + assert!(builder.bind_query::().is_err()); + let registry = Arc::new(builder.finish().unwrap()); + assert_eq!(registry.release_bytes(), baseline.release_bytes()); + let target = CellTarget::new( + TenantId::from_bytes([1; 16]), + ApplicationId::from_bytes([3; 16]), + QUEUE_NAMESPACE, + &0_u32.to_be_bytes(), + ) + .unwrap(); + let cell = target.cell_id(); + let incarnation = IncarnationId::from_bytes([2; 16]); + let store = Store::new(Arc::new(InMemory::new())); + let layout = CellStorageLayout::new(store, Path::from("runtime"), [3; 16]); + let replica = CellReplica::new( + layout.clone(), + *cell.as_bytes(), + *incarnation.as_bytes(), + Limits::default(), + ) + .unwrap(); + let catalog = CellCatalog::new(layout.clone(), target.tenant()); + let proof = catalog + .provision( + CatalogEntry::new( + &target, + CatalogRole::Queue, + registry.module_code(QUEUE_MODULE).unwrap(), + 1, + ) + .unwrap(), + ) + .await + .unwrap(); + let authority = CellAuthority::new(layout.clone()); + let first_session = SessionId::from_bytes([4; 16]); + let observed = authority + .create_initial( + &proof, + incarnation, + Owner { + session: first_session, + endpoint: "https://first.internal:8081".into(), + }, + ) + .await + .unwrap(); + let directory = tempfile::TempDir::new().unwrap(); + let runtime = CellRuntime::new( + SqlWorkerPool::new(1, 10).unwrap(), + 16 * 1024 * 1024, + first_session, + ) + .unwrap(); + let handle = runtime + .bootstrap( + proof.clone(), + replica.clone(), + authority.clone(), + observed, + directory.path().join("first.sqlite"), + install_queue_schema, + ) + .await + .unwrap(); + let queue = QueueNamespace::::new( + CellClient::local(registry.clone(), handle.clone()), + target.tenant(), + target.application(), + ) + .unwrap(); + QueueRuntimeFixture { + registry, + target, + incarnation, + session: first_session, + directory, + runtime, + handle, + queue, + replica, + authority, + } +} + +async fn source_identity(fixture: &QueueRuntimeFixture) -> (u64, u64) { + use cellule_runtime::cell::actor::CellInventoryEntry; + tokio::time::timeout(std::time::Duration::from_secs(3), async { + loop { + let page = fixture.runtime.fleet_cells_page(None, 128).await.unwrap(); + if let CellInventoryEntry::Owned(owner) = &page.entries()[0] + && let Some(position) = &owner.position + { + return (owner.generation, position.epoch); + } + drop(page); + tokio::task::yield_now().await; + } + }) + .await + .unwrap() +} + +async fn wait_quiescing(fixture: &QueueRuntimeFixture) { + use cellule_runtime::cell::actor::CellInventoryEntry; + tokio::time::timeout(std::time::Duration::from_secs(3), async { + loop { + let page = fixture.runtime.fleet_cells_page(None, 128).await.unwrap(); + if let CellInventoryEntry::Owned(owner) = &page.entries()[0] + && owner.quiescing + { + return; + } + drop(page); + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); +} + +async fn maintenance_queue_movement(drop_waiter: bool) { + use cellule_runtime::cell::actor::MaintenanceCellRelease; + use cellule_runtime::cell::executor::Resolution; + use cellule_runtime::control::ControlState; + use cellule_runtime::primitives::queue::QueueSendCommand; + let fixture = queue_fixture().await; + let client = CellClient::local(fixture.registry.clone(), fixture.handle.clone()); + let prepared = client + .prepare_command::>( + &fixture.target, + mutation_identity(30), + QueueSendRequest { + producer_id: [30; 16], + payload: b"leased".to_vec(), + available_at_ms: now_ms(), + }, + ) + .await + .unwrap(); + let evidence = prepared.evidence().clone(); + let sent = prepared.execute().await.unwrap(); + let claimed = fixture + .queue + .claim( + mutation_identity(31), + 0, + QueueClaimRequest { + limit: 1, + lease_ms: 30_000, + }, + ) + .await + .unwrap(); + assert_eq!(claimed.output.len(), 1); + assert_eq!(claimed.output[0].payload, b"leased"); + fixture + .queue + .send( + mutation_identity(32), + QueueSendRequest { + producer_id: [32; 16], + payload: b"unclaimed-survivor".to_vec(), + available_at_ms: now_ms(), + }, + ) + .await + .unwrap(); + let (generation, epoch) = source_identity(&fixture).await; + let moving = { + let runtime = fixture.runtime.clone(); + let cell = fixture.target.cell_id(); + let source = fixture.session; + let incarnation = fixture.incarnation; + tokio::spawn(async move { + runtime + .release_maintenance_cell_at( + cell, + source, + generation, + incarnation, + epoch, + tokio::time::Instant::now() + std::time::Duration::from_secs(5), + ) + .await + }) + }; + wait_quiescing(&fixture).await; + assert!(!moving.is_finished()); + assert_eq!( + fixture + .authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .state, + ControlState::Serving + ); + assert!( + fixture + .queue + .validate_claim(0, claimed.output.clone(), Some(claimed.receipt)) + .await + .unwrap() + .output + ); + if drop_waiter { + moving.abort(); + } + fixture + .queue + .ack( + mutation_identity(33), + 0, + claimed.output[0].message_id, + claimed.output[0].token, + ) + .await + .unwrap(); + let release = if drop_waiter { + assert!(moving.await.unwrap_err().is_cancelled()); + None + } else { + match moving.await.unwrap().unwrap() { + MaintenanceCellRelease::Released(position) => Some(position), + refusal => panic!("unexpected refusal: {refusal:?}"), + } + }; + let idle = tokio::time::timeout(std::time::Duration::from_secs(5), async { + loop { + let observed = fixture + .authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + if observed.value().state == ControlState::Idle + && fixture.runtime.unreleased_cell_count().await.unwrap() == 0 + { + return observed; + } + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + if let Some(release) = release { + assert_eq!(idle.value().root.as_ref(), Some(&release.root)); + assert_eq!(release.epoch, epoch); + assert!(release.root.commit_sequence >= sent.receipt.commit_sequence); + } + assert!( + fixture + .handle + .query(1, 1, |_| panic!("released source accepted SQL")) + .await + .is_err() + ); + let receiver = SessionId::from_bytes([40; 16]); + let destination = + CellRuntime::new(SqlWorkerPool::new(1, 10).unwrap(), 16 << 20, receiver).unwrap(); + let restored = destination + .acquire_idle_restored( + fixture.handle.catalog().clone(), + fixture.replica.clone(), + fixture.authority.clone(), + idle, + fixture.directory.path().join("receiver.sqlite"), + Owner { + session: receiver, + endpoint: "https://receiver.internal:8081".into(), + }, + ) + .await + .unwrap(); + let client = CellClient::local(fixture.registry.clone(), restored.clone()); + assert!( + matches!(client.resolve(&evidence).await.unwrap(), Resolution::Committed(outcome) if outcome.commit_sequence() == sent.receipt.commit_sequence) + ); + let queue = QueueNamespace::::new( + client, + fixture.target.tenant(), + fixture.target.application(), + ) + .unwrap(); + let survivor = queue + .claim( + mutation_identity(34), + 0, + QueueClaimRequest { + limit: 10, + lease_ms: 30_000, + }, + ) + .await + .unwrap(); + assert_eq!(survivor.output.len(), 1); + assert_eq!(survivor.output[0].payload, b"unclaimed-survivor"); + queue + .ack( + mutation_identity(35), + 0, + survivor.output[0].message_id, + survivor.output[0].token, + ) + .await + .unwrap(); + restored.drain().await.unwrap(); + destination.shutdown().await.unwrap(); + fixture.runtime.shutdown().await.unwrap(); + assert_eq!(destination.stats().retained_bytes(), 0); + assert_eq!(fixture.runtime.stats().retained_bytes(), 0); +} + +#[tokio::test] +async fn busy_maintenance_releases_exact_queue_root_and_resumes_unclaimed_work() { + maintenance_queue_movement(false).await; +} + +#[tokio::test] +async fn dropped_busy_maintenance_waiter_keeps_release_owned_and_joined() { + maintenance_queue_movement(true).await; +} + +#[tokio::test] +async fn maintenance_deadline_refuses_before_release_and_keeps_native_completion_available() { + use cellule_runtime::cell::actor::MaintenanceCellRelease; + use cellule_runtime::control::ControlState; + use cellule_runtime::fleet::operations::DrainBlocker; + let fixture = queue_fixture().await; + fixture + .queue + .send( + mutation_identity(50), + QueueSendRequest { + producer_id: [50; 16], + payload: vec![1], + available_at_ms: now_ms(), + }, + ) + .await + .unwrap(); + let claimed = fixture + .queue + .claim( + mutation_identity(51), + 0, + QueueClaimRequest { + limit: 1, + lease_ms: 30_000, + }, + ) + .await + .unwrap(); + let (generation, epoch) = source_identity(&fixture).await; + let before = fixture + .authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .root + .clone(); + let result = fixture + .runtime + .release_maintenance_cell_at( + fixture.target.cell_id(), + fixture.session, + generation, + fixture.incarnation, + epoch, + tokio::time::Instant::now() + std::time::Duration::from_millis(150), + ) + .await + .unwrap(); + assert!(matches!( + result, + MaintenanceCellRelease::Refused { + blocker: DrainBlocker::Deadline, + error: None + } + )); + let current = fixture + .authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(current.value().state, ControlState::Serving); + assert_eq!(current.value().root, before); + assert!(matches!( + fixture.queue.info(0, None).await, + Err(InvocationError::NotStarted( + cellule_runtime::Error::CellDraining + )) + )); + assert!( + fixture + .queue + .validate_claim(0, claimed.output.clone(), Some(claimed.receipt)) + .await + .unwrap() + .output + ); + fixture + .queue + .ack( + mutation_identity(52), + 0, + claimed.output[0].message_id, + claimed.output[0].token, + ) + .await + .unwrap(); + let result = fixture + .runtime + .release_maintenance_cell_at( + fixture.target.cell_id(), + fixture.session, + generation, + fixture.incarnation, + epoch, + tokio::time::Instant::now() + std::time::Duration::from_secs(5), + ) + .await + .unwrap(); + assert!(matches!(result, MaintenanceCellRelease::Released(_))); + fixture.runtime.shutdown().await.unwrap(); + assert_eq!(fixture.runtime.stats().retained_bytes(), 0); +} diff --git a/crates/cellule-runtime/tests/primitives/workflow_api.rs b/crates/cellule-runtime/tests/primitives/workflow_api.rs index 6cf96882..14c79c29 100644 --- a/crates/cellule-runtime/tests/primitives/workflow_api.rs +++ b/crates/cellule-runtime/tests/primitives/workflow_api.rs @@ -506,5 +506,6 @@ fn registry_rejects_workflow_effect_target_drift() { } mod activity; +mod maintenance; mod namespace; mod retry; diff --git a/crates/cellule-runtime/tests/primitives/workflow_api/maintenance.rs b/crates/cellule-runtime/tests/primitives/workflow_api/maintenance.rs new file mode 100644 index 00000000..83384fcb --- /dev/null +++ b/crates/cellule-runtime/tests/primitives/workflow_api/maintenance.rs @@ -0,0 +1,529 @@ +//! Exact Activity completion, lease expiry, timers and external waits across a move. + +use super::*; +use cellule_runtime::cell::actor::{CellHandle, CellInventoryEntry, MaintenanceCellRelease}; +use cellule_runtime::control::ControlState; +use cellule_runtime::primitives::workflow::{ + ActivityClaim, ActivityCompletion, ActivityCompletionOutcome, ActivityLeaseOutcome, + WorkflowActivityClaimRequest, WorkflowActivityExtendRequest, WorkflowActivityValidateRequest, +}; + +struct Fixture { + registry: Arc, + target: CellTarget, + incarnation: IncarnationId, + session: SessionId, + directory: tempfile::TempDir, + replica: CellReplica, + authority: CellAuthority, + runtime: CellRuntime, + handle: CellHandle, +} + +async fn fixture() -> Fixture { + let registry = registry(); + let target = CellTarget::new( + TenantId::from_bytes([61; 16]), + ApplicationId::from_bytes([62; 16]), + WORKFLOW_NAMESPACE, + &0_u32.to_be_bytes(), + ) + .unwrap(); + let incarnation = IncarnationId::from_bytes([63; 16]); + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("workflow-maintenance"), + [62; 16], + ); + let replica = CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + Limits::default(), + ) + .unwrap(); + let proof = CellCatalog::new(layout.clone(), target.tenant()) + .provision( + CatalogEntry::new( + &target, + CatalogRole::Workflow, + registry.module_code(WORKFLOW_MODULE).unwrap(), + 1, + ) + .unwrap(), + ) + .await + .unwrap(); + let authority = CellAuthority::new(layout); + let session = SessionId::from_bytes([64; 16]); + let control = authority + .create_initial( + &proof, + incarnation, + Owner { + session, + endpoint: "https://source.internal:8081".into(), + }, + ) + .await + .unwrap(); + let directory = tempfile::TempDir::new().unwrap(); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 10).unwrap(), 16 << 20, session).unwrap(); + let handle = runtime + .bootstrap( + proof, + replica.clone(), + authority.clone(), + control, + directory.path().join("source.sqlite"), + install_workflow_schema, + ) + .await + .unwrap(); + Fixture { + registry, + target, + incarnation, + session, + directory, + replica, + authority, + runtime, + handle, + } +} + +fn completion(claim: &ActivityClaim) -> ActivityCompletion { + ActivityCompletion { + run_id: claim.run_id, + activity_id: claim.activity_id, + attempt: claim.attempt, + lease_token: claim.token, + completion_token: [80; 16], + result: b"settled".to_vec(), + failed: false, + retryable: false, + } +} + +async fn maintenance_workflow_movement(expire: bool) { + let fixture = fixture().await; + let client = CellClient::local(fixture.registry.clone(), fixture.handle.clone()); + let workflows = WorkflowNamespace::::new( + client.clone(), + fixture.target.tenant(), + fixture.target.application(), + ) + .unwrap(); + workflows + .start( + mutation_identity(65), + b"leased".to_vec(), + b"activity".to_vec(), + ) + .await + .unwrap(); + let claimed = client + .command::>( + &fixture.target, + mutation_identity(66), + WorkflowActivityClaimRequest { + limit: 1, + lease_ms: if expire { 5_000 } else { 30_000 }, + }, + ) + .await + .unwrap(); + assert_eq!(claimed.output.len(), 1); + let mut claim = claimed.output[0].clone(); + workflows + .start( + mutation_identity(67), + b"unclaimed".to_vec(), + b"activity-blocking".to_vec(), + ) + .await + .unwrap(); + let waiting = workflows + .start( + mutation_identity(68), + b"external-wait".to_vec(), + b"waiting".to_vec(), + ) + .await + .unwrap(); + let WorkflowOutcome::Applied { + run_id: waiting_run, + .. + } = waiting.output + else { + panic!("wait not applied") + }; + let timer = workflows + .start(mutation_identity(69), b"timer".to_vec(), b"timer".to_vec()) + .await + .unwrap(); + let page = fixture.runtime.fleet_cells_page(None, 128).await.unwrap(); + let CellInventoryEntry::Owned(owner) = &page.entries()[0] else { + panic!("missing owner") + }; + let generation = owner.generation; + let epoch = owner.position.as_ref().unwrap().epoch; + drop(page); + let mut moving = { + let runtime = fixture.runtime.clone(); + let cell = fixture.target.cell_id(); + let session = fixture.session; + let incarnation = fixture.incarnation; + tokio::spawn(async move { + runtime + .release_maintenance_cell_at( + cell, + session, + generation, + incarnation, + epoch, + tokio::time::Instant::now() + Duration::from_secs(10), + ) + .await + }) + }; + tokio::time::timeout(Duration::from_secs(3), async { + loop { + let page = fixture.runtime.fleet_cells_page(None, 128).await.unwrap(); + if let CellInventoryEntry::Owned(owner) = &page.entries()[0] + && owner.quiescing + { + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + assert!( + tokio::time::timeout(Duration::from_millis(150), &mut moving) + .await + .is_err() + ); + assert_eq!( + fixture + .authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .state, + ControlState::Serving + ); + assert!( + client + .query::>( + &fixture.target, + Some(claimed.receipt), + WorkflowActivityValidateRequest { + claimed: vec![claim.clone()] + } + ) + .await + .unwrap() + .output + ); + assert!(matches!( + client + .command::>( + &fixture.target, + mutation_identity(70), + WorkflowActivityClaimRequest { + limit: 1, + lease_ms: 5_000 + } + ) + .await, + Err(InvocationError::NotStarted(Error::CellDraining)) + )); + assert!(matches!( + fixture + .registry + .run_maintenance_once( + client.clone(), + fixture.target.clone(), + mutation_identity(71), + MaintenanceTickRequest { + expected_commit_sequence: timer.receipt.commit_sequence + } + ) + .await, + Err(InvocationError::NotStarted(Error::CellDraining)) + )); + assert!(matches!( + workflows.state(b"external-wait".to_vec(), None).await, + Err(InvocationError::NotStarted(Error::CellDraining)) + )); + let original_completion = completion(&claim); + if !expire { + let extended = client + .command::>( + &fixture.target, + mutation_identity(72), + WorkflowActivityExtendRequest { + claim: claim.clone(), + extension_ms: 30_000, + }, + ) + .await + .unwrap(); + let ActivityLeaseOutcome::Extended { lease_until_ms } = extended.output else { + panic!("extension lost") + }; + assert!(lease_until_ms > claim.lease_until_ms); + claim.lease_until_ms = lease_until_ms; + assert!( + client + .query::>( + &fixture.target, + Some(extended.receipt), + WorkflowActivityValidateRequest { + claimed: vec![claim.clone()] + } + ) + .await + .unwrap() + .output + ); + assert!(!moving.is_finished()); + let completed = client + .command::>( + &fixture.target, + mutation_identity(73), + original_completion.clone(), + ) + .await + .unwrap(); + assert!(matches!( + completed.output, + ActivityCompletionOutcome::Applied(_) + )); + } + let MaintenanceCellRelease::Released(position) = moving.await.unwrap().unwrap() else { + panic!("activity move refused") + }; + let idle = fixture + .authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(idle.value().state, ControlState::Idle); + assert_eq!(idle.value().root.as_ref(), Some(&position.root)); + let session = SessionId::from_bytes([81; 16]); + let destination = + CellRuntime::new(SqlWorkerPool::new(1, 10).unwrap(), 16 << 20, session).unwrap(); + let restored = destination + .acquire_idle_restored( + fixture.handle.catalog().clone(), + fixture.replica.clone(), + fixture.authority.clone(), + idle, + fixture.directory.path().join("receiver.sqlite"), + Owner { + session, + endpoint: "https://receiver.internal:8081".into(), + }, + ) + .await + .unwrap(); + let client = CellClient::local(fixture.registry.clone(), restored.clone()); + let workflows = WorkflowNamespace::::new( + client.clone(), + fixture.target.tenant(), + fixture.target.application(), + ) + .unwrap(); + let late = client + .command::>( + &fixture.target, + mutation_identity(74), + original_completion.clone(), + ) + .await; + // The receiver uses the registered native driver to resume the unclaimed + // activity; no raw row mutation or alternate scheduler performs the work. + let pool = BlockingActivityPool::new(1).unwrap(); + assert!(matches!( + fixture + .registry + .run_activity_once( + client.clone(), + &fixture.target, + 5_000, + pool.try_reserve().unwrap() + ) + .await + .unwrap(), + ActivityRunOutcome::Completed { .. } + )); + assert_eq!( + workflows + .state(b"unclaimed".to_vec(), None) + .await + .unwrap() + .output + .unwrap() + .status, + WorkflowStatus::Completed + ); + if expire { + assert!( + matches!(late, Err(InvocationError::Rejected(outcome)) if outcome.output == ActivityCompletionOutcome::LeaseLost) + ); + let reclaimed = client + .command::>( + &fixture.target, + mutation_identity(75), + WorkflowActivityClaimRequest { + limit: 1, + lease_ms: 30_000, + }, + ) + .await + .unwrap(); + assert_eq!(reclaimed.output.len(), 1); + let retry = &reclaimed.output[0]; + assert_eq!(retry.activity_id, claim.activity_id); + assert_eq!(retry.attempt, 2); + assert_ne!(retry.token, claim.token); + assert_eq!(retry.definition_digest, claim.definition_digest); + assert!( + matches!(client.command::>(&fixture.target, mutation_identity(76), original_completion).await, + Err(InvocationError::Rejected(outcome)) if outcome.output == ActivityCompletionOutcome::LeaseLost) + ); + assert!(matches!( + client + .command::>( + &fixture.target, + mutation_identity(77), + completion(retry) + ) + .await + .unwrap() + .output, + ActivityCompletionOutcome::Applied(_) + )); + } else { + assert!( + matches!(late.unwrap().output, ActivityCompletionOutcome::Duplicate { result } if result == b"settled") + ); + } + assert_eq!( + workflows + .state(b"leased".to_vec(), None) + .await + .unwrap() + .output + .unwrap() + .status, + WorkflowStatus::Completed + ); + let state = workflows + .state(b"external-wait".to_vec(), None) + .await + .unwrap() + .output + .unwrap(); + assert_eq!(state.run_id, waiting_run); + assert_eq!(state.state, b"waiting"); + assert_eq!(state.status, WorkflowStatus::Running); + let signal = WorkflowSignal { + workflow_id: b"external-wait".to_vec(), + run_id: waiting_run, + signal_id: [82; 16], + event: b"finish".to_vec(), + }; + workflows + .signal(mutation_identity(78), signal.clone()) + .await + .unwrap(); + assert!(matches!( + workflows + .signal(mutation_identity(79), signal) + .await + .unwrap() + .output, + WorkflowOutcome::Duplicate { .. } + )); + assert_eq!( + workflows + .state(b"external-wait".to_vec(), None) + .await + .unwrap() + .output + .unwrap() + .status, + WorkflowStatus::Completed + ); + assert_eq!( + workflows + .state(b"timer".to_vec(), None) + .await + .unwrap() + .output + .unwrap() + .state, + b"timer-waiting" + ); + let sequence = fixture + .authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .root + .as_ref() + .unwrap() + .commit_sequence; + let tick = fixture + .registry + .run_maintenance_once( + client, + fixture.target.clone(), + mutation_identity(80), + MaintenanceTickRequest { + expected_commit_sequence: sequence, + }, + ) + .await + .unwrap(); + assert_eq!( + tick.output, + MaintenanceTickOutcome::Applied { processed: 1 } + ); + assert_eq!( + workflows + .state(b"timer".to_vec(), Some(tick.receipt)) + .await + .unwrap() + .output + .unwrap() + .status, + WorkflowStatus::Completed + ); + pool.shutdown().await.unwrap(); + restored.drain().await.unwrap(); + destination.shutdown().await.unwrap(); + fixture.runtime.shutdown().await.unwrap(); + assert_eq!(destination.stats().retained_bytes(), 0); + assert_eq!(fixture.runtime.stats().retained_bytes(), 0); +} + +#[tokio::test] +async fn maintenance_preserves_activity_heartbeat_completion_workflow_wait_and_timer() { + maintenance_workflow_movement(false).await; +} + +#[tokio::test] +async fn maintenance_preserves_expired_activity_for_reclaim_and_fences_late_completion() { + maintenance_workflow_movement(true).await; +} diff --git a/crates/cellule-runtime/tests/protocol/client.rs b/crates/cellule-runtime/tests/protocol/client.rs index 1d770e86..541bf2d4 100644 --- a/crates/cellule-runtime/tests/protocol/client.rs +++ b/crates/cellule-runtime/tests/protocol/client.rs @@ -654,6 +654,7 @@ impl PeerRoundTrip for LoopbackRoundTrip { } } +mod effect_maintenance; mod effects; mod publication; mod routing; diff --git a/crates/cellule-runtime/tests/protocol/client/effect_maintenance.rs b/crates/cellule-runtime/tests/protocol/client/effect_maintenance.rs new file mode 100644 index 00000000..f2bbfa47 --- /dev/null +++ b/crates/cellule-runtime/tests/protocol/client/effect_maintenance.rs @@ -0,0 +1,387 @@ +//! Native Effect completion and exact inbox recovery during maintenance. + +use super::*; +use cellule_runtime::Error; +use cellule_runtime::cell::actor::{CellInventoryEntry, MaintenanceCellRelease}; +use cellule_runtime::control::ControlState; +use cellule_runtime::primitives::effects::EffectState; + +fn peer(fixture: &Fixture, handle: cellule_runtime::cell::actor::CellHandle) -> EffectPeerClient { + let signer = Arc::new(PeerSigner::new( + SessionId::from_bytes([42; 16]), + fixture.registry.release_digest(), + ed25519_dalek::SigningKey::from_bytes(&[43; 32]), + )); + EffectPeerClient::new( + signer.clone(), + PeerPrincipal { + issuer: "crab-runtime:test".into(), + subject: "source-session".into(), + actions: vec!["repository.issue.create".into()], + }, + Arc::new(LoopbackRoundTrip { + verifier: Arc::new(PeerVerifier::new( + SessionId::from_bytes([42; 16]), + fixture.registry.release_digest(), + signer.verifying_key(), + )), + dispatcher: Arc::new(PeerDispatcher::new( + fixture.registry.clone(), + Arc::new(LocalResolver { + target: fixture.target.clone(), + handle, + }), + Arc::new(RepositoryAuthorizer), + )), + }), + ) +} + +fn now_ms() -> i64 { + i64::try_from( + std::time::SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap() + .as_millis(), + ) + .unwrap() +} + +async fn maintenance_effect_movement(drop_waiter: bool, expire: bool) { + let fixture = fixture().await; + let runtime = fixture.runtime.as_ref().unwrap(); + let client = CellClient::local(fixture.registry.clone(), fixture.handle().clone()); + let mut encoder = BoundedEncoder::new(64).unwrap(); + b"leased-effect".to_vec().encode(&mut encoder).unwrap(); + let prepared = client + .prepare_command::( + &fixture.target, + mutation_identity(70), + encoder.finish(), + ) + .await + .unwrap(); + let evidence = prepared.evidence().clone(); + let emitted = prepared.execute().await.unwrap(); + let source = EffectSource::::new(client.clone(), fixture.target.clone()); + let claimed = source + .claim( + mutation_identity(71), + EffectClaimRequest { + limit: 1, + lease_ms: if expire { 5_000 } else { 30_000 }, + }, + ) + .await + .unwrap(); + assert_eq!(claimed.output.len(), 1); + let lease = claimed.output[0].clone(); + // A destination publication can precede a lost source acknowledgement. + let delivered = peer(&fixture, fixture.handle().clone()) + .deliver(&lease, now_ms()) + .await + .unwrap(); + let mut encoder = BoundedEncoder::new(64).unwrap(); + b"unclaimed-effect".to_vec().encode(&mut encoder).unwrap(); + client + .command::(&fixture.target, mutation_identity(72), encoder.finish()) + .await + .unwrap(); + let page = runtime.fleet_cells_page(None, 128).await.unwrap(); + let CellInventoryEntry::Owned(owner) = &page.entries()[0] else { + panic!("missing owner") + }; + let generation = owner.generation; + let epoch = owner.position.as_ref().unwrap().epoch; + drop(page); + let mut moving = { + let runtime = runtime.clone(); + let cell = fixture.target.cell_id(); + let session = fixture.session; + let incarnation = fixture.incarnation; + tokio::spawn(async move { + runtime + .release_maintenance_cell_at( + cell, + session, + generation, + incarnation, + epoch, + tokio::time::Instant::now() + Duration::from_secs(10), + ) + .await + }) + }; + tokio::time::timeout(Duration::from_secs(3), async { + loop { + let page = runtime.fleet_cells_page(None, 128).await.unwrap(); + if let CellInventoryEntry::Owned(owner) = &page.entries()[0] + && owner.quiescing + { + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + assert!( + tokio::time::timeout(Duration::from_millis(150), &mut moving) + .await + .is_err() + ); + assert_eq!( + fixture + .authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .state, + ControlState::Serving + ); + assert!( + source + .validate(vec![lease.clone()], claimed.receipt) + .await + .unwrap() + .output + ); + let mut wrong = lease.clone(); + wrong.token = [99; 16]; + assert!( + !source + .validate(vec![wrong], claimed.receipt) + .await + .unwrap() + .output + ); + assert!(matches!( + source + .claim( + mutation_identity(73), + EffectClaimRequest { + limit: 1, + lease_ms: 5_000 + } + ) + .await, + Err(InvocationError::NotStarted(Error::CellDraining)) + )); + assert!(matches!( + source.status(lease.effect_id, None).await, + Err(InvocationError::NotStarted(Error::CellDraining)) + )); + if drop_waiter { + moving.abort(); + } + if !expire { + assert_eq!( + source + .ack( + mutation_identity(74), + lease.clone(), + b"published destination".to_vec() + ) + .await + .unwrap() + .output, + EffectLeaseOutcome::Delivered + ); + } + let release = if drop_waiter { + assert!(moving.await.unwrap_err().is_cancelled()); + None + } else { + let MaintenanceCellRelease::Released(position) = moving.await.unwrap().unwrap() else { + panic!("effect release refused") + }; + Some(position) + }; + let idle = tokio::time::timeout(Duration::from_secs(10), async { + loop { + let observed = fixture + .authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + if observed.value().state == ControlState::Idle + && runtime.unreleased_cell_count().await.unwrap() == 0 + { + return observed; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); + if let Some(position) = release { + assert_eq!(idle.value().root.as_ref(), Some(&position.root)); + assert_eq!(position.epoch, epoch); + } + let session = SessionId::from_bytes([80; 16]); + let destination = + CellRuntime::new(SqlWorkerPool::new(1, 10).unwrap(), 16 << 20, session).unwrap(); + let restored = destination + .acquire_idle_restored( + fixture.proof.clone(), + fixture.replica.clone(), + fixture.authority.clone(), + idle, + fixture + ._directory + .path() + .join("maintenance-receiver.sqlite"), + Owner { + session, + endpoint: "https://receiver.internal:8081".into(), + }, + ) + .await + .unwrap(); + let client = CellClient::local(fixture.registry.clone(), restored.clone()); + assert!( + matches!(client.resolve(&evidence).await.unwrap(), Resolution::Committed(outcome) + if outcome.commit_sequence() == emitted.receipt.commit_sequence) + ); + let source = EffectSource::::new(client.clone(), fixture.target.clone()); + assert!( + !source + .validate(vec![lease.clone()], claimed.receipt) + .await + .unwrap() + .output + ); + assert!( + matches!(source.ack(mutation_identity(75), lease.clone(), b"late result".to_vec()).await, + Err(InvocationError::Rejected(outcome)) if outcome.output == EffectLeaseOutcome::LeaseLost) + ); + let peer = peer(&fixture, restored.clone()); + // The migrated inbox answers the identical destination receipt, even when + // an expired source lease will be reclaimed and delivered again. + assert_eq!( + peer.resolve(&lease, now_ms()).await.unwrap(), + Resolution::Committed(delivered.clone()) + ); + if expire { + let mut reclaimed = source + .claim( + mutation_identity(76), + EffectClaimRequest { + limit: 2, + lease_ms: 30_000, + }, + ) + .await + .unwrap(); + // Expiry uses the existing retry backoff. The already-ready survivor + // is claimable first; the old lease is retained as Ready, not erased. + assert_eq!(reclaimed.output.len(), 1); + assert_ne!(reclaimed.output[0].effect_id, lease.effect_id); + assert_eq!( + source + .status(lease.effect_id, Some(reclaimed.receipt)) + .await + .unwrap() + .output + .unwrap() + .state, + EffectState::Ready + ); + tokio::time::sleep(Duration::from_millis(250)).await; + let retried = source + .claim( + mutation_identity(79), + EffectClaimRequest { + limit: 1, + lease_ms: 30_000, + }, + ) + .await + .unwrap(); + assert_eq!(retried.output.len(), 1); + reclaimed.receipt = retried.receipt; + reclaimed.output.extend(retried.output); + assert_eq!(reclaimed.output.len(), 2); + let retry = reclaimed + .output + .iter() + .find(|claim| claim.effect_id == lease.effect_id) + .unwrap(); + assert_eq!(retry.attempt, 2); + assert_ne!(retry.token, lease.token); + assert_eq!(retry.operation_digest, lease.operation_digest); + assert!( + source + .validate(reclaimed.output.clone(), reclaimed.receipt) + .await + .unwrap() + .output + ); + for claim in reclaimed.output { + let outcome = peer.deliver(&claim, now_ms()).await.unwrap(); + if claim.effect_id == lease.effect_id { + assert_eq!(outcome, delivered); + } + source + .ack( + mutation_identity(if claim.effect_id == lease.effect_id { + 77 + } else { + 78 + }), + claim, + b"delivered".to_vec(), + ) + .await + .unwrap(); + } + } else { + let outcome = fixture + .registry + .run_effect_once(client.clone(), fixture.target.clone(), peer, 5_000) + .await + .unwrap(); + assert!(matches!(outcome, EffectRunOutcome::Delivered { .. })); + } + assert_eq!( + source + .status(lease.effect_id, None) + .await + .unwrap() + .output + .unwrap() + .state, + EffectState::Delivered + ); + assert_eq!( + client + .query::(&fixture.target, None, ()) + .await + .unwrap() + .output, + 2 + ); + restored.drain().await.unwrap(); + destination.shutdown().await.unwrap(); + runtime.shutdown().await.unwrap(); + assert_eq!(destination.stats().retained_bytes(), 0); + assert_eq!(runtime.stats().retained_bytes(), 0); +} + +#[tokio::test] +async fn maintenance_waits_for_effect_ack_and_resumes_native_delivery() { + maintenance_effect_movement(false, false).await; +} + +#[tokio::test] +async fn dropped_maintenance_waiter_preserves_effect_completion_and_inbox() { + maintenance_effect_movement(true, false).await; +} + +#[tokio::test] +async fn maintenance_preserves_expired_effect_for_normal_reclaim_and_inbox_dedup() { + maintenance_effect_movement(false, true).await; +} diff --git a/crates/cellule-runtime/tests/runtime/catalog.rs b/crates/cellule-runtime/tests/runtime/catalog.rs index 2f09da40..1cc9602e 100644 --- a/crates/cellule-runtime/tests/runtime/catalog.rs +++ b/crates/cellule-runtime/tests/runtime/catalog.rs @@ -545,3 +545,306 @@ fn nibble(byte: u8) -> u8 { _ => panic!("catalog digest must be lowercase hex"), } } + +#[tokio::test] +async fn complete_catalog_scan_covers_empty_heads_and_rechecks_them() { + let (layout, _, target) = fixture(); + let recorder = Arc::new(CatalogReadRecorder::default()); + let catalog = CellCatalog::with_telemetry( + layout, + target.tenant(), + cellule_runtime::fleet::telemetry::CellTelemetryHandle::from_sink(recorder.clone()), + ); + assert_eq!(catalog.tenant(), target.tenant()); + let mut scan = catalog.scan_all(1).await.unwrap(); + assert_eq!(recorder.reads.lock().unwrap().len(), 256); + assert!(scan.next_page().await.unwrap().is_none()); + let receipt = scan.finish().await.unwrap(); + assert_eq!(receipt.tenant(), target.tenant()); + assert_eq!(receipt.application(), target.application()); + assert_eq!(receipt.entry_count(), 0); + for shard in 0..=u8::MAX { + assert_eq!(receipt.revision(shard), 0); + assert!(receipt.page_digests(shard).is_empty()); + } + assert_eq!(recorder.reads.lock().unwrap().len(), 512); + assert!( + recorder + .reads + .lock() + .unwrap() + .iter() + .all(|read| *read == (CatalogReadKind::Head, true)) + ); + receipt.revalidate().await.unwrap(); +} + +#[tokio::test] +async fn complete_catalog_scan_streams_multiple_pages_and_shards_in_one_scope() { + let (layout, tenant, _, mut entries) = provisioned_shard(257).await; + let first_shard = entries[0].cell().as_bytes()[0]; + let application = ApplicationId::from_bytes(*layout.application_id()); + let target = (0_u32..) + .map(|partition| { + CellTarget::new( + tenant, + application, + NamespaceId::from_bytes([22; 16]), + &partition.to_be_bytes(), + ) + .unwrap() + }) + .find(|target| target.cell_id().as_bytes()[0] != first_shard) + .unwrap(); + let catalog = CellCatalog::new(layout.clone(), tenant); + let other = entry(&target, CatalogRole::Sql, 10); + catalog.provision(other.clone()).await.unwrap(); + entries.push(other); + entries.sort_unstable_by_key(|entry| *entry.cell().as_bytes()); + let foreign_tenant = TenantId::from_bytes([98; 16]); + let foreign = + CellTarget::new(foreign_tenant, application, target.namespace(), b"foreign").unwrap(); + CellCatalog::new(layout.clone(), foreign_tenant) + .provision(entry(&foreign, CatalogRole::Kv, 11)) + .await + .unwrap(); + // Independent adapter; no resident cache or caller-supplied shard subset. + let mut scan = CellCatalog::new(layout.clone(), tenant) + .scan_all(258) + .await + .unwrap(); + let mut actual = Vec::new(); + let mut page_count = 0; + while let Some(page) = scan.next_page().await.unwrap() { + assert!(page.entries().len() <= 256); + assert!( + page.entries() + .iter() + .all(|proof| proof.revision() == page.revision()) + ); + actual.extend(page.entries().iter().map(|proof| proof.entry().clone())); + page_count += 1; + } + assert_eq!(actual, entries); + assert_eq!(page_count, 3); + let receipt = scan.finish().await.unwrap(); + assert_eq!(receipt.entry_count(), 258); + assert_eq!(receipt.revision(first_shard), 257); + let head = head_json(&layout, tenant, first_shard).await; + let digests = head["pages"] + .as_array() + .unwrap() + .iter() + .map(|page| Digest::from_bytes(decode_digest(page["digest"].as_str().unwrap()))) + .collect::>(); + assert_eq!(receipt.page_digests(first_shard), digests); + assert_eq!(receipt.revision(target.cell_id().as_bytes()[0]), 1); +} + +#[tokio::test] +async fn complete_catalog_scan_refuses_partial_finish() { + let (_, catalog, target) = fixture(); + catalog + .provision(entry(&target, CatalogRole::Sql, 4)) + .await + .unwrap(); + assert!(matches!( + catalog.scan_all(1).await.unwrap().finish().await, + Err(cellule_runtime::Error::Catalog( + "complete catalog scan is not exhausted" + )) + )); + let mut scan = catalog.scan_all(1).await.unwrap(); + assert!(scan.next_page().await.unwrap().is_some()); + // Even after the last page, the consumer must observe end of traversal. + assert!(matches!( + scan.finish().await, + Err(cellule_runtime::Error::Catalog( + "complete catalog scan is not exhausted" + )) + )); +} + +#[tokio::test] +async fn complete_catalog_scan_limits_are_checked_before_io_and_failure_is_terminal() { + let (layout, _, first) = fixture(); + let recorder = Arc::new(CatalogReadRecorder::default()); + let catalog = CellCatalog::with_telemetry( + layout, + first.tenant(), + cellule_runtime::fleet::telemetry::CellTelemetryHandle::from_sink(recorder.clone()), + ); + for limit in [0, 16_777_217] { + assert!(matches!( + catalog.scan_all(limit).await, + Err(cellule_runtime::Error::Capacity(_)) + )); + } + assert!(recorder.reads.lock().unwrap().is_empty()); + catalog + .provision(entry(&first, CatalogRole::Sql, 4)) + .await + .unwrap(); + let second = (0_u32..) + .map(|partition| { + CellTarget::new( + first.tenant(), + first.application(), + first.namespace(), + &partition.to_be_bytes(), + ) + .unwrap() + }) + .find(|target| target.cell_id().as_bytes()[0] != first.cell_id().as_bytes()[0]) + .unwrap(); + catalog + .provision(entry(&second, CatalogRole::Sql, 4)) + .await + .unwrap(); + let mut scan = catalog.scan_all(1).await.unwrap(); + assert!(scan.next_page().await.unwrap().is_some()); + assert!(matches!( + scan.next_page().await, + Err(cellule_runtime::Error::Capacity(_)) + )); + assert!(matches!( + scan.next_page().await, + Err(cellule_runtime::Error::Catalog( + "complete catalog scan previously failed" + )) + )); + assert!(scan.finish().await.is_err()); +} + +#[tokio::test] +async fn complete_catalog_scan_refuses_changed_populated_and_absent_heads() { + let (_, catalog, first, second) = same_shard_targets(); + catalog + .provision(entry(&first, CatalogRole::Sql, 4)) + .await + .unwrap(); + let mut scan = catalog.scan_all(2).await.unwrap(); + catalog + .provision(entry(&second, CatalogRole::Sql, 4)) + .await + .unwrap(); + let page = scan.next_page().await.unwrap().unwrap(); + assert_eq!(page.entries().len(), 1); + assert_eq!(page.entries()[0].entry().cell(), first.cell_id()); + assert!(scan.next_page().await.unwrap().is_none()); + assert!(matches!( + scan.finish().await, + Err(cellule_runtime::Error::Catalog( + "complete catalog scan head changed" + )) + )); + let (_, empty, target) = fixture(); + let mut scan = empty.scan_all(1).await.unwrap(); + empty + .provision(entry(&target, CatalogRole::Sql, 4)) + .await + .unwrap(); + assert!(scan.next_page().await.unwrap().is_none()); + assert!(scan.finish().await.is_err()); +} + +#[tokio::test] +async fn complete_catalog_scan_refuses_head_deletion_and_same_body_rewrite() { + for delete in [false, true] { + let (layout, catalog, target) = fixture(); + catalog + .provision(entry(&target, CatalogRole::Sql, 4)) + .await + .unwrap(); + let mut scan = catalog.scan_all(1).await.unwrap(); + while scan.next_page().await.unwrap().is_some() {} + let path = + layout.catalog_head_path(target.tenant().as_bytes(), target.cell_id().as_bytes()[0]); + if delete { + layout.store().delete(&path).await.unwrap(); + } else { + let (body, token) = layout + .store() + .get_with_etag_bounded(&path, 64 * 1024) + .await + .unwrap(); + layout.store().update(&path, body, token).await.unwrap(); + } + assert!(matches!( + scan.finish().await, + Err(cellule_runtime::Error::Catalog( + "complete catalog scan head changed" + )) + )); + } +} + +#[tokio::test] +async fn complete_catalog_scan_preserves_missing_page_error_and_cannot_skip_it() { + let (layout, catalog, target) = fixture(); + catalog + .provision(entry(&target, CatalogRole::Sql, 4)) + .await + .unwrap(); + let mut scan = catalog.scan_all(1).await.unwrap(); + let head = head_json(&layout, target.tenant(), target.cell_id().as_bytes()[0]).await; + let digest = decode_digest(head["pages"][0]["digest"].as_str().unwrap()); + let path = layout.catalog_object_path(&digest); + layout.store().delete(&path).await.unwrap(); + assert!(matches!(scan.next_page().await, + Err(cellule_runtime::Error::Storage(cellule_store::StorageError::NotFound { path: missing })) if missing == path.to_string())); + assert!(scan.next_page().await.is_err()); + assert!(scan.finish().await.is_err()); +} + +#[tokio::test] +async fn complete_catalog_scan_reuses_digest_validation_and_receipt_rechecks() { + let (layout, catalog, target) = fixture(); + catalog + .provision(entry(&target, CatalogRole::Sql, 4)) + .await + .unwrap(); + let mut scan = catalog.scan_all(1).await.unwrap(); + while scan.next_page().await.unwrap().is_some() {} + let receipt = scan.finish().await.unwrap(); + catalog + .provision(entry(&target, CatalogRole::Sql, 4)) + .await + .unwrap(); + receipt.revalidate().await.unwrap(); // Idempotent provisioning did not write a head. + let head = head_json(&layout, target.tenant(), target.cell_id().as_bytes()[0]).await; + let digest = decode_digest(head["pages"][0]["digest"].as_str().unwrap()); + let mut scan = catalog.scan_all(1).await.unwrap(); + layout + .store() + .put_overwrite( + &layout.catalog_object_path(&digest), + Bytes::from_static(b"{}"), + ) + .await + .unwrap(); + assert!(matches!( + scan.next_page().await, + Err(cellule_runtime::Error::Catalog("page digest mismatch")) + )); + assert!(scan.finish().await.is_err()); + // A head receipt verifies head identity only, not continuing page availability. + receipt.revalidate().await.unwrap(); + let fresh = (0_u32..) + .map(|partition| { + CellTarget::new( + target.tenant(), + target.application(), + target.namespace(), + &partition.to_be_bytes(), + ) + .unwrap() + }) + .find(|candidate| candidate.cell_id().as_bytes()[0] != target.cell_id().as_bytes()[0]) + .unwrap(); + catalog + .provision(entry(&fresh, CatalogRole::Sql, 4)) + .await + .unwrap(); + assert!(receipt.revalidate().await.is_err()); +} diff --git a/crates/cellule-runtime/tests/runtime/lifecycle.rs b/crates/cellule-runtime/tests/runtime/lifecycle.rs index 41810c98..65b60c56 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle.rs @@ -52,8 +52,11 @@ use crate::support::fixtures::{mutation_identity_window, now_ms}; pub mod durability; pub mod execution; pub mod idle; +pub mod inventory; +pub mod maintenance; pub mod ownership; pub mod read_replica; +pub mod receiver; pub mod residency; #[derive(Debug)] @@ -64,6 +67,7 @@ pub(super) struct PausingStore { failing: AtomicBool, transient_put_failures: AtomicUsize, lost_update_response: AtomicBool, + acquisition_fault: AtomicUsize, failed: AtomicBool, blocked: AtomicBool, released: AtomicBool, @@ -90,6 +94,7 @@ impl PausingStore { failing: AtomicBool::new(false), transient_put_failures: AtomicUsize::new(0), lost_update_response: AtomicBool::new(false), + acquisition_fault: AtomicUsize::new(0), failed: AtomicBool::new(false), blocked: AtomicBool::new(false), released: AtomicBool::new(false), @@ -191,6 +196,20 @@ impl ObjectStore for PausingStore { payload: PutPayload, options: PutOptions, ) -> object_store::Result { + let acquisition_fault = if location.as_ref().contains("/acquisitions/") { + self.acquisition_fault.swap(0, Ordering::AcqRel) + } else { + 0 + }; + if acquisition_fault == 1 { + return Err(object_store::Error::PermissionDenied { + path: location.to_string(), + source: Box::new(std::io::Error::new( + std::io::ErrorKind::PermissionDenied, + "injected acquisition metadata failure", + )), + }); + } if self .transient_put_failures .try_update(Ordering::AcqRel, Ordering::Acquire, |remaining| { @@ -230,6 +249,15 @@ impl ObjectStore for PausingStore { } } let result = self.inner.put_opts(location, payload, options).await?; + if acquisition_fault == 2 { + return Err(object_store::Error::PermissionDenied { + path: location.to_string(), + source: Box::new(std::io::Error::new( + std::io::ErrorKind::PermissionDenied, + "acquisition metadata response lost after commit", + )), + }); + } if update && self.lost_update_response.swap(false, Ordering::AcqRel) { return Err(object_store::Error::Generic { store: "pausing-store", @@ -355,8 +383,9 @@ impl NodeLogAuthority for TestNodeAuthority { fn close<'a>( &'a self, - barrier: &'a NodeLogRotationBarrier, + retirement: &'a cellule_runtime::node::log::NodeLogRetirementObservation, ) -> futures_util::future::BoxFuture<'a, cellule_runtime::Result<()>> { + let barrier = retirement.barrier(); Box::pin(async move { self.closes.lock().unwrap().push(barrier.log_epoch()); Ok(()) @@ -534,7 +563,7 @@ pub(super) async fn fence_log_session( .unwrap(); directory .create( - signed(member, "https://follower.internal:8081", 10_000, 20_000), + signed(member, "https://follower.internal:8081", 1, 20_000), 2, ) .await @@ -550,7 +579,7 @@ pub(super) async fn fence_log_session( if claimant != member { directory .create( - signed(claimant, "https://claimant.internal:8081", 10_000, 20_000), + signed(claimant, "https://claimant.internal:8081", 1, 20_000), 3, ) .await diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/durability.rs b/crates/cellule-runtime/tests/runtime/lifecycle/durability.rs index d463533e..d2115a9a 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/durability.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/durability.rs @@ -6,6 +6,7 @@ use cellule_runtime::fleet::telemetry::{CellTelemetry, CommandResponseSource, Pu mod admission; mod proofs; mod recovery; +mod retirement; #[derive(Default)] pub(super) struct RecordingResponses( diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/durability/proofs.rs b/crates/cellule-runtime/tests/runtime/lifecycle/durability/proofs.rs index 1799b90d..544f21e7 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/durability/proofs.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/durability/proofs.rs @@ -290,7 +290,7 @@ async fn runtime_replaces_node_durability_binding_for_a_new_epoch() { let lease = NodeLeaseGuard::new(0, 60_000).unwrap(); runtime.install_node_lease(lease.clone()).unwrap(); let transport: Arc = Arc::new(TestNodeTransport(None)); - let make_durability = |log_epoch| { + let make_binding = |session, leader, log_epoch| { let gate = DurabilityGate::new(session, leader, log_epoch, [follower]).unwrap(); let shipper = NodeLogShipper::new(gate.clone(), Arc::clone(&transport), Limits::default()).unwrap(); @@ -303,6 +303,16 @@ async fn runtime_replaces_node_durability_binding_for_a_new_epoch() { lease.clone(), )) }; + let make_durability = |epoch| make_binding(session, leader, epoch); + let foreign_boot = make_binding(SessionId::from_bytes([63; 16]), leader, 1); + assert!(matches!( + runtime.install_node_durability(fixture.target.application(), foreign_boot.clone()), + Err(cellule_runtime::Error::Control( + "Cell runtime node durability boot differs" + )) + )); + assert!(runtime.node_durability().is_none()); + foreign_boot.shutdown().await.unwrap(); let first = make_durability(1); runtime .install_node_durability(fixture.target.application(), Arc::clone(&first)) @@ -310,13 +320,46 @@ async fn runtime_replaces_node_durability_binding_for_a_new_epoch() { let second = make_durability(2); let previous = runtime - .replace_node_durability(fixture.target.application(), Arc::clone(&second)) + .replace_node_durability(fixture.target.application(), &first, Arc::clone(&second)) .unwrap(); assert!(Arc::ptr_eq(&previous, &first)); let (application, current) = runtime.node_durability().unwrap(); assert_eq!(application, fixture.target.application()); assert!(Arc::ptr_eq(¤t, &second)); + let rejected = make_durability(3); + assert!(matches!( + runtime.replace_node_durability( + fixture.target.application(), + &first, + Arc::clone(&rejected) + ), + Err(cellule_runtime::Error::Control( + "Cell runtime node durability binding changed" + )) + )); + assert!(Arc::ptr_eq(&runtime.node_durability().unwrap().1, &second)); + rejected.shutdown().await.unwrap(); + for (candidate, refusal) in [ + ( + make_binding(SessionId::from_bytes([63; 16]), leader, 3), + "Cell runtime node durability identity changed", + ), + ( + make_binding(session, NodeId::from_bytes([64; 16]), 3), + "Cell runtime node durability identity changed", + ), + ( + make_durability(2), + "Cell runtime node durability epoch did not advance", + ), + ] { + assert!( + matches!(runtime.replace_node_durability(fixture.target.application(), &second, candidate.clone()), Err(cellule_runtime::Error::Control(message)) if message == refusal) + ); + assert!(Arc::ptr_eq(&runtime.node_durability().unwrap().1, &second)); + candidate.shutdown().await.unwrap(); + } runtime.shutdown().await.unwrap(); } #[tokio::test(flavor = "multi_thread")] diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/durability/recovery.rs b/crates/cellule-runtime/tests/runtime/lifecycle/durability/recovery.rs index 4974e795..5ae2bf3e 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/durability/recovery.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/durability/recovery.rs @@ -221,6 +221,42 @@ async fn lost_ack_suffix_recovers_an_ambiguous_command_without_reexecution() { .recover_and_seal(&directory, fenced, inventory, 10_002) .await .unwrap(); + // Retire the failed owner's actual suffix only after its overlay is pinned. + // The unchanged runtime readback below must still recover the lost outcome. + let retirement_transport = Arc::new( + cellule_runtime::node::log_transport::LocalRecoveredFollowerTransport::new( + cellule_runtime::node::log_transport::LocalFollowerTransport::new( + member, + follower_store.clone(), + ), + directory.clone(), + successor, + || Ok(10_003), + ) + .unwrap(), + ); + let retirement = cellule_runtime::node::log_recovery::retirement::retire_recovered_members( + retirement_transport, + &completed.sealed, + ) + .await + .unwrap() + .confirmed() + .unwrap(); + assert_eq!(retirement.members()[0].result().unwrap().durable_through, 2); + // Commit the tombstone transition but lose its reply. Fresh canonical + // inspection must adopt it without requiring a second native retirement. + store.lose_next_update_response(); + let retired = directory + .retire_recovered_log(&retirement, successor, 10_003) + .await + .unwrap(); + assert_eq!( + retired.log().phase(), + cellule_runtime::node::log_state::NodeLogPhase::Retired + ); + assert!(!directory.log_epoch_referenced(leader, 1).await.unwrap()); + assert_eq!(follower_store.retained_bytes(), 8); let attached = completed.controls.into_iter().next().unwrap(); let successor_runtime = CellRuntime::new( SqlWorkerPool::new(1, 1).unwrap(), diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/durability/retirement.rs b/crates/cellule-runtime/tests/runtime/lifecycle/durability/retirement.rs new file mode 100644 index 00000000..a1ba6126 --- /dev/null +++ b/crates/cellule-runtime/tests/runtime/lifecycle/durability/retirement.rs @@ -0,0 +1,797 @@ +//! Maintenance evidence from actual native follower fences and Cell publication. +use super::*; +use cellule_runtime::node::log_transport::LocalFollowerTransport; +use cellule_runtime::node::{NodeDirectory, VersionedNodeAdvertisement}; +use futures_util::future::BoxFuture; + +struct RetirementTransport { + locals: Vec<(NodeId, LocalFollowerTransport)>, + requests: Mutex>, + frames: Mutex>, + lose_reply: AtomicBool, + contradict_reply: AtomicBool, + pause: AtomicBool, + entered: tokio::sync::Semaphore, + release: tokio::sync::Semaphore, + directory: Option, +} + +impl RetirementTransport { + fn local(&self, member: NodeId) -> &LocalFollowerTransport { + &self.locals.iter().find(|(id, _)| *id == member).unwrap().1 + } +} + +impl NodeLogTransport for RetirementTransport { + fn append<'a>( + &'a self, + member: NodeId, + request: AppendRequest, + ) -> BoxFuture<'a, cellule_runtime::Result> { + self.frames.lock().unwrap().extend(request.frames.clone()); + Box::pin(async move { + if let Some(directory) = &self.directory { + directory + .authorize_log_append( + request.leader_session, + member, + request.log_epoch, + request.covered_through, + now_ms(), + ) + .await?; + } + self.local(member).append(member, request).await + }) + } + fn seal<'a>( + &'a self, + member: NodeId, + request: SealRequest, + ) -> BoxFuture<'a, cellule_runtime::Result> { + self.local(member).seal(member, request) + } + fn tail<'a>( + &'a self, + member: NodeId, + request: TailRequest, + ) -> BoxFuture<'a, cellule_runtime::Result>> { + self.local(member).tail(member, request) + } + fn retire<'a>( + &'a self, + member: NodeId, + request: RetireRequest, + ) -> BoxFuture<'a, cellule_runtime::Result> { + Box::pin(async move { + self.requests.lock().unwrap().push((member, request)); + if let Some(directory) = &self.directory { + directory + .authorize_log_retire( + request.leader_session, + member, + request.log_epoch, + request.covered_through, + now_ms(), + ) + .await?; + } + let mut receipt = self.local(member).retire(member, request).await?; + if member == self.locals[0].0 { + if self.lose_reply.swap(false, Ordering::AcqRel) { + return Err(cellule_runtime::Error::Node("lost native retirement reply")); + } + if self.contradict_reply.swap(false, Ordering::AcqRel) { + receipt.durable_through += 1; + } + } else if self.pause.swap(false, Ordering::AcqRel) { + self.entered.add_permits(1); + self.release.acquire().await.unwrap().forget(); + } + Ok(receipt) + }) + } +} + +// The fixture owns the exact successful close observation. A retry reconciles +// that observation rather than treating an arbitrary absent log as its result. +struct SignedRetirementAuthority { + directory: NodeDirectory, + observed: tokio::sync::Mutex, + closed_barrier: Mutex>, + attempts: AtomicUsize, + lose_reply: AtomicBool, + pause_reply: AtomicBool, + entered: tokio::sync::Semaphore, + release: tokio::sync::Semaphore, +} + +impl NodeLogAuthority for SignedRetirementAuthority { + fn activate<'a>(&'a self, epoch: u64) -> BoxFuture<'a, cellule_runtime::Result<()>> { + Box::pin(async move { + let mut observed = self.observed.lock().await; + if observed + .advertisement() + .log() + .is_none_or(|log| log.epoch() != epoch) + { + return Err(cellule_runtime::Error::Fenced); + } + *observed = self.directory.activate_log(&observed, now_ms()).await?; + Ok(()) + }) + } + fn advance_coverage<'a>( + &'a self, + epoch: u64, + through: u64, + ) -> BoxFuture<'a, cellule_runtime::Result<()>> { + Box::pin(async move { + let mut observed = self.observed.lock().await; + if observed + .advertisement() + .log() + .is_none_or(|log| log.epoch() != epoch) + { + return Err(cellule_runtime::Error::Fenced); + } + *observed = self + .directory + .advance_log_coverage(&observed, through, now_ms()) + .await?; + Ok(()) + }) + } + fn close<'a>( + &'a self, + retirement: &'a cellule_runtime::node::log::NodeLogRetirementObservation, + ) -> BoxFuture<'a, cellule_runtime::Result<()>> { + let barrier = retirement.barrier(); + Box::pin(async move { + self.attempts.fetch_add(1, Ordering::AcqRel); + let mut observed = self.observed.lock().await; + if let Some(closed) = self.closed_barrier.lock().unwrap().as_ref() { + assert_eq!(closed, barrier); + assert!(observed.advertisement().log().is_none()); + return Ok(()); + } + *observed = self + .directory + .close_log(&observed, barrier, now_ms()) + .await?; + *self.closed_barrier.lock().unwrap() = Some(barrier.clone()); + // Both faults happen after the canonical CAS, not before dispatch. + if self.pause_reply.swap(false, Ordering::AcqRel) { + self.entered.add_permits(1); + self.release.acquire().await.unwrap().forget(); + } + if self.lose_reply.swap(false, Ordering::AcqRel) { + return Err(cellule_runtime::Error::Facility { + name: "signed-close-reply", + source: Box::new(std::io::Error::new( + std::io::ErrorKind::ConnectionReset, + "canonical close accepted but reply lost", + )), + }); + } + Ok(()) + }) + } +} + +struct RetirementFixture { + fixture: Fixture, + runtime: CellRuntime, + handle: cellule_runtime::cell::actor::CellHandle, + durability: Arc, + authority: Arc, + transport: Arc, + directories: Vec, +} + +impl RetirementFixture { + async fn new(key: &[u8], store: Store) -> Self { + Self::new_protocol(key, store, false).await.0 + } + + async fn new_protocol( + key: &[u8], + store: Store, + authenticated: bool, + ) -> (Self, Option>) { + let fixture = fixture_with_limits_and_store(key, Limits::default(), store); + let session = SessionId::from_bytes([191; 16]); + let leader = NodeId::from_bytes([192; 16]); + let members = [NodeId::from_bytes([193; 16]), NodeId::from_bytes([194; 16])]; + let signed_authority = if authenticated { + let fleet = Digest::from_bytes([181; 32]); + let image = Digest::from_bytes([182; 32]); + let release = Digest::from_bytes([183; 32]); + let directory = NodeDirectory::new(fixture.layout.clone(), fleet, image, release); + let key = ed25519_dalek::SigningKey::from_bytes(&[184; 32]); + let now = now_ms(); + let signed = |node, session| { + cellule_runtime::node::NodeAdvertisement::sign( + node, + session, + "https://retirement.internal:8789".into(), + fleet, + Digest::from_bytes([185; 32]), + image, + release, + &key, + 1, + now, + now + 20_000, + vec![Digest::from_bytes([186; 32])], + vec![1], + cellule_runtime::node::NodeFailureDomain::default(), + cellule_runtime::node::NodeCapacity { + free_memory_bytes: 1, + free_disk_bytes: 1, + follower_free_bytes: 1, + follower_retained_bytes: 0, + job_credits: 1, + log_protocol: cellule_runtime::node::NODE_LOG_PROTOCOL_VERSION, + }, + ) + .unwrap() + }; + let created = directory + .create(signed(leader, session), now) + .await + .unwrap(); + for (index, member) in members.iter().enumerate() { + directory + .create( + signed(*member, SessionId::from_bytes([197 + index as u8; 16])), + now, + ) + .await + .unwrap(); + } + let enrolled = directory.recruit_log(&created, 7, 1, 3, now).await.unwrap(); + assert_eq!(enrolled.advertisement().log().unwrap().members(), members); + Some(Arc::new(SignedRetirementAuthority { + directory, + observed: tokio::sync::Mutex::new(enrolled), + closed_barrier: Mutex::new(None), + attempts: AtomicUsize::new(0), + lose_reply: AtomicBool::new(false), + pause_reply: AtomicBool::new(false), + entered: tokio::sync::Semaphore::new(0), + release: tokio::sync::Semaphore::new(0), + })) + } else { + None + }; + let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( + SqlWorkerPool::new(1, 1).unwrap(), + 2 * 1024 * 1024, + session, + ReplicaHost::default().with_local_disk_budget(DiskBudget::new(1 << 30)), + ) + .unwrap(); + let lease = NodeLeaseGuard::new(0, 60_000).unwrap(); + runtime.install_node_lease(lease.clone()).unwrap(); + let directories: Vec<_> = members + .iter() + .map(|_| tempfile::tempdir().unwrap()) + .collect(); + let locals = members + .iter() + .zip(&directories) + .map(|(member, directory)| { + let store = cellule_runtime::FollowerStore::open( + directory.path().to_owned(), + Limits::default(), + DiskBudget::new(1 << 30), + ) + .unwrap(); + (*member, LocalFollowerTransport::new(*member, store)) + }) + .collect(); + let transport = Arc::new(RetirementTransport { + locals, + requests: Mutex::new(Vec::new()), + frames: Mutex::new(Vec::new()), + lose_reply: AtomicBool::new(false), + contradict_reply: AtomicBool::new(false), + pause: AtomicBool::new(false), + entered: tokio::sync::Semaphore::new(0), + release: tokio::sync::Semaphore::new(0), + directory: signed_authority + .as_ref() + .map(|authority| authority.directory.clone()), + }); + let gate = DurabilityGate::new(session, leader, 7, members).unwrap(); + let node_transport: Arc = transport.clone(); + let shipper = + NodeLogShipper::new(gate.clone(), node_transport.clone(), Limits::default()).unwrap(); + let authority = Arc::new(TestNodeAuthority::default()); + let log_authority: Arc = match &signed_authority { + Some(authority) => authority.clone(), + None => authority.clone(), + }; + let durability = Arc::new(NodeDurability::new( + gate, + shipper, + log_authority, + node_transport, + lease, + )); + runtime + .install_node_durability(fixture.target.application(), durability.clone()) + .unwrap(); + let handle = bootstrap_on(&runtime, &fixture, session).await; + ( + Self { + fixture, + runtime, + handle, + durability, + authority, + transport, + directories, + }, + signed_authority, + ) + } + + async fn command(&self) { + let outcome = self + .handle + .execute( + mutation_identity_window(195, 10, 10_000), + Digest::from_bytes([196; 32]), + 20, + 1_024, + 1_024, + |transaction| { + transaction.execute("UPDATE counter SET value = value + 1", [])?; + Ok(HandlerOutcome::Success(b"durable maintenance".to_vec())) + }, + ) + .await + .unwrap(); + assert!( + matches!(outcome, StoredOutcome::Success { result, commit_sequence: 1 } if result == b"durable maintenance") + ); + } + + async fn assert_native_fences(&self) { + let frame = self.transport.frames.lock().unwrap()[0].clone(); + for directory in &self.directories { + // Reopening independently proves the append fence survived fsync. + let store = cellule_runtime::FollowerStore::open( + directory.path().to_owned(), + Limits::default(), + DiskBudget::new(1 << 30), + ) + .unwrap(); + let page = store.fleet_lanes_page(None, 128, now_ms()).await.unwrap(); + assert_eq!(page.entries().len(), 1); + assert_eq!( + page.entries()[0].state, + cellule_runtime::follower::FollowerLaneState::Retired + ); + assert_eq!(page.entries()[0].retired_through, Some(1)); + assert!( + store + .append(SessionId::from_bytes([191; 16]), 7, vec![frame.clone()], 0) + .await + .is_err() + ); + } + let control = CellAuthority::new(self.fixture.layout.clone()) + .load(self.fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + let root = control.value().ltx_root().unwrap(); + assert_eq!(root.commit_sequence, 1); + let restored = self + .fixture + ._directory + .path() + .join("maintenance-restored.sqlite"); + let verified = self.fixture.replica.open_root(&root).await.unwrap(); + assert_eq!(verified.restore(&restored).await.unwrap(), root.position); + let connection = cellule_ltx::rusqlite::Connection::open(restored).unwrap(); + assert_eq!( + connection + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) + .unwrap(), + 1 + ); + let request = mutation_identity_window(195, 10, 10_000); + assert_eq!( + connection + .query_row( + "SELECT result FROM sys_requests WHERE request_id = ?1", + [request.request_id.as_bytes().as_slice()], + |row| row.get::<_, Vec>(0) + ) + .unwrap(), + b"durable maintenance" + ); + } +} + +fn store() -> Store { + Store::new(Arc::new(InMemory::new())) +} + +#[tokio::test(flavor = "multi_thread")] +async fn maintenance_joins_every_member_and_retries_original_scope_after_lost_reply() { + let test = RetirementFixture::new(b"maintenance-retire-lost", store()).await; + test.command().await; + test.handle.drain().await.unwrap(); + test.transport.lose_reply.store(true, Ordering::Release); + test.transport.pause.store(true, Ordering::Release); + let durability = test.durability.clone(); + let task = tokio::spawn(async move { durability.shutdown_for_maintenance().await }); + test.transport.entered.acquire().await.unwrap().forget(); + let pending = !task.is_finished(); + let closed = test.authority.closes.lock().unwrap().clone(); + test.transport.release.add_permits(1); + let error = task.await.unwrap().unwrap_err(); + assert!(pending); + assert!(closed.is_empty()); + assert!(std::error::Error::source(&error).is_some()); + let observation = test.durability.retirement_observation().unwrap().unwrap(); + assert_eq!(observation.members().len(), 2); + assert!( + matches!(observation.members()[0].result(), Err(error) if matches!(error.as_ref(), cellule_runtime::Error::Node("lost native retirement reply"))) + ); + assert!(observation.members()[1].result().is_ok()); + assert!(observation.confirmed().is_err()); + assert!(test.authority.closes.lock().unwrap().is_empty()); + let proof = test.durability.shutdown_for_maintenance().await.unwrap(); + assert_eq!(proof.barrier(), observation.barrier()); + assert_eq!(proof.barrier().log_epoch(), 7); + assert_eq!(proof.barrier().covered_through(), 1); + assert_eq!(proof.barrier().members().len(), 2); + assert!(Arc::ptr_eq( + &proof, + &test.durability.shutdown_for_maintenance().await.unwrap() + )); + assert_eq!(*test.authority.closes.lock().unwrap(), vec![7]); + let requests = test.transport.requests.lock().unwrap().clone(); + assert_eq!(&requests[..2], &requests[2..]); + test.assert_native_fences().await; + test.runtime.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread")] +async fn ordinary_best_effort_close_cannot_be_upgraded_to_maintenance_proof() { + let test = RetirementFixture::new(b"maintenance-retire-ordinary", store()).await; + test.command().await; + test.handle.drain().await.unwrap(); + test.transport.lose_reply.store(true, Ordering::Release); + test.durability.shutdown().await.unwrap(); + assert_eq!(*test.authority.closes.lock().unwrap(), vec![7]); + assert!( + test.durability + .retirement_observation() + .unwrap() + .unwrap() + .confirmed() + .is_err() + ); + assert!(matches!( + test.durability.shutdown_for_maintenance().await, + Err(cellule_runtime::Error::Node( + "node-log member retirement remains unconfirmed" + )) + )); + assert_eq!(test.transport.requests.lock().unwrap().len(), 2); + test.assert_native_fences().await; + test.runtime.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread")] +async fn contradictory_success_blocks_both_close_paths() { + let test = RetirementFixture::new(b"maintenance-retire-contradict", store()).await; + test.command().await; + test.handle.drain().await.unwrap(); + for strict in [false, true] { + test.transport + .contradict_reply + .store(true, Ordering::Release); + let result = if strict { + test.durability.shutdown_for_maintenance().await.map(|_| ()) + } else { + test.durability.shutdown().await + }; + assert!(matches!( + result, + Err(cellule_runtime::Error::Node( + "follower retire receipt differs" + )) + )); + assert!(test.authority.closes.lock().unwrap().is_empty()); + } + test.durability.shutdown_for_maintenance().await.unwrap(); + test.assert_native_fences().await; + test.runtime.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread")] +async fn cancelled_maintenance_waiter_leaves_no_proof_or_authority_close() { + let test = RetirementFixture::new(b"maintenance-retire-cancel", store()).await; + test.command().await; + test.handle.drain().await.unwrap(); + test.transport.pause.store(true, Ordering::Release); + let durability = test.durability.clone(); + let task = tokio::spawn(async move { durability.shutdown_for_maintenance().await }); + test.transport.entered.acquire().await.unwrap().forget(); + task.abort(); + assert!(task.await.unwrap_err().is_cancelled()); + assert!(test.authority.closes.lock().unwrap().is_empty()); + assert!(test.durability.retirement_observation().unwrap().is_none()); + test.durability.shutdown_for_maintenance().await.unwrap(); + test.assert_native_fences().await; + test.runtime.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread")] +async fn uncovered_cell_publication_blocks_member_retirement() { + let pausing = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); + let test = + RetirementFixture::new(b"maintenance-retire-coverage", Store::new(pausing.clone())).await; + pausing.arm_next_update(); + test.command().await; + pausing.wait_until_blocked().await; + let result = test.durability.shutdown_for_maintenance().await; + let requests = test.transport.requests.lock().unwrap().len(); + let closes = test.authority.closes.lock().unwrap().len(); + pausing.release(); + test.handle.drain().await.unwrap(); + assert!(matches!( + result, + Err(cellule_runtime::Error::PendingPublication) + )); + assert_eq!(requests, 0); + assert_eq!(closes, 0); + test.durability.shutdown_for_maintenance().await.unwrap(); + test.assert_native_fences().await; + test.runtime.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread")] +async fn confirmed_fences_survive_lost_signed_authority_close_reply() { + let (test, authority) = + RetirementFixture::new_protocol(b"signed-close-lost", store(), true).await; + let authority = authority.unwrap(); + test.command().await; + test.handle.drain().await.unwrap(); + authority.lose_reply.store(true, Ordering::Release); + let error = test + .durability + .shutdown_for_maintenance() + .await + .unwrap_err(); + assert!(matches!( + error, + cellule_runtime::Error::Facility { + name: "signed-close-reply", + .. + } + )); + assert!(std::error::Error::source(&error).is_some()); + let observation = test.durability.retirement_observation().unwrap().unwrap(); + let confirmed = observation.confirmed().unwrap(); + assert_eq!(test.transport.requests.lock().unwrap().len(), 2); + // A fresh request now fails the real signed-directory authorization path. + assert!( + authority + .directory + .authorize_log_retire( + SessionId::from_bytes([191; 16]), + NodeId::from_bytes([193; 16]), + 7, + 1, + now_ms() + ) + .await + .is_err() + ); + let proof = test.durability.shutdown_for_maintenance().await.unwrap(); + assert_eq!(proof.as_ref(), &confirmed); + assert!(Arc::ptr_eq( + &observation, + &test.durability.retirement_observation().unwrap().unwrap() + )); + assert_eq!(test.transport.requests.lock().unwrap().len(), 2); + assert_eq!(authority.attempts.load(Ordering::Acquire), 2); + test.durability.shutdown().await.unwrap(); + assert!(Arc::ptr_eq( + &proof, + &test.durability.shutdown_for_maintenance().await.unwrap() + )); + assert_eq!(authority.attempts.load(Ordering::Acquire), 2); + test.assert_native_fences().await; + test.runtime.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread")] +async fn cancelled_signed_close_waiter_reuses_confirmed_fences_for_both_paths() { + for strict_retry in [false, true] { + let (test, authority) = RetirementFixture::new_protocol( + if strict_retry { + b"signed-close-cancel-strict" + } else { + b"signed-close-cancel-ordinary" + }, + store(), + true, + ) + .await; + let authority = authority.unwrap(); + test.command().await; + test.handle.drain().await.unwrap(); + authority.pause_reply.store(true, Ordering::Release); + let durability = test.durability.clone(); + let task = tokio::spawn(async move { durability.shutdown_for_maintenance().await }); + authority.entered.acquire().await.unwrap().forget(); + let original = test.durability.retirement_observation().unwrap().unwrap(); + assert!(original.confirmed().is_ok()); + task.abort(); + assert!(task.await.unwrap_err().is_cancelled()); + assert!( + authority + .directory + .authorize_log_retire( + SessionId::from_bytes([191; 16]), + NodeId::from_bytes([194; 16]), + 7, + 1, + now_ms() + ) + .await + .is_err() + ); + if strict_retry { + test.durability.shutdown_for_maintenance().await.unwrap(); + } else { + test.durability.shutdown().await.unwrap(); + } + let proof = test.durability.shutdown_for_maintenance().await.unwrap(); + assert_eq!(proof.as_ref(), &original.confirmed().unwrap()); + assert!(Arc::ptr_eq( + &original, + &test.durability.retirement_observation().unwrap().unwrap() + )); + assert_eq!(test.transport.requests.lock().unwrap().len(), 2); + assert_eq!(authority.attempts.load(Ordering::Acquire), 2); + test.assert_native_fences().await; + test.runtime.shutdown().await.unwrap(); + } +} + +#[tokio::test(flavor = "multi_thread")] +async fn lost_enrollment_and_refusal_replies_reconcile_only_the_original_retained_attempt() { + for refusing in [false, true] { + let pausing = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); + let (test, authority) = RetirementFixture::new_protocol( + if refusing { + b"enrollment-refusal-lost" + } else { + b"enrollment-commit-lost" + }, + Store::new(pausing.clone()), + true, + ) + .await; + let authority = authority.unwrap(); + test.command().await; + test.handle.drain().await.unwrap(); + test.durability.shutdown_for_maintenance().await.unwrap(); + let source = authority.observed.lock().await.clone(); + let directory = authority.directory.clone(); + let prepared = directory + .prepare_log_enrollment(&source, 8, 1, 3, now_ms()) + .await + .unwrap() + .unwrap(); + let attempt = directory + .prepare_log_enrollment_attempt(&prepared, now_ms()) + .await + .unwrap(); + pausing.lost_update_response.store(true, Ordering::Release); + pausing.arm_gets(); + let job = { + let directory = directory.clone(); + let attempt = attempt.clone(); + tokio::spawn(async move { + if refusing { + directory + .fence_log_enrollment(&attempt, now_ms()) + .await + .map(|_| ()) + } else { + directory + .commit_log_enrollment(&attempt, now_ms()) + .await + .map(|_| ()) + } + }) + }; + // The backend accepted the CAS, lost its reply and is now paused in the + // canonical reconciliation read. No new selection occurs after abort. + pausing.wait_until_get_blocked().await; + job.abort(); + assert!(job.await.unwrap_err().is_cancelled()); + pausing.release_gets(); + if refusing { + assert!( + directory + .commit_log_enrollment(&attempt, now_ms()) + .await + .is_err() + ); + let proof = directory + .fence_log_enrollment(&attempt, now_ms()) + .await + .unwrap(); + assert_eq!(proof.prepared().followers(), prepared.followers()); + assert!(proof.refusal().advertisement().log().is_none()); + assert!( + directory + .inspect_log_enrollment(&attempt, now_ms()) + .await + .unwrap() + .is_none() + ); + } else { + assert!( + directory + .fence_log_enrollment(&attempt, now_ms()) + .await + .is_err() + ); + let proof = directory + .inspect_log_enrollment(&attempt, now_ms()) + .await + .unwrap() + .unwrap(); + assert_eq!(proof.prepared().followers(), prepared.followers()); + assert_eq!( + proof.enrollment().advertisement().log(), + Some(prepared.log()) + ); + let gate = DurabilityGate::new( + source.advertisement().session(), + source.advertisement().node(), + 8, + prepared.log().members().iter().copied(), + ) + .unwrap(); + directory + .close_log( + proof.enrollment(), + &gate.begin_rotation().unwrap(), + now_ms(), + ) + .await + .unwrap(); + assert!( + directory + .inspect_log_enrollment(&attempt, now_ms()) + .await + .unwrap() + .is_none() + ); + // Enrollment followed by closure is not a refusal of that enrollment. + assert!( + directory + .fence_log_enrollment(&attempt, now_ms()) + .await + .is_err() + ); + } + test.assert_native_fences().await; + test.runtime.shutdown().await.unwrap(); + } +} diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/idle/acquire.rs b/crates/cellule-runtime/tests/runtime/lifecycle/idle/acquire.rs index b3ef36e6..ffed4d7f 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/idle/acquire.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/idle/acquire.rs @@ -2,6 +2,114 @@ use super::*; +#[tokio::test] +async fn acquisition_metadata_failure_blocks_admission_and_releases_resources() { + acquisition_metadata_fault(1).await; +} + +#[tokio::test] +async fn lost_acquisition_metadata_reply_adopts_the_original_before_serving() { + acquisition_metadata_fault(2).await; +} + +async fn acquisition_metadata_fault(fault: usize) { + let backend = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); + let fixture = fixture_with_limits_and_store( + b"acquisition-metadata-fault", + Limits::default(), + Store::new(backend.clone()), + ); + let first = SessionId::from_bytes([112; 16]); + let original = CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 16 << 20, first).unwrap(); + let handle = bootstrap_on(&original, &fixture, first).await; + let catalog = handle.catalog().clone(); + handle.drain().await.unwrap(); + original.shutdown().await.unwrap(); + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + let input = idle.value().clone(); + let next = SessionId::from_bytes([113; 16]); + let receiver = CellRuntime::new_with_replica_host( + SqlWorkerPool::new(1, 1).unwrap(), + 16 << 20, + next, + ReplicaHost::default().with_local_disk_budget(DiskBudget::new(1 << 30)), + ) + .unwrap(); + backend.acquisition_fault.store(fault, Ordering::Release); + let result = receiver + .acquire_idle_restored( + catalog.clone(), + fixture.replica.clone(), + authority.clone(), + idle, + fixture._directory.path().join("metadata-receiver.sqlite"), + Owner { + session: next, + endpoint: "https://metadata-receiver.internal:8081".into(), + }, + ) + .await; + assert_eq!(backend.acquisition_fault.load(Ordering::Acquire), 0); + let current = authority.load(input.cell).await.unwrap().unwrap(); + let retained = authority + .acquisition_record(input.cell, input.incarnation, input.epoch + 1) + .await + .unwrap(); + if fault == 1 { + assert!(matches!(result, Err(cellule_runtime::Error::Storage(_)))); + assert_eq!(current.value().state, ControlState::Idle); + assert_eq!(current.value().root, input.root); + assert!(current.value().owner.is_none()); + assert!(retained.is_none()); + assert!( + receiver + .local_handle(catalog, ¤t) + .await + .unwrap() + .is_none() + ); + } else { + let handle = result.unwrap(); + assert_eq!(current.value().state, ControlState::Serving); + let retained = retained.unwrap(); + assert_eq!(retained.input(), &input); + assert_eq!(retained.materialized().root, input.root); + assert_eq!(retained.materialized().state, ControlState::Recovering); + assert_eq!( + handle + .query(64, 64, |connection| { + Ok(connection + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))? + .to_be_bytes() + .to_vec()) + }) + .await + .unwrap(), + 0_i64.to_be_bytes(), + ); + handle.drain().await.unwrap(); + assert_eq!( + authority + .acquisition_record(input.cell, input.incarnation, input.epoch + 1) + .await + .unwrap(), + Some(retained), + ); + } + receiver.shutdown().await.unwrap(); + let stats = receiver.stats(); + assert_eq!(stats.active_cells(), 0); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.file_descriptors(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); +} + #[tokio::test(flavor = "multi_thread")] async fn failed_idle_receiver_does_not_leave_authority_owned() { let fixture = fixture_for(b"receiver-activation-failure"); @@ -28,6 +136,7 @@ async fn failed_idle_receiver_does_not_leave_authority_owned() { .unwrap() .unwrap(); let expected_root = idle.value().root.clone(); + let original_input = idle.value().clone(); let successor = SessionId::from_bytes([113; 16]); let runtime = CellRuntime::new( SqlWorkerPool::new(1, 1).unwrap(), @@ -65,6 +174,23 @@ async fn failed_idle_receiver_does_not_leave_authority_owned() { assert_eq!(current.value().state, ControlState::Idle); assert!(current.value().owner.is_none()); assert_eq!(current.value().root, expected_root); + // Retained input survives a failed restore; it cannot certify serving. + let acquisition = authority + .acquisition_record( + original_input.cell, + original_input.incarnation, + current.value().epoch, + ) + .await + .unwrap() + .unwrap(); + assert_eq!(acquisition.input(), &original_input); + assert_eq!(acquisition.materialized().state, ControlState::Recovering); + assert_eq!( + acquisition.materialized().owner.as_ref().unwrap().session, + successor + ); + runtime.shutdown().await.unwrap(); } #[tokio::test] diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/idle/release.rs b/crates/cellule-runtime/tests/runtime/lifecycle/idle/release.rs index e5fde4e9..caa757d6 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/idle/release.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/idle/release.rs @@ -2,6 +2,142 @@ use super::*; +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn exact_position_release_rejects_stale_identity_and_keeps_the_final_root() { + let fixture = fixture_for(b"exact-release-position"); + let (runtime, handle, _) = activate_runtime(&fixture, 64 << 20).await; + let authority = CellAuthority::new(fixture.layout.clone()); + let before = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + let (_, generation, _, _) = runtime.idle_transfer_candidates().await.unwrap()[0]; + let cell = fixture.target.cell_id(); + let session = SessionId::from_bytes([4; 16]); + let incarnation = handle.incarnation(); + let epoch = before.value().epoch; + for (source, activation, identity, ownership) in [ + ( + SessionId::from_bytes([171; 16]), + generation, + incarnation, + epoch, + ), + (session, generation + 1, incarnation, epoch), + ( + session, + generation, + IncarnationId::from_bytes([172; 16]), + epoch, + ), + (session, generation, incarnation, epoch + 1), + (session, generation, incarnation, 0), + ] { + assert!( + runtime + .release_idle_cell_at(cell, source, activation, identity, ownership) + .await + .is_err() + ); + assert_eq!( + authority.load(cell).await.unwrap().unwrap().value(), + before.value() + ); + } + let clock = now_ms(); + let identity = mutation_identity_window(173, clock, clock + 60_000); + let digest = Digest::from_bytes([173; 32]); + let acknowledged = handle + .execute(identity, digest, clock, 64, 64, |transaction| { + transaction.execute("UPDATE counter SET value = 41", [])?; + Ok(HandlerOutcome::Success(Vec::new())) + }) + .await + .unwrap(); + let published = authority.load(cell).await.unwrap().unwrap(); + assert_ne!(published.value().root, before.value().root); + let position = tokio::time::timeout(std::time::Duration::from_secs(5), async { + loop { + match runtime + .release_idle_cell_at(cell, session, generation, incarnation, epoch) + .await + { + Ok(position) => return position, + Err(cellule_runtime::Error::CellReleaseRefused { source, .. }) + if matches!(*source, cellule_runtime::Error::CellDraining) => + { + tokio::task::yield_now().await + } + Err(error) => panic!("exact release failed: {error}"), + } + } + }) + .await + .unwrap(); + assert_eq!(position.incarnation, incarnation); + assert_eq!(position.epoch, epoch); + assert_eq!(Some(&position.root), published.value().root.as_ref()); + assert!(position.root.commit_sequence >= acknowledged.commit_sequence()); + assert_eq!(runtime.stats().active_cells(), 0); + let idle = authority.load(cell).await.unwrap().unwrap(); + assert_eq!(idle.value().state, ControlState::Idle); + assert!(idle.value().owner.is_none()); + assert_eq!(Some(&position.root), idle.value().root.as_ref()); + + let receiver = CellRuntime::new_with_replica_host( + SqlWorkerPool::new(1, 2).unwrap(), + 64 << 20, + SessionId::from_bytes([174; 16]), + ReplicaHost::default().with_local_disk_budget(DiskBudget::new(8 << 30)), + ) + .unwrap(); + let successor = receiver + .acquire_idle_restored( + handle.catalog().clone(), + fixture.replica.clone(), + authority.clone(), + idle, + fixture._directory.path().join("successor.sqlite"), + Owner { + session: SessionId::from_bytes([174; 16]), + endpoint: "https://successor.internal:8081".into(), + }, + ) + .await + .unwrap(); + assert_eq!( + successor + .resolve(identity, digest, now_ms(), 64) + .await + .unwrap(), + Resolution::Committed(acknowledged) + ); + successor + .execute( + mutation_identity_window(175, clock, clock + 60_000), + Digest::from_bytes([175; 32]), + now_ms(), + 64, + 64, + |transaction| { + transaction.execute("UPDATE counter SET value = 42", [])?; + Ok(HandlerOutcome::Success(Vec::new())) + }, + ) + .await + .unwrap(); + let advanced = authority.load(cell).await.unwrap().unwrap(); + assert!(advanced.value().epoch > position.epoch); + assert!( + advanced.value().root.as_ref().unwrap().commit_sequence > position.root.commit_sequence + ); + assert_eq!(Some(&position.root), published.value().root.as_ref()); + successor.drain().await.unwrap(); + receiver.shutdown().await.unwrap(); + runtime.shutdown().await.unwrap(); +} + #[tokio::test] async fn exact_idle_release_checks_generation_and_confirms_authority_release() { let fixture = fixture_for(b"exact-idle-release"); diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/inventory.rs b/crates/cellule-runtime/tests/runtime/lifecycle/inventory.rs new file mode 100644 index 00000000..b4c3118e --- /dev/null +++ b/crates/cellule-runtime/tests/runtime/lifecycle/inventory.rs @@ -0,0 +1,294 @@ +//! Public ownership inventory, residence tracking, and native-byte accounting. + +use super::*; +use cellule_runtime::cell::actor::{CellInventoryCursor, CellInventoryEntry}; +use cellule_runtime::fleet::operations::{DrainBlocker, MAX_PAGE_BYTES}; + +pub(super) async fn stable_owner( + runtime: &CellRuntime, +) -> cellule_runtime::cell::actor::OwnedCellObservation { + tokio::time::timeout(std::time::Duration::from_secs(5), async { + loop { + let page = runtime.fleet_cells_page(None, 128).await.unwrap(); + if let Some(CellInventoryEntry::Owned(owner)) = page.entries().first() + && owner.cost.is_some() + && owner.stable_observations == 2 + { + return (**owner).clone(); + } + drop(page); + tokio::time::sleep(std::time::Duration::from_millis(25)).await; + } + }) + .await + .unwrap() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn paged_inventory_preserves_all_owners_and_rejects_a_release_changed_cursor() { + let first = fixture_for(b"inventory-first"); + let second = fixture_for(b"inventory-second"); + let session = SessionId::from_bytes([4; 16]); + let pool = SqlWorkerPool::new(2, 10).unwrap(); + let runtime = CellRuntime::new(pool.clone(), 64 << 20, session).unwrap(); + let first_handle = bootstrap_on(&runtime, &first, session).await; + let second_handle = bootstrap_on(&runtime, &second, session).await; + let before = runtime.stats().retained_bytes(); + let page = runtime.fleet_cells_page(None, 1).await.unwrap(); + assert_eq!(page.session(), session); + assert_eq!(page.owned_cells(), 2); + assert_eq!(page.transitioning_cells(), 0); + assert_eq!(page.entries().len(), 1); + assert_eq!( + runtime.stats().retained_bytes(), + before + MAX_PAGE_BYTES as usize + ); + let cursor = CellInventoryCursor::from_bytes(&page.next().unwrap().to_bytes()).unwrap(); + let next = runtime.fleet_cells_page(Some(cursor), 1).await.unwrap(); + assert!(next.next().is_none()); + assert_eq!(next.topology(), page.topology()); + let mut actual = [page.entries()[0].cell(), next.entries()[0].cell()]; + actual.sort_by_key(|cell| *cell.as_bytes()); + let mut expected = [first.target.cell_id(), second.target.cell_id()]; + expected.sort_by_key(|cell| *cell.as_bytes()); + assert_eq!(actual, expected); + drop(next); + drop(page); + assert_eq!(runtime.stats().retained_bytes(), before); + first_handle.drain().await.unwrap(); + assert!(matches!( + runtime.fleet_cells_page(Some(cursor), 1).await, + Err(cellule_runtime::Error::Node(_)) + )); + let remaining = runtime.fleet_cells_page(None, 128).await.unwrap(); + assert_eq!(remaining.owned_cells(), 1); + assert_eq!(remaining.entries()[0].cell(), second.target.cell_id()); + drop(remaining); + second_handle.drain().await.unwrap(); + let empty = runtime.fleet_cells_page(None, 128).await.unwrap(); + assert!(empty.entries().is_empty()); + assert!(empty.next().is_none()); + drop(empty); + runtime.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn busy_owner_stays_visible_and_later_use_cannot_reset_residence_time() { + let fixture = fixture(); + let (runtime, handle, _pool) = activate_runtime(&fixture, 64 << 20).await; + let page = runtime.fleet_cells_page(None, 128).await.unwrap(); + let CellInventoryEntry::Owned(initial) = &page.entries()[0] else { + panic!("active owner omitted") + }; + let since = initial.resident_since_ms; + let topology = page.topology(); + assert_eq!(initial.target, fixture.target); + assert!(initial.position.is_some()); + assert!(initial.cost.is_some()); + assert_eq!(initial.maintenance_cost, initial.cost); + assert!(initial.database_bytes.is_some_and(|bytes| bytes > 0)); + assert!(initial.sampled_at_ms.is_some()); + assert!(initial.stable_observations >= 1); + drop(page); + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + let started = Arc::new(Notify::new()); + let (release_tx, release_rx) = mpsc::channel(); + let query = { + let handle = handle.clone(); + let started = Arc::clone(&started); + tokio::spawn(async move { + handle + .query(64, 64, move |_| { + started.notify_one(); + release_rx.recv().unwrap(); + Ok(Vec::new()) + }) + .await + }) + }; + tokio::time::timeout(std::time::Duration::from_secs(3), started.notified()) + .await + .unwrap(); + let observation = tokio::time::timeout( + std::time::Duration::from_secs(3), + runtime.fleet_cells_page(None, 128), + ) + .await; + // Release before assertions so a failed inventory cannot strand SQLite. + release_tx.send(()).unwrap(); + query.await.unwrap().unwrap(); + let page = observation.unwrap().unwrap(); + assert_eq!(page.topology(), topology); + assert_eq!(page.owned_cells(), 1); + let CellInventoryEntry::Owned(owner) = &page.entries()[0] else { + panic!("busy owner omitted") + }; + assert_eq!(owner.resident_since_ms, since); + assert!(owner.last_used_ms > since); + assert!(owner.blockers.contains(&DrainBlocker::BusyExecution)); + assert!(owner.maintenance_cost.is_some()); + drop(page); + handle.drain().await.unwrap(); + runtime.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn mutation_invalidates_demand_then_real_worker_samples_restore_stability() { + let fixture = fixture_for(b"inventory-growing-cell"); + let (runtime, handle, _pool) = activate_runtime(&fixture, 64 << 20).await; + let initial = stable_owner(&runtime).await; + let initial_bytes = initial.database_bytes.unwrap(); + let started = Arc::new(Notify::new()); + let (release_tx, release_rx) = mpsc::channel(); + let clock = now_ms(); + let write = { + let handle = handle.clone(); + let started = Arc::clone(&started); + tokio::spawn(async move { + handle.execute(mutation_identity_window(81, clock, clock + 60_000), + Digest::from_bytes([81; 32]), clock, 64, 64, move |transaction| { + started.notify_one(); + release_rx.recv().unwrap(); + transaction.execute_batch("CREATE TABLE payload(value BLOB NOT NULL); INSERT INTO payload VALUES(randomblob(1048576))")?; + Ok(HandlerOutcome::Success(Vec::new())) + }).await + }) + }; + tokio::time::timeout(std::time::Duration::from_secs(3), started.notified()) + .await + .unwrap(); + let observed = runtime.fleet_cells_page(None, 128).await; + release_tx.send(()).unwrap(); + let committed = write.await.unwrap().unwrap(); + let page = observed.unwrap(); + let CellInventoryEntry::Owned(during) = &page.entries()[0] else { + panic!("executing owner omitted") + }; + assert!(during.cost.is_none()); + assert_eq!(during.maintenance_cost, initial.cost); + assert_eq!(during.stable_observations, 0); + assert!(during.blockers.contains(&DrainBlocker::UnknownInventory)); + drop(page); + let settled = stable_owner(&runtime).await; + assert!(settled.database_bytes.unwrap() > initial_bytes); + assert_eq!(settled.resident_since_ms, initial.resident_since_ms); + assert_eq!(settled.generation, initial.generation); + assert_eq!( + settled.position.unwrap().root.commit_sequence, + committed.commit_sequence() + ); + assert_eq!(settled.cost, initial.cost); + assert_eq!(settled.maintenance_cost, initial.maintenance_cost); + assert!(settled.work_blocker.is_none()); + handle.drain().await.unwrap(); + runtime.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn unknown_schema_inventory_has_no_cost_and_cannot_release_its_owner() { + let fixture = fixture_for(b"inventory-missing-queue-schema"); + let session = SessionId::from_bytes([4; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 64 << 20, session).unwrap(); + let _handle = bootstrap_role_on( + &runtime, + &fixture, + session, + CatalogRole::Queue, + |transaction| { + transaction.execute_batch("CREATE TABLE payload(value INTEGER)")?; + Ok(()) + }, + ) + .await; + let page = runtime.fleet_cells_page(None, 128).await.unwrap(); + let CellInventoryEntry::Owned(owner) = &page.entries()[0] else { + panic!("unknown owner omitted") + }; + assert!(owner.cost.is_none()); + assert!(owner.database_bytes.is_none()); + assert_eq!(owner.stable_observations, 0); + assert!(owner.blockers.contains(&DrainBlocker::UnknownInventory)); + let generation = owner.generation; + drop(page); + assert!( + runtime + .release_idle_cell(fixture.target.cell_id(), session, generation) + .await + .is_err() + ); + let authority = CellAuthority::new(fixture.layout.clone()) + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(authority.value().owner.as_ref().unwrap().session, session); + assert_eq!(runtime.unreleased_cell_count().await.unwrap(), 1); + runtime.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn blob_owner_has_no_busy_maintenance_envelope_and_refusal_leaves_foreground_open() { + use cellule_runtime::cell::actor::MaintenanceCellRelease; + let fixture = fixture_for(b"inventory-unproven-blob-owner"); + let session = SessionId::from_bytes([4; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 64 << 20, session).unwrap(); + let handle = bootstrap_role_on( + &runtime, + &fixture, + session, + CatalogRole::Blob, + |transaction| { + transaction.execute_batch( + "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES(42)", + )?; + Ok(()) + }, + ) + .await; + let page = runtime.fleet_cells_page(None, 128).await.unwrap(); + let CellInventoryEntry::Owned(owner) = &page.entries()[0] else { + panic!("Blob owner omitted") + }; + assert!(owner.maintenance_cost.is_none()); + let generation = owner.generation; + let incarnation = owner.incarnation; + let epoch = owner.position.as_ref().unwrap().epoch; + drop(page); + let result = runtime + .release_maintenance_cell_at( + fixture.target.cell_id(), + session, + generation, + incarnation, + epoch, + tokio::time::Instant::now() + std::time::Duration::from_secs(3), + ) + .await + .unwrap(); + assert!(matches!( + result, + MaintenanceCellRelease::Refused { + blocker: DrainBlocker::UnknownInventory, + error: None + } + )); + let page = runtime.fleet_cells_page(None, 128).await.unwrap(); + let CellInventoryEntry::Owned(owner) = &page.entries()[0] else { + panic!("Blob owner omitted") + }; + assert!(!owner.quiescing); + drop(page); + assert_eq!( + handle + .query(64, 64, |connection| { + let value = connection + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; + Ok(value.to_be_bytes().to_vec()) + }) + .await + .unwrap(), + 42_i64.to_be_bytes() + ); + runtime.shutdown().await.unwrap(); + assert_eq!(runtime.stats().retained_bytes(), 0); +} diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/maintenance.rs b/crates/cellule-runtime/tests/runtime/lifecycle/maintenance.rs new file mode 100644 index 00000000..b98a4360 --- /dev/null +++ b/crates/cellule-runtime/tests/runtime/lifecycle/maintenance.rs @@ -0,0 +1,195 @@ +//! Public maintenance admission while accepted SQLite work is still running. + +use super::*; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn maintenance_quiescence_keeps_accepted_work_and_original_resolution() { + let fixture = fixture(); + let (runtime, handle, _) = activate_runtime(&fixture, 16 << 20).await; + let owner = inventory::stable_owner(&runtime).await; + let identity = mutation_identity_window(91, 10, 10_000); + let digest = Digest::from_bytes([92; 32]); + let started = Arc::new(Notify::new()); + let (release_tx, release_rx) = mpsc::channel(); + let executing = { + let handle = handle.clone(); + let started = Arc::clone(&started); + tokio::spawn(async move { + handle + .execute(identity, digest, 20, 1024, 1024, move |transaction| { + started.notify_one(); + release_rx.recv().unwrap(); + transaction.execute("UPDATE counter SET value = value + 1", [])?; + Ok(HandlerOutcome::Success( + b"accepted-before-quiescence".to_vec(), + )) + }) + .await + }) + }; + tokio::time::timeout(std::time::Duration::from_secs(3), started.notified()) + .await + .unwrap(); + let closed = tokio::time::timeout( + std::time::Duration::from_secs(3), + runtime.quiesce_cell_at( + fixture.target.cell_id(), + SessionId::from_bytes([4; 16]), + owner.generation, + owner.incarnation, + owner.position.unwrap().epoch, + ), + ) + .await; + let refused = tokio::time::timeout( + std::time::Duration::from_secs(3), + handle.query(1, 1, |_| panic!("foreground query admitted")), + ) + .await; + // Join the accepted effect before asserting, including on a failed barrier. + release_tx.send(()).unwrap(); + let outcome = executing.await.unwrap().unwrap(); + closed.unwrap().unwrap(); + assert!(matches!( + refused.unwrap(), + Err(cellule_runtime::Error::CellDraining) + )); + assert!( + matches!(outcome, StoredOutcome::Success { ref result, commit_sequence: 1 } if result == b"accepted-before-quiescence") + ); + assert_eq!( + handle.resolve(identity, digest, 21, 1024).await.unwrap(), + Resolution::Committed(outcome) + ); + handle.drain().await.unwrap(); + runtime.shutdown().await.unwrap(); + assert_eq!(runtime.stats().retained_bytes(), 0); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn expired_maintenance_inventory_cannot_refuse_the_new_request_and_accepted_sql_moves() { + use cellule_runtime::cell::actor::MaintenanceCellRelease; + use cellule_runtime::fleet::operations::DrainBlocker; + let fixture = fixture(); + let (runtime, handle, _) = activate_runtime(&fixture, 16 << 20).await; + let owner = inventory::stable_owner(&runtime).await; + let identity = mutation_identity_window(93, 10, 10_000); + let digest = Digest::from_bytes([94; 32]); + let started = Arc::new(Notify::new()); + let (release_tx, release_rx) = mpsc::channel(); + let executing = { + let handle = handle.clone(); + let started = Arc::clone(&started); + tokio::spawn(async move { + handle + .execute(identity, digest, 20, 1024, 1024, move |transaction| { + started.notify_one(); + release_rx.recv().unwrap(); + transaction.execute("UPDATE counter SET value = value + 1", [])?; + Ok(HandlerOutcome::Success(b"accepted".to_vec())) + }) + .await + }) + }; + tokio::time::timeout(std::time::Duration::from_secs(3), started.notified()) + .await + .unwrap(); + let source = SessionId::from_bytes([4; 16]); + let epoch = owner.position.unwrap().epoch; + let first = tokio::time::timeout( + std::time::Duration::from_secs(2), + runtime.release_maintenance_cell_at( + fixture.target.cell_id(), + source, + owner.generation, + owner.incarnation, + epoch, + tokio::time::Instant::now() + std::time::Duration::from_millis(150), + ), + ) + .await; + let second = { + let runtime = runtime.clone(); + let cell = fixture.target.cell_id(); + tokio::spawn(async move { + runtime + .release_maintenance_cell_at( + cell, + source, + owner.generation, + owner.incarnation, + epoch, + tokio::time::Instant::now() + std::time::Duration::from_secs(5), + ) + .await + }) + }; + // Let the replacement request reach the actor while the original read is + // still queued behind SQL. Join SQL on every path before assertions. + tokio::task::yield_now().await; + release_tx.send(()).unwrap(); + let outcome = executing.await.unwrap().unwrap(); + assert!(matches!( + first.unwrap().unwrap(), + MaintenanceCellRelease::Refused { + blocker: DrainBlocker::Deadline, + error: None + } + )); + let MaintenanceCellRelease::Released(position) = second.await.unwrap().unwrap() else { + panic!("new request was refused by stale inventory"); + }; + assert_eq!(position.root.commit_sequence, 1); + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(idle.value().root.as_ref(), Some(&position.root)); + let receiver = SessionId::from_bytes([95; 16]); + let destination = + CellRuntime::new(SqlWorkerPool::new(1, 10).unwrap(), 16 << 20, receiver).unwrap(); + let restored = destination + .acquire_idle_restored( + handle.catalog().clone(), + fixture.replica.clone(), + authority, + idle, + fixture + ._directory + .path() + .join("maintenance-receiver.sqlite"), + Owner { + session: receiver, + endpoint: "https://receiver.internal:8081".into(), + }, + ) + .await + .unwrap(); + assert_eq!( + restored + .query(64, 64, |connection| Ok(connection + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))? + .to_be_bytes() + .to_vec())) + .await + .unwrap(), + 1_i64.to_be_bytes() + ); + assert_eq!( + restored.resolve(identity, digest, 21, 1024).await.unwrap(), + Resolution::Committed(outcome) + ); + assert!( + handle + .query(1, 1, |_| panic!("released SQL owner ran")) + .await + .is_err() + ); + restored.drain().await.unwrap(); + destination.shutdown().await.unwrap(); + runtime.shutdown().await.unwrap(); + assert_eq!(runtime.stats().retained_bytes(), 0); + assert_eq!(destination.stats().retained_bytes(), 0); +} diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/ownership.rs b/crates/cellule-runtime/tests/runtime/lifecycle/ownership.rs index b96a61f9..7acc881a 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/ownership.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/ownership.rs @@ -3,6 +3,7 @@ use super::*; mod lease; +mod prefix; mod recovery; mod sparse_process; mod succession; diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/ownership/lease.rs b/crates/cellule-runtime/tests/runtime/lifecycle/ownership/lease.rs index 815d0e9e..e0b73b6e 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/ownership/lease.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/ownership/lease.rs @@ -2,6 +2,60 @@ use super::*; +#[tokio::test] +async fn lifecycle_metadata_uses_shared_credit_without_opening_fenced_native_work() { + let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( + SqlWorkerPool::new(1, 1).unwrap(), + 1_024, + SessionId::from_bytes([46; 16]), + ReplicaHost::default(), + ) + .unwrap(); + assert!(matches!( + runtime.try_reserve_node_metadata_bytes(0), + Err(cellule_runtime::Error::Capacity("node retained bytes")) + )); + let metadata = runtime.try_reserve_node_metadata_bytes(1_024).unwrap(); + assert_eq!(runtime.stats().retained_bytes(), 1_024); + assert!(matches!( + runtime.try_reserve_node_metadata_bytes(1), + Err(cellule_runtime::Error::Capacity("node retained bytes")) + )); + assert!(matches!( + runtime.try_reserve_node_bytes(1), + Err(cellule_runtime::Error::Fenced) + )); + drop(metadata); + let lease = NodeLeaseGuard::new(0, 60_000).unwrap(); + runtime.install_node_lease(lease.clone()).unwrap(); + let native = runtime.try_reserve_node_bytes(1_023).unwrap(); + let metadata = runtime.try_reserve_node_metadata_bytes(1).unwrap(); + assert_eq!(runtime.stats().retained_bytes(), 1_024); + assert!(matches!( + runtime.try_reserve_node_metadata_bytes(1), + Err(cellule_runtime::Error::Capacity("node retained bytes")) + )); + drop(native); + lease.fence(); + let fenced_metadata = runtime.try_reserve_node_metadata_bytes(1_023).unwrap(); + assert!(matches!( + runtime.try_reserve_node_bytes(1), + Err(cellule_runtime::Error::Fenced) + )); + runtime.shutdown().await.unwrap(); + assert!(matches!( + runtime.try_reserve_node_metadata_bytes(1), + Err(cellule_runtime::Error::RuntimeClosed) + )); + assert!(matches!( + runtime.try_reserve_node_bytes(1), + Err(cellule_runtime::Error::RuntimeClosed) + )); + assert_eq!(runtime.stats().retained_bytes(), 1_024); + drop((metadata, fenced_metadata)); + assert_eq!(runtime.stats().retained_bytes(), 0); +} + #[tokio::test] async fn fleet_runtime_stays_fenced_until_one_live_node_lease_is_installed() { let session = SessionId::from_bytes([41; 16]); diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/ownership/prefix.rs b/crates/cellule-runtime/tests/runtime/lifecycle/ownership/prefix.rs new file mode 100644 index 00000000..25678275 --- /dev/null +++ b/crates/cellule-runtime/tests/runtime/lifecycle/ownership/prefix.rs @@ -0,0 +1,346 @@ +//! Complete origin verification and native derivation across owner movement. +use super::*; +use object_store::ObjectStoreExt; +use std::time::Duration; + +async fn increment(handle: &cellule_runtime::cell::actor::CellHandle, byte: u8) { + let clock = now_ms(); + handle + .execute( + mutation_identity_window(byte, clock, clock + 60_000), + Digest::from_bytes([byte; 32]), + clock, + 64, + 64, + |transaction| { + transaction.execute("UPDATE counter SET value = value + 1", [])?; + Ok(HandlerOutcome::Success(Vec::new())) + }, + ) + .await + .unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn exact_prefix_survives_compaction_movement_and_origin_loss_of_old_roots() { + let backend = Arc::new(InMemory::new()); + let fixture = fixture_with_limits_and_store( + b"root-prefix-movement", + Limits::default(), + Store::new(backend.clone()), + ); + let (source, handle, _) = activate_runtime(&fixture, 64 << 20).await; + let catalog = handle.catalog().clone(); + let authority = CellAuthority::new(fixture.layout.clone()); + increment(&handle, 1).await; + let first = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .unwrap(); + for byte in 2..=20 { + increment(&handle, byte).await; + } + let original = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .unwrap(); + handle.drain().await.unwrap(); + source.shutdown().await.unwrap(); + let idle = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + let next = SessionId::from_bytes([81; 16]); + let receiver = CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 64 << 20, next).unwrap(); + let handle = receiver + .acquire_idle_restored( + catalog, + fixture.replica.clone(), + authority.clone(), + idle, + fixture._directory.path().join("prefix-receiver.sqlite"), + Owner { + session: next, + endpoint: "https://prefix-receiver.internal".into(), + }, + ) + .await + .unwrap(); + for byte in 21..=40 { + increment(&handle, byte).await; + } + let current = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + let root = current.value().ltx_root().unwrap(); + let independent = CellAuthority::new(fixture.layout.clone()); + let proof = independent + .verify_root_prefix(original, root, &fixture.replica, 128) + .await + .unwrap(); + assert_eq!(proof.root(), root); + assert_eq!(proof.prefix(), original); + assert!(proof.dependency_count() > 0); + let lineage = independent.root_lineage(root).await.unwrap().unwrap(); + assert!(!lineage.predecessors().is_empty()); + // Old root documents are not selected dependencies after compaction. Their + // historical verified links remain; no test fabricates a preparation link. + let old_path = fixture.layout.incarnation_object_path( + &first.cell, + &first.incarnation, + &first.digest, + CellObjectKind::Root, + ); + backend.delete(&old_path).await.unwrap(); + assert!(fixture.replica.open_root(&first).await.is_err()); + independent + .verify_root_prefix(first, root, &fixture.replica, 128) + .await + .unwrap(); + assert_eq!( + handle + .query(64, 64, |connection| { + Ok(connection + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))? + .to_be_bytes() + .to_vec()) + }) + .await + .unwrap(), + 40_i64.to_be_bytes() + ); + handle.drain().await.unwrap(); + receiver.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn cached_root_metadata_cannot_hide_a_missing_or_corrupt_current_dependency() { + for corrupt in [false, true] { + let backend = Arc::new(InMemory::new()); + let fixture = fixture_with_limits_and_store( + b"root-prefix-origin-loss", + Limits::default(), + Store::new(backend.clone()), + ); + let (runtime, handle, _) = activate_runtime(&fixture, 64 << 20).await; + let authority = CellAuthority::new(fixture.layout.clone()); + let prefix = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .unwrap(); + increment(&handle, 1).await; + let root = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .unwrap(); + let objects = fixture.replica.reachable_objects(&root).await.unwrap(); + let proof = authority + .verify_root_prefix(prefix, root, &fixture.replica, 8) + .await + .unwrap(); + assert_eq!(proof.dependency_count(), objects.len()); + handle.drain().await.unwrap(); + runtime.shutdown().await.unwrap(); + let object = objects + .iter() + .find(|object| object.kind == CellObjectKind::Ltx) + .unwrap(); + let path = fixture.layout.incarnation_object_path( + &root.cell, + &root.incarnation, + &object.digest, + object.kind, + ); + if corrupt { + backend + .put(&path, Bytes::from_static(b"corrupt-current-ltx").into()) + .await + .unwrap(); + } else { + backend.delete(&path).await.unwrap(); + } + assert!(matches!( + authority + .verify_root_prefix(prefix, root, &fixture.replica, 8) + .await, + Err(cellule_runtime::Error::Ltx(_)) + )); + } +} + +#[tokio::test] +async fn unrelated_or_missing_lineage_and_invalid_limits_cannot_certify_a_prefix() { + let backend = Arc::new(InMemory::new()); + let fixture = fixture_with_limits_and_store( + b"root-prefix-missing", + Limits::default(), + Store::new(backend.clone()), + ); + let (runtime, handle, _) = activate_runtime(&fixture, 64 << 20).await; + let authority = CellAuthority::new(fixture.layout.clone()); + let prefix = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .unwrap(); + increment(&handle, 1).await; + let root = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .unwrap(); + let mut unrelated = prefix; + unrelated.digest = [99; 32]; + assert!(matches!( + authority + .verify_root_prefix(unrelated, root, &fixture.replica, 8) + .await, + Err(cellule_runtime::Error::RootPrefixUnproven { .. }) + )); + assert!(matches!( + authority + .verify_root_prefix(prefix, root, &fixture.replica, 0) + .await, + Err(cellule_runtime::Error::Capacity(_)) + )); + let mut foreign = prefix; + foreign.incarnation = [99; 16]; + assert!(matches!( + authority + .verify_root_prefix(foreign, root, &fixture.replica, 8) + .await, + Err(cellule_runtime::Error::Control(_)) + )); + backend + .delete( + &fixture + .layout + .root_lineage_path(&root.cell, &root.incarnation, &root.digest), + ) + .await + .unwrap(); + assert!( + matches!(authority.verify_root_prefix(prefix, root, &fixture.replica, 8).await, + Err(cellule_runtime::Error::RootLineageIncomplete { root: missing }) if missing == root) + ); + // Exact identity still requires actual current graph availability. + authority + .verify_root_prefix(root, root, &fixture.replica, 1) + .await + .unwrap(); + handle.drain().await.unwrap(); + runtime.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn runtime_prefix_verification_admits_memory_before_io_and_releases_every_exit() { + let backend = Arc::new(InMemory::new()); + let fixture = fixture_with_limits_and_store( + b"root-prefix-budget", + Limits::default(), + Store::new(backend.clone()), + ); + let (source, handle, _) = activate_runtime(&fixture, 64 << 20).await; + let catalog = handle.catalog().clone(); + let authority = CellAuthority::new(fixture.layout.clone()); + let root = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .unwrap(); + let slots = Arc::new(tokio::sync::Semaphore::new(1)); + let runtime = CellRuntime::new_with_replica_host( + SqlWorkerPool::new(1, 1).unwrap(), + 64 << 20, + SessionId::from_bytes([82; 16]), + ReplicaHost::default().with_io_slots(slots.clone()), + ) + .unwrap(); + // Block the shared origin facility, then poll the same owned proof future. + // Cancellation drops its memory token and all native admission; it does not + // fabricate completion or leave another task running outside the caller. + let io = slots.acquire().await.unwrap(); + let mut proof = Box::pin(runtime.verify_root_prefix( + &catalog, + &authority, + fixture.replica.clone(), + root, + root, + 8, + )); + assert!( + tokio::time::timeout(Duration::from_millis(10), &mut proof) + .await + .is_err() + ); + assert!(runtime.stats().retained_bytes() > 16 << 20); + drop(proof); + assert_eq!(runtime.stats().retained_bytes(), 0); + drop(io); + runtime + .verify_root_prefix(&catalog, &authority, fixture.replica.clone(), root, root, 8) + .await + .unwrap(); + assert_eq!(runtime.stats().retained_bytes(), 0); + // A missing origin root cannot hide refusal before the first origin read. + let path = fixture.layout.incarnation_object_path( + &root.cell, + &root.incarnation, + &root.digest, + CellObjectKind::Root, + ); + backend.delete(&path).await.unwrap(); + let held = runtime.try_reserve_node_bytes(64 << 20).unwrap(); + assert!(matches!( + runtime + .verify_root_prefix(&catalog, &authority, fixture.replica.clone(), root, root, 8,) + .await, + Err(cellule_runtime::Error::Capacity("node retained bytes")) + )); + assert_eq!(runtime.stats().retained_bytes(), 64 << 20); + drop(held); + assert!(matches!( + runtime + .verify_root_prefix(&catalog, &authority, fixture.replica.clone(), root, root, 8,) + .await, + Err(cellule_runtime::Error::Ltx(_)) + )); + assert_eq!(runtime.stats().retained_bytes(), 0); + runtime.shutdown().await.unwrap(); + assert!(matches!( + runtime + .verify_root_prefix(&catalog, &authority, fixture.replica.clone(), root, root, 8,) + .await, + Err(cellule_runtime::Error::RuntimeClosed) + )); + handle.drain().await.unwrap(); + source.shutdown().await.unwrap(); +} diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/ownership/recovery.rs b/crates/cellule-runtime/tests/runtime/lifecycle/ownership/recovery.rs index 5d898737..ee377e0e 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/ownership/recovery.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/ownership/recovery.rs @@ -2,15 +2,70 @@ use super::*; +use cellule_runtime::cell::actor::{AcquisitionObservation, AcquisitionObserver}; + +#[derive(Default)] +struct RecordedAcquisition { + before: std::sync::Mutex>, + restored: std::sync::Mutex>, + fail_at: u8, +} +impl AcquisitionObserver for RecordedAcquisition { + fn before_claim<'a>( + &'a self, + input: &'a cellule_runtime::control::Control, + ) -> AcquisitionObservation<'a> { + Box::pin(async move { + self.before.lock().unwrap().push(input.clone()); + if self.fail_at == 1 { + return Err(cellule_runtime::Error::Peer( + "injected lost recovery input recording reply", + )); + } + Ok(()) + }) + } + fn before_activation<'a>( + &'a self, + input: &'a cellule_runtime::control::Control, + restored: &'a cellule_runtime::control::Control, + ) -> AcquisitionObservation<'a> { + Box::pin(async move { + assert_eq!(self.before.lock().unwrap().last(), Some(input)); + assert_eq!(restored.state, ControlState::Recovering); + assert!(restored.recovery.is_none()); + self.restored.lock().unwrap().push(restored.clone()); + if self.fail_at == 2 { + return Err(cellule_runtime::Error::Peer( + "injected lost recovered position recording reply", + )); + } + Ok(()) + }) + } +} + #[tokio::test] async fn takeover_resumes_pinned_recovery_before_serving() { - recover_retained_tail(false).await; + recover_retained_tail(false, 0, false).await; } #[tokio::test] async fn recovery_seals_already_rooted_tail_without_an_empty_manifest() { - recover_retained_tail(true).await; + recover_retained_tail(true, 0, false).await; +} +#[tokio::test] +async fn recovery_recording_failure_before_cas_preserves_attached_tail_and_owner() { + recover_retained_tail(false, 1, false).await; +} +#[tokio::test] +async fn recovery_recording_failure_before_admission_keeps_materialized_root_recoverable() { + recover_retained_tail(false, 2, false).await; +} +#[tokio::test] +async fn exact_suffix_survives_an_interrupted_claim_without_acquisition_metadata() { + recover_retained_tail(false, 0, true).await; } -async fn recover_retained_tail(rooted: bool) { +async fn recover_retained_tail(rooted: bool, fail_at: u8, interrupted: bool) { let fixture = fixture_for(b"recovered-takeover"); let handle = activate(&fixture, 16 * 1024 * 1024).await; drop(handle); @@ -222,30 +277,174 @@ async fn recover_retained_tail(rooted: bool) { .await .unwrap(); assert_eq!(repeated.sealed, completed.sealed); - let attached = repeated.controls.into_iter().next().unwrap(); - let takeover = repeated.takeover; + let manifest_digest = repeated.sealed.log().recovery_manifest().unwrap(); + let original_inventory = manifests + .load_manifest(leader, repeated.sealed.log().epoch(), manifest_digest) + .await + .unwrap(); + assert_eq!(original_inventory.cells().len(), 1); + assert_eq!( + original_inventory.cells()[0].application, + fixture.target.application() + ); + assert_eq!(original_inventory.cells()[0].cell, fixture.target.cell_id()); + assert_eq!( + original_inventory.cells()[0].recovery, + *repeated.controls[0].value().recovery.as_ref().unwrap() + ); + let mut attached = repeated.controls.into_iter().next().unwrap(); + let original_owner = attached.value().clone(); + let mut takeover = repeated.takeover; + let mut successor = successor; + if interrupted { + // The real ownership CAS committed, then its acquisition was interrupted + // before metadata/materialization. Use the ordinary authority transition, + // never fabricate a successful acquisition record or prepared root. + attached = authority + .transition( + &attached, + attached + .value() + .takeover(Owner { + session: successor, + endpoint: "https://interrupted.internal".into(), + }) + .unwrap(), + Transition::Takeover, + ) + .await + .unwrap(); + assert!( + authority + .acquisition_record( + original_owner.cell, + original_owner.incarnation, + attached.value().epoch + ) + .await + .unwrap() + .is_none() + ); + let next = SessionId::from_bytes([84; 16]); + directory + .create( + cellule_runtime::node::NodeAdvertisement::sign( + NodeId::from_bytes(*next.as_bytes()), + next, + "https://later.internal".into(), + Digest::from_bytes([90; 32]), + Digest::from_bytes([94; 32]), + Digest::from_bytes([91; 32]), + Digest::from_bytes([92; 32]), + &ed25519_dalek::SigningKey::from_bytes(&[93; 32]), + 1, + 20_000, + 40_000, + vec![Digest::from_bytes([95; 32])], + vec![1], + cellule_runtime::node::NodeFailureDomain::default(), + cellule_runtime::node::NodeCapacity { + free_memory_bytes: 1, + free_disk_bytes: 1, + job_credits: 1, + ..cellule_runtime::node::NodeCapacity::default() + }, + ) + .unwrap(), + 20_000, + ) + .await + .unwrap(); + takeover = directory + .claim_expired_for_takeover(successor, next, 20_000) + .await + .unwrap(); + successor = next; + } let runtime = CellRuntime::new( SqlWorkerPool::new(1, 10).unwrap(), 16 * 1024 * 1024, successor, ) .unwrap(); + let input = attached.value().clone(); + let recorder = Arc::new(RecordedAcquisition { + fail_at, + ..RecordedAcquisition::default() + }); let restored = runtime - .takeover_restored( + .takeover_restored_observed( proof, fixture.replica.clone(), authority.clone(), attached, takeover, - manifests, + manifests.clone(), fixture._directory.path().join("recovered-takeover.sqlite"), Owner { session: successor, endpoint: "https://recovered-successor.internal:8081".into(), }, + Some(recorder.clone()), ) - .await - .unwrap(); + .await; + assert_eq!( + recorder.before.lock().unwrap().as_slice(), + std::slice::from_ref(&input) + ); + if fail_at != 0 { + let error = restored.err().unwrap(); + assert!(matches!(error, cellule_runtime::Error::Peer(message) + if message == if fail_at == 1 { "injected lost recovery input recording reply" } + else { "injected lost recovered position recording reply" })); + let current = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(runtime.stats().active_cells(), 0); + assert_eq!(runtime.stats().worker_jobs(), 0); + if fail_at == 1 { + assert_eq!(current.value(), &input); + assert!(current.value().recovery.is_some()); + assert!(recorder.restored.lock().unwrap().is_empty()); + assert!( + authority + .acquisition_record(input.cell, input.incarnation, input.epoch + 1) + .await + .unwrap() + .is_none() + ); + } else { + let recorded = recorder.restored.lock().unwrap()[0].clone(); + let retained = authority + .acquisition_record(input.cell, input.incarnation, recorded.epoch) + .await + .unwrap() + .unwrap(); + assert_eq!(retained.input(), &input); + assert_eq!(retained.materialized(), &recorded); + assert_eq!(current.value().state, ControlState::Idle); + assert!(current.value().recovery.is_none()); + assert_eq!(current.value().root, recorded.root); + assert_eq!( + current.value().root.as_ref().unwrap().commit_sequence, + predecessor.commit_sequence + 1 + ); + let root = current.value().ltx_root().unwrap(); + fixture.replica.open_root(&root).await.unwrap(); + } + runtime.shutdown().await.unwrap(); + return; + } + let restored = restored.unwrap(); + let recorded = recorder.restored.lock().unwrap()[0].clone(); + assert!(input.recovery.is_some() && recorded.recovery.is_none()); + assert_eq!(recorded.epoch, input.epoch + 1); + assert_eq!( + recorded.root.as_ref().unwrap().commit_sequence, + predecessor.commit_sequence + 1 + ); assert_eq!( restored @@ -265,6 +464,274 @@ async fn recover_retained_tail(rooted: bool) { .unwrap(); assert_eq!(serving.value().state, ControlState::Serving); assert!(serving.value().recovery.is_none()); + // The original owner remains discoverable after materialization erases + // its overlay and a successor serves the acknowledged recovered state. + let independent_authority = CellAuthority::new(fixture.layout.clone()); + let owner_history = independent_authority + .owner_history(fixture.target.cell_id(), if interrupted { 3 } else { 2 }) + .await + .unwrap(); + assert_eq!( + owner_history.owners().len(), + if interrupted { 3 } else { 2 } + ); + assert_eq!(owner_history.owners()[0], original_owner); + if interrupted { + assert_eq!(owner_history.owners()[1], input); + } + assert_eq!(owner_history.current(), serving.value()); + let acquisition = independent_authority + .acquisition_record(input.cell, input.incarnation, recorded.epoch) + .await + .unwrap() + .unwrap(); + assert_eq!(acquisition.input(), &input); + assert_eq!(acquisition.materialized(), &recorded); + let recovered_root = recorded.ltx_root().unwrap(); + independent_authority + .verify_root_prefix(predecessor, recovered_root, &fixture.replica, 16) + .await + .unwrap(); + let clock = now_ms(); + restored + .execute( + mutation_identity_window(47, clock, clock + 60_000), + Digest::from_bytes([47; 32]), + clock, + 64, + 64, + |transaction| { + transaction.execute("UPDATE counter SET value = value + 1", [])?; + Ok(HandlerOutcome::Success(Vec::new())) + }, + ) + .await + .unwrap(); + let advanced = authority + .load(input.cell) + .await + .unwrap() + .unwrap() + .value() + .ltx_root() + .unwrap(); + independent_authority + .verify_root_prefix(recovered_root, advanced, &fixture.replica, 16) + .await + .unwrap(); + + let required = &original_inventory.cells()[0]; + let verifier = CellRuntime::new( + SqlWorkerPool::new(1, 1).unwrap(), + 64 << 20, + SessionId::from_bytes([83; 16]), + ) + .unwrap(); + let native = verifier + .verify_recovered_prefix( + restored.catalog(), + &independent_authority, + fixture.replica.clone(), + required, + advanced, + 16, + ) + .await + .unwrap(); + assert_eq!(native.required().recovery, required.recovery); + assert_eq!(native.acquisition_epoch(), recorded.epoch); + assert_eq!(native.proof().prefix(), recovered_root); + assert_eq!(native.proof().root(), advanced); + assert_eq!(verifier.stats().retained_bytes(), 0); + // Every original manifest boundary is compared against the canonical input; + // a currently valid advanced graph cannot certify a different sealed suffix. + for change in 0..16 { + let mut wrong = cellule_runtime::recovery::manifest::PinnedRecoveryCell { + application: required.application, + cell: required.cell, + incarnation: required.incarnation, + cell_epoch: required.cell_epoch, + recovery: required.recovery.clone(), + }; + match change { + 0 => wrong.application = ApplicationId::from_bytes([99; 16]), + 1 => wrong.cell = cellule_runtime::identity::CellId::from_bytes([99; 32]), + 2 => wrong.incarnation = IncarnationId::from_bytes([99; 16]), + 3 => wrong.cell_epoch += 1, + 4 => wrong.recovery.leader_session = SessionId::from_bytes([99; 16]), + 5 => wrong.recovery.log_epoch += 1, + 6 => wrong.recovery.manifest_digest = Digest::from_bytes([99; 32]), + 7 => wrong.recovery.first_node_sequence += 1, + 8 => wrong.recovery.last_node_sequence += 1, + 9 => wrong.recovery.predecessor.digest = Digest::from_bytes([99; 32]), + 10 => wrong.recovery.predecessor.txid += 1, + 11 => wrong.recovery.predecessor.checksum ^= 1, + 12 => wrong.recovery.predecessor.commit_sequence += 1, + 13 => wrong.recovery.final_txid += 1, + 14 => wrong.recovery.final_checksum ^= 1, + _ => wrong.recovery.final_commit_sequence += 1, + } + assert!( + verifier + .verify_recovered_prefix( + restored.catalog(), + &independent_authority, + fixture.replica.clone(), + &wrong, + advanced, + 16, + ) + .await + .is_err() + ); + assert_eq!(verifier.stats().retained_bytes(), 0); + } + // Shared runtime admission refuses before origin I/O without lending the + // active writer's smaller node envelope to independent verification. + assert!(matches!( + runtime + .verify_recovered_prefix( + restored.catalog(), + &independent_authority, + fixture.replica.clone(), + required, + advanced, + 16, + ) + .await, + Err(cellule_runtime::Error::Capacity(_)) + )); + let acquisition_path = fixture.layout.acquisition_record_path( + input.cell.as_bytes(), + input.incarnation.as_bytes(), + recorded.epoch, + ); + let (acquisition_bytes, _) = fixture + .layout + .store() + .get_with_etag(&acquisition_path) + .await + .unwrap(); + fixture + .layout + .store() + .delete(&acquisition_path) + .await + .unwrap(); + assert!(matches!( + verifier + .verify_recovered_prefix( + restored.catalog(), + &independent_authority, + fixture.replica.clone(), + required, + advanced, + 16, + ) + .await, + Err(cellule_runtime::Error::AcquisitionHistoryIncomplete { .. }) + )); + assert_eq!(verifier.stats().retained_bytes(), 0); + fixture + .layout + .store() + .create_strict( + &acquisition_path, + Bytes::from_static(b"corrupt-acquisition"), + ) + .await + .unwrap(); + assert!(matches!( + verifier + .verify_recovered_prefix( + restored.catalog(), + &independent_authority, + fixture.replica.clone(), + required, + advanced, + 16, + ) + .await, + Err(cellule_runtime::Error::Control(_)) + )); + assert_eq!(verifier.stats().retained_bytes(), 0); + fixture + .layout + .store() + .delete(&acquisition_path) + .await + .unwrap(); + fixture + .layout + .store() + .create_strict(&acquisition_path, acquisition_bytes) + .await + .unwrap(); + let root_path = fixture.layout.incarnation_object_path( + &advanced.cell, + &advanced.incarnation, + &advanced.digest, + cellule_ltx::CellObjectKind::Root, + ); + let (root_bytes, _) = fixture + .layout + .store() + .get_with_etag(&root_path) + .await + .unwrap(); + fixture.layout.store().delete(&root_path).await.unwrap(); + assert!(matches!( + verifier + .verify_recovered_prefix( + restored.catalog(), + &independent_authority, + fixture.replica.clone(), + required, + advanced, + 16, + ) + .await, + Err(cellule_runtime::Error::Ltx(_)) + )); + assert_eq!(verifier.stats().retained_bytes(), 0); + fixture + .layout + .store() + .create_strict(&root_path, root_bytes) + .await + .unwrap(); + verifier + .verify_recovered_prefix( + restored.catalog(), + &independent_authority, + fixture.replica.clone(), + required, + advanced, + 16, + ) + .await + .unwrap(); + assert_eq!(verifier.stats().retained_bytes(), 0); + verifier.shutdown().await.unwrap(); + + // Materialization clears the control's overlay pointer. The canonical + // sealed manifest still retains the original scope after adapter restart. + let reconstructed = cellule_runtime::recovery::manifest::RecoveryManifestStore::new( + fixture.layout.clone(), + Limits::default(), + ); + let retained_inventory = reconstructed + .load_manifest(leader, 1, manifest_digest) + .await + .unwrap(); + assert_eq!( + retained_inventory.cells()[0].recovery, + original_inventory.cells()[0].recovery + ); + assert_eq!( + retained_inventory.cells()[0].incarnation, + serving.value().incarnation + ); assert_eq!( serving.value().root.as_ref().unwrap().commit_sequence, predecessor.commit_sequence + 1 @@ -317,6 +784,7 @@ async fn unchanged_unpublished_owner_is_taken_over_then_bootstrapped() { session, ) .unwrap(); + let original_input = stale.value().clone(); let restored = runtime .takeover_unpublished( proof, @@ -360,6 +828,23 @@ async fn unchanged_unpublished_owner_is_taken_over_then_bootstrapped() { .unwrap(); assert_eq!(owned.value().epoch, 2); assert_eq!(owned.value().owner.as_ref().unwrap().session, session); + let independent_authority = CellAuthority::new(fixture.layout.clone()); + let owner_history = independent_authority + .owner_history(fixture.target.cell_id(), 2) + .await + .unwrap(); + assert_eq!(owner_history.owners()[0], original_input); + assert!(owner_history.owners()[0].root.is_none()); + assert_eq!(owner_history.current(), owned.value()); + let acquisition = independent_authority + .acquisition_record(original_input.cell, original_input.incarnation, 2) + .await + .unwrap() + .unwrap(); + assert_eq!(acquisition.input(), &original_input); + assert!(acquisition.materialized().root.is_none()); + assert_eq!(acquisition.materialized().owner, owned.value().owner); + restored.drain().await.unwrap(); } #[tokio::test] diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/read_replica.rs b/crates/cellule-runtime/tests/runtime/lifecycle/read_replica.rs index 9efc36fa..a8def6b0 100644 --- a/crates/cellule-runtime/tests/runtime/lifecycle/read_replica.rs +++ b/crates/cellule-runtime/tests/runtime/lifecycle/read_replica.rs @@ -23,6 +23,7 @@ use cellule_runtime::registry::{ OperationDescriptor, Query, QueryContext, Registry, RegistryBuilder, RetainedCodeDescriptor, }; +mod closing; mod lifecycle; const MODULE: &str = "replica-counter"; diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/read_replica/closing.rs b/crates/cellule-runtime/tests/runtime/lifecycle/read_replica/closing.rs new file mode 100644 index 00000000..d11a3285 --- /dev/null +++ b/crates/cellule-runtime/tests/runtime/lifecycle/read_replica/closing.rs @@ -0,0 +1,432 @@ +//! Joined reader closure across retained clones and cancelled waiters. + +use super::*; +use std::time::Duration; + +struct ClosingFixture { + fixture: Fixture, + source: CellRuntime, + handle: CellHandle, + runtime: CellRuntime, + reader: CellReadReplica, + original: std::path::PathBuf, +} + +async fn opened(fixture: Fixture) -> ClosingFixture { + let owner = SessionId::from_bytes([44; 16]); + let source = CellRuntime::new_with_replica_host( + SqlWorkerPool::new(1, 1).unwrap(), + 8 << 20, + owner, + ReplicaHost::default().with_local_disk_budget(DiskBudget::new(8 << 30)), + ) + .unwrap(); + let handle = bootstrap_on(&source, &fixture, owner).await; + let registry = compiled_reader_registry(); + let directory = owner_directory(&fixture, owner, ®istry).await; + let runtime = CellRuntime::new_with_replica_host( + SqlWorkerPool::new(2, 512).unwrap(), + 8 << 20, + SessionId::from_bytes([45; 16]), + ReplicaHost::default().with_local_disk_budget(DiskBudget::new(8 << 30)), + ) + .unwrap(); + let original = fixture._directory.path().join("joined-reader.sqlite"); + let reader = CellReadReplica::open( + runtime.clone(), + registry, + CellAuthority::new(fixture.layout.clone()), + directory, + fixture.replica.clone(), + fixture.target.clone(), + &original, + ) + .await + .unwrap(); + ClosingFixture { + fixture, + source, + handle, + runtime, + reader, + original, + } +} + +fn empty(runtime: &CellRuntime) { + let stats = runtime.stats(); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.retained_bytes(), 0); + assert_eq!(stats.file_descriptors(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); + assert_eq!(stats.io_slots(), 0); +} + +async fn finish(fixture: ClosingFixture) { + fixture.reader.close_and_join().await; + fixture.runtime.shutdown().await.unwrap(); + fixture.handle.drain().await.unwrap(); + fixture.source.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn prepared_source_opens_the_journaled_root_after_newer_publication_and_still_obeys_fencing() +{ + let fixture = opened(fixture_for(b"reader-pinned-enrollment-source")).await; + let original = fixture.reader.close_and_join().await; + empty(&fixture.runtime); + let registry = compiled_reader_registry(); + let authority = CellAuthority::new(fixture.fixture.layout.clone()); + let directory = NodeDirectory::new( + fixture.fixture.layout.clone(), + Digest::from_bytes([9; 32]), + Digest::from_bytes([10; 32]), + registry.release_digest(), + ); + let source = CellReadReplica::prepare_source( + ®istry, + &authority, + &directory, + fixture.fixture.target.clone(), + ) + .await + .unwrap(); + assert_eq!(source.target(), &fixture.fixture.target); + assert_eq!(source.description().incarnation, original.incarnation); + assert_eq!(source.owner().session, SessionId::from_bytes([44; 16])); + assert_eq!(source.node(), NodeId::from_bytes([11; 16])); + assert_eq!(source.fleet(), Digest::from_bytes([9; 32])); + assert_eq!(source.root().commit_sequence, original.commit_sequence); + empty(&fixture.runtime); + fixture + .handle + .execute( + crate::support::fixtures::mutation_identity(49), + Digest::from_bytes([50; 32]), + now_ms(), + 64, + 64, + |tx| { + tx.execute("UPDATE counter SET value = 1", [])?; + Ok(HandlerOutcome::Success(Vec::new())) + }, + ) + .await + .unwrap(); + let current = authority.load(original.cell).await.unwrap().unwrap(); + assert_eq!(source.epoch(), current.value().epoch); + assert!(current.value().root.as_ref().unwrap().commit_sequence > source.root().commit_sequence); + let path = fixture + .fixture + ._directory + .path() + .join("pinned-source.sqlite"); + let reader = CellReadReplica::open_source( + fixture.runtime.clone(), + registry.clone(), + authority.clone(), + directory.clone(), + fixture.fixture.replica.clone(), + source.clone(), + &path, + ) + .await + .unwrap(); + let observed = reader.query::(None, 0).await.unwrap(); + assert_eq!(observed.receipt, original); + assert_eq!(observed.output, 0); + let next = fixture + .fixture + ._directory + .path() + .join("source-refreshed.sqlite"); + assert!(reader.refresh(&next).await.unwrap().commit_sequence > original.commit_sequence); + assert_eq!( + reader.query::(None, 0).await.unwrap().output, + 1 + ); + reader.close_and_join().await; + empty(&fixture.runtime); + fixture.handle.drain().await.unwrap(); + let rejected = fixture + .fixture + ._directory + .path() + .join("stale-source.sqlite"); + assert!(matches!( + CellReadReplica::open_source( + fixture.runtime.clone(), + registry.clone(), + authority.clone(), + directory.clone(), + fixture.fixture.replica.clone(), + source.clone(), + &rejected, + ) + .await, + Err(cellule_runtime::Error::Fenced) + )); + assert!(!rejected.exists()); + empty(&fixture.runtime); + fixture.runtime.node_admission().cordon().unwrap(); + assert!(matches!( + CellReadReplica::open_source( + fixture.runtime.clone(), + registry, + authority, + directory, + fixture.fixture.replica.clone(), + source, + &rejected, + ) + .await, + Err(cellule_runtime::Error::CellDraining) + )); + empty(&fixture.runtime); + fixture.runtime.shutdown().await.unwrap(); + // This case already performed canonical source release to test fencing. + fixture.source.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn joined_close_detaches_native_snapshots_from_retained_peer_clones() { + let fixture = opened(fixture_for(b"reader-close-retained-clones")).await; + let peer = fixture.reader.clone(); + let receipt = fixture.reader.receipt().await; + assert!(fixture.original.exists()); + let (first, duplicate) = tokio::join!(fixture.reader.close_and_join(), peer.close_and_join(),); + assert_eq!((first, duplicate), (receipt, receipt)); + assert_eq!(peer.receipt().await, receipt); + let joined = peer.lifecycle_observation().await; + assert!(joined.locally_joined()); + assert_eq!(joined.receipt(), receipt); + assert!(!fixture.original.exists()); + empty(&fixture.runtime); + assert!(matches!( + peer.readiness().await, + Err(cellule_runtime::Error::Fenced) + )); + assert!(matches!( + peer.query::(None, 0).await, + Err(cellule_runtime::Error::Fenced) + )); + let replacement = fixture + .fixture + ._directory + .path() + .join("closed-refresh.sqlite"); + assert!(matches!( + peer.refresh(&replacement).await, + Err(cellule_runtime::Error::Fenced) + )); + assert!(!replacement.exists()); + assert_eq!(peer.close_and_join().await, receipt); + empty(&fixture.runtime); + // The external capability remains alive through complete node drain. + finish(fixture).await; + assert_eq!(peer.receipt().await, receipt); + assert_eq!(peer.lifecycle_observation().await, joined); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn joined_close_keeps_old_native_query_owned_after_query_or_close_waiter_cancellation() { + for (drop_query, drop_close) in [(false, false), (true, false), (false, true), (true, true)] { + let fixture = opened(fixture_for(b"reader-close-owned-query")).await; + let pause = QueryPause::new(); + let token = pause.id; + let reader = fixture.reader.clone(); + let query = tokio::spawn(async move { reader.query::(None, token).await }); + pause.entered().await; + fixture + .handle + .execute( + crate::support::fixtures::mutation_identity(46), + Digest::from_bytes([47; 32]), + now_ms(), + 64, + 64, + |tx| { + tx.execute("UPDATE counter SET value = 1", [])?; + Ok(HandlerOutcome::Success(Vec::new())) + }, + ) + .await + .unwrap(); + let replacement = fixture.fixture._directory.path().join("new-reader.sqlite"); + let receipt = fixture.reader.refresh(&replacement).await.unwrap(); + let peer = fixture.reader.clone(); + let mut closing = Box::pin(fixture.reader.close_and_join()); + let first_pending = futures_util::poll!(closing.as_mut()).is_pending(); + let old_retained = fixture.original.exists(); + let new_detached = !replacement.exists(); + if drop_close { + drop(closing); + closing = Box::pin(peer.close_and_join()); + } + if drop_query { + query.abort(); + } + let mut sibling = Box::pin(peer.close_and_join()); + let second_pending = futures_util::poll!(sibling.as_mut()).is_pending(); + let retained = fixture.runtime.stats(); + let joining = peer.lifecycle_observation().await; + // Always unblock accepted native SQL before checking fixture assertions. + pause.release(); + if drop_query { + assert!(query.await.unwrap_err().is_cancelled()); + } else { + assert!(matches!( + query.await.unwrap(), + Err(cellule_runtime::Error::Fenced) + )); + } + assert_eq!(closing.await, receipt); + assert_eq!(sibling.await, receipt); + assert!(joining.admission_closed() && !joining.snapshot_attached()); + assert!(joining.retained_lifetimes() > 0 && !joining.locally_joined()); + assert_eq!(joining.receipt(), receipt); + assert!(peer.lifecycle_observation().await.locally_joined()); + assert!(first_pending && second_pending && old_retained && new_detached); + assert_eq!(retained.worker_jobs(), 1); + assert!(retained.resident_bytes() > 0); + assert!(!fixture.original.exists() && !replacement.exists()); + empty(&fixture.runtime); + finish(fixture).await; + assert_eq!(peer.receipt().await, receipt); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn joined_close_owns_cancelled_refresh_native_open_until_its_uninstalled_view_is_released() { + let store = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); + let runtime_slot = Arc::new(Mutex::new(None::)); + let slot = runtime_slot.clone(); + let armed = Arc::new(AtomicBool::new(false)); + let once = armed.clone(); + let pausing = store.clone(); + let fixture = opened(fixture_with_limits_and_store( + b"reader-close-cancelled-native-open", + Limits::default(), + Store::new(store.clone()).with_read_request_observer(Arc::new(move |kind| { + // Async root preparation has no SQL job. Arm only the actual VFS + // page fault inside the accepted native snapshot-open job. + if kind == cellule_store::StorageReadKind::Range + && slot + .lock() + .unwrap() + .as_ref() + .is_some_and(|runtime| runtime.stats().worker_jobs() == 1) + && !once.swap(true, Ordering::AcqRel) + { + pausing.arm_gets(); + } + })), + )) + .await; + *runtime_slot.lock().unwrap() = Some(fixture.runtime.clone()); + let original_receipt = fixture.reader.receipt().await; + fixture + .handle + .execute( + crate::support::fixtures::mutation_identity(46), + Digest::from_bytes([47; 32]), + now_ms(), + 64, + 64, + |tx| { + // Change SQLite's schema page, ensuring open faults a new page + // rather than reusing the unchanged first page from the old view. + tx.execute_batch( + "CREATE TABLE extra(value INTEGER); UPDATE counter SET value = 1", + )?; + Ok(HandlerOutcome::Success(Vec::new())) + }, + ) + .await + .unwrap(); + let destination = fixture + .fixture + ._directory + .path() + .join("cancelled-refresh.sqlite"); + let reader = fixture.reader.clone(); + let path = destination.clone(); + let refresh = tokio::spawn(async move { reader.refresh(&path).await }); + let entered = + tokio::time::timeout(Duration::from_secs(3), store.wait_until_get_blocked()).await; + if entered.is_err() { + store.release_gets(); + let _ = refresh.await; + finish(fixture).await; + panic!("native snapshot open did not reach its page fault"); + } + refresh.abort(); + assert!(refresh.await.unwrap_err().is_cancelled()); + let mut closing = Box::pin(fixture.reader.close_and_join()); + let pending = futures_util::poll!(closing.as_mut()).is_pending(); + let before = fixture.runtime.stats(); + let joining = fixture.reader.lifecycle_observation().await; + store.release_gets(); + assert_eq!(closing.await, original_receipt); + assert!(armed.load(Ordering::Acquire) && pending); + assert!(joining.admission_closed() && !joining.snapshot_attached()); + assert!(joining.retained_lifetimes() > 0 && !joining.locally_joined()); + assert_eq!(joining.receipt(), original_receipt); + assert!( + fixture + .reader + .lifecycle_observation() + .await + .locally_joined() + ); + assert_eq!(before.worker_jobs(), 1); + assert!(before.resident_bytes() > 0); + assert!(!destination.exists() && !fixture.original.exists()); + empty(&fixture.runtime); + runtime_slot.lock().unwrap().take(); + finish(fixture).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn joined_close_waits_for_accepted_authority_read_and_fences_its_late_readiness_reply() { + let store = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); + let fixture = opened(fixture_with_limits_and_store( + b"reader-close-stalled-observation", + Limits::default(), + Store::new(store.clone()), + )) + .await; + let receipt = fixture.reader.receipt().await; + store.arm_gets(); + let reader = fixture.reader.clone(); + let readiness = tokio::spawn(async move { reader.readiness().await }); + tokio::time::timeout( + std::time::Duration::from_secs(3), + store.wait_until_get_blocked(), + ) + .await + .unwrap(); + let mut closing = Box::pin(fixture.reader.close_and_join()); + let pending = futures_util::poll!(closing.as_mut()).is_pending(); + let retained = fixture.original.exists(); + let joining = fixture.reader.lifecycle_observation().await; + store.release_gets(); + assert!(matches!( + readiness.await.unwrap(), + Err(cellule_runtime::Error::Fenced) + )); + assert_eq!(closing.await, receipt); + assert!(pending && retained); + assert!(joining.admission_closed() && !joining.snapshot_attached()); + assert!(joining.retained_lifetimes() > 0 && !joining.locally_joined()); + assert!( + fixture + .reader + .lifecycle_observation() + .await + .locally_joined() + ); + empty(&fixture.runtime); + finish(fixture).await; +} diff --git a/crates/cellule-runtime/tests/runtime/lifecycle/receiver.rs b/crates/cellule-runtime/tests/runtime/lifecycle/receiver.rs new file mode 100644 index 00000000..f3968acb --- /dev/null +++ b/crates/cellule-runtime/tests/runtime/lifecycle/receiver.rs @@ -0,0 +1,727 @@ +//! Real receiver credit, canonical acquisition, cancellation and shutdown. + +use super::*; +use cellule_runtime::cell::actor::{CellHandle, PreparedCellReceiver, ReceiverState}; +use cellule_runtime::fleet::operations::{ + AttemptId, MAX_RECORD_BYTES, MoveAttemptSpec, OperationError, OperationId, +}; + +const RECEIVER: SessionId = SessionId::from_bytes([121; 16]); + +fn receiver_runtime() -> CellRuntime { + runtime_with_capacity(64 << 20, 8, 8 << 30) +} + +fn runtime_with_capacity(memory: usize, cells: usize, disk: u64) -> CellRuntime { + let pool = SqlWorkerPool::new(1, cells) + .unwrap() + .with_native_memory_limit(memory) + .unwrap(); + CellRuntime::new_with_replica_host( + pool, + 64 << 20, + RECEIVER, + ReplicaHost::default().with_local_disk_budget(DiskBudget::new(disk)), + ) + .unwrap() +} + +async fn attempt(source: &CellRuntime, fixture: &Fixture, sequence: u64) -> MoveAttemptSpec { + let owner = super::inventory::stable_owner(source).await; + assert_eq!(owner.target, fixture.target); + MoveAttemptSpec { + id: AttemptId { + operation: OperationId::from_bytes([122; 16]).unwrap(), + sequence, + }, + target: owner.target, + incarnation: owner.incarnation, + source_node: NodeId::from_bytes([123; 16]), + source: source.fleet_cells_page(None, 1).await.unwrap().session(), + generation: owner.generation, + source_epoch: owner.position.unwrap().epoch, + destination_node: NodeId::from_bytes([124; 16]), + destination: RECEIVER, + cost: owner.cost.unwrap(), + snapshot_digest: Digest::from_bytes([125; 32]), + deadline_ms: now_ms() + 60_000, + } +} + +fn prepare( + runtime: &CellRuntime, + spec: &MoveAttemptSpec, + handle: &CellHandle, + fixture: &Fixture, +) -> cellule_runtime::Result { + runtime.prepare_receiver( + spec.clone(), + handle.catalog().clone(), + fixture.replica.clone(), + fixture._directory.path().join("receiver.sqlite"), + spec.deadline_ms, + now_ms(), + ) +} + +fn owner() -> Owner { + Owner { + session: RECEIVER, + endpoint: "https://receiver.internal:8081".into(), + } +} + +async fn counter(handle: &CellHandle) -> i64 { + let bytes = handle + .query(64, 64, |connection| { + Ok(connection + .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))? + .to_be_bytes() + .to_vec()) + }) + .await + .unwrap(); + i64::from_be_bytes(bytes.try_into().unwrap()) +} + +async fn wait_state(prepared: &PreparedCellReceiver, expected: ReceiverState) { + tokio::time::timeout(std::time::Duration::from_secs(10), async { + while prepared.state().unwrap() != expected { + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + }) + .await + .unwrap(); +} + +async fn retire(runtime: &CellRuntime, prepared: &PreparedCellReceiver) { + tokio::time::timeout(std::time::Duration::from_secs(5), async { + loop { + match runtime.retire_prepared_receiver(prepared).await { + Ok(()) => return, + Err(cellule_runtime::Error::FleetOperation(error)) + if matches!(*error, OperationError::Busy) => + { + tokio::task::yield_now().await + } + Err(error) => panic!("retirement failed: {error}"), + } + } + }) + .await + .unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn prepared_credit_transfers_into_exact_root_sql_and_one_fenced_writer() { + let fixture = fixture_for(b"prepared-receiver-root"); + let (source, handle, _) = activate_runtime(&fixture, 64 << 20).await; + let clock = now_ms(); + let acknowledged = handle + .execute( + mutation_identity_window(126, clock, clock + 60_000), + Digest::from_bytes([126; 32]), + clock, + 64, + 64, + |transaction| { + transaction.execute("UPDATE counter SET value = 77", [])?; + Ok(HandlerOutcome::Success(Vec::new())) + }, + ) + .await + .unwrap(); + let spec = attempt(&source, &fixture, 1).await; + let receiver = receiver_runtime(); + let prepared = prepare(&receiver, &spec, &handle, &fixture).unwrap(); + let reserved = receiver.stats(); + assert_eq!(reserved.active_cells(), 1); + assert_eq!(reserved.resident_bytes(), spec.cost.memory_bytes as usize); + assert_eq!( + reserved.file_descriptors(), + spec.cost.file_descriptors as usize + ); + assert_eq!(reserved.worker_jobs(), 1); + assert_eq!(reserved.local_disk_reserved_bytes(), spec.cost.disk_bytes); + assert_eq!(reserved.retained_bytes(), MAX_RECORD_BYTES as usize); + let duplicate = prepare(&receiver, &spec, &handle, &fixture).unwrap(); + assert_eq!(receiver.stats(), reserved); + assert_eq!(duplicate.spec().unwrap(), spec); + drop(prepared); + let prepared = receiver.prepared_receiver(spec.id).unwrap().unwrap(); + assert_eq!(counter(&handle).await, 77); + let released = source + .release_idle_cell_at( + spec.target.cell_id(), + spec.source, + spec.generation, + spec.incarnation, + spec.source_epoch, + ) + .await + .unwrap(); + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert!(idle.value().root.as_ref().unwrap().commit_sequence >= acknowledged.commit_sequence()); + assert_eq!(Some(&released.root), idle.value().root.as_ref()); + assert_eq!(released.epoch, spec.source_epoch); + let root = idle.value().root.clone(); + let original_input = idle.value().clone(); + let activated = receiver + .activate_prepared_receiver(&prepared, authority.clone(), idle, owner(), now_ms()) + .await + .unwrap(); + assert_eq!(prepared.state().unwrap(), ReceiverState::Activated); + assert_eq!(counter(&activated).await, 77); + assert_eq!( + activated + .resolve( + mutation_identity_window(126, clock, clock + 60_000), + Digest::from_bytes([126; 32]), + now_ms(), + 64, + ) + .await + .unwrap(), + Resolution::Committed(acknowledged) + ); + let serving = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(serving.value().state, ControlState::Serving); + assert_eq!(serving.value().owner.as_ref().unwrap().session, RECEIVER); + assert_eq!(serving.value().epoch, spec.source_epoch + 1); + assert_eq!(serving.value().root, root); + let independent = CellAuthority::new(fixture.layout.clone()); + let acquisition = independent + .acquisition_record( + fixture.target.cell_id(), + spec.incarnation, + serving.value().epoch, + ) + .await + .unwrap() + .unwrap(); + assert_eq!(acquisition.input(), &original_input); + assert_eq!(acquisition.materialized().root, root); + assert_eq!(acquisition.materialized().owner, serving.value().owner); + assert_eq!(acquisition.materialized().state, ControlState::Recovering); + + let after = receiver.stats(); + assert_eq!(after.active_cells(), 1); + assert_eq!(after.worker_jobs(), 0); + assert_eq!(after.resident_bytes(), reserved.resident_bytes()); + assert_eq!(after.file_descriptors(), reserved.file_descriptors()); + assert!( + after.local_disk_reserved_bytes() > 0 + && after.local_disk_reserved_bytes() < reserved.local_disk_reserved_bytes() + ); + assert!( + source + .local_handle(handle.catalog().clone(), &serving) + .await + .unwrap() + .is_none() + ); + retire(&receiver, &prepared).await; + assert_eq!(receiver.stats().retained_bytes(), 0); + assert_eq!(receiver.stats().active_cells(), 1); + activated.drain().await.unwrap(); + receiver.shutdown().await.unwrap(); + assert_eq!(receiver.stats().active_cells(), 0); + assert_eq!(receiver.stats().resident_bytes(), 0); + assert_eq!(receiver.stats().file_descriptors(), 0); + assert_eq!(receiver.stats().local_disk_reserved_bytes(), 0); + source.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn expiry_does_not_free_credit_and_cancellation_never_reprepares_same_attempt() { + let fixture = fixture_for(b"prepared-receiver-cancel"); + let (source, handle, _) = activate_runtime(&fixture, 64 << 20).await; + let spec = attempt(&source, &fixture, 1).await; + let receiver = receiver_runtime(); + let prepared = prepare(&receiver, &spec, &handle, &fixture).unwrap(); + let retained = prepared.clone(); + drop(prepared); + assert_eq!( + receiver.stats().local_disk_reserved_bytes(), + spec.cost.disk_bytes + ); + handle.drain().await.unwrap(); + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert!( + matches!(receiver.activate_prepared_receiver(&retained, authority.clone(), idle.clone(), owner(), spec.deadline_ms).await, + Err(cellule_runtime::Error::FleetOperation(error)) if matches!(*error, OperationError::Deadline)) + ); + assert_eq!(retained.state().unwrap(), ReceiverState::Prepared); + assert_eq!( + receiver.stats().local_disk_reserved_bytes(), + spec.cost.disk_bytes + ); + receiver.cancel_prepared_receiver(&retained).unwrap(); + receiver.cancel_prepared_receiver(&retained).unwrap(); + assert_eq!(receiver.stats().local_disk_reserved_bytes(), 0); + assert_eq!(receiver.stats().worker_jobs(), 0); + assert_eq!(receiver.stats().active_cells(), 0); + assert_eq!(receiver.stats().retained_bytes(), MAX_RECORD_BYTES as usize); + let duplicate = prepare(&receiver, &spec, &handle, &fixture).unwrap(); + assert_eq!(duplicate.state().unwrap(), ReceiverState::Cancelled); + let mut changed = spec.clone(); + changed.snapshot_digest = Digest::from_bytes([127; 32]); + assert!( + matches!(prepare(&receiver, &changed, &handle, &fixture), Err(cellule_runtime::Error::FleetOperation(error)) if matches!(*error, OperationError::Conflict)) + ); + assert_eq!( + authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap() + .value(), + idle.value() + ); + retire(&receiver, &retained).await; + assert!(matches!( + retained.state(), + Err(cellule_runtime::Error::RuntimeClosed) + )); + receiver.shutdown().await.unwrap(); + source.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn preparation_refusal_leaves_source_serving_and_returns_partial_charges() { + let fixture = fixture_for(b"prepared-receiver-refusal"); + let (source, handle, _) = activate_runtime(&fixture, 64 << 20).await; + let spec = attempt(&source, &fixture, 1).await; + // Fail resident memory, file descriptors, then disk after the earlier tokens + // were admitted. Every refusal must undo its own partial preparation. + for receiver in [ + runtime_with_capacity(64 << 10, 8, 8 << 30), + runtime_with_capacity(64 << 20, 1, 8 << 30), + runtime_with_capacity(64 << 20, 8, 1), + ] { + assert!(matches!( + prepare(&receiver, &spec, &handle, &fixture), + Err(cellule_runtime::Error::Capacity(_) | cellule_runtime::Error::Ltx(_)) + )); + assert_eq!(receiver.stats().active_cells(), 0); + assert_eq!(receiver.stats().worker_jobs(), 0); + assert_eq!(receiver.stats().resident_bytes(), 0); + assert_eq!(receiver.stats().file_descriptors(), 0); + assert_eq!(receiver.stats().retained_bytes(), 0); + assert_eq!(receiver.stats().local_disk_reserved_bytes(), 0); + assert!(receiver.prepared_receiver(spec.id).unwrap().is_none()); + assert_eq!(counter(&handle).await, 0); + receiver.shutdown().await.unwrap(); + } + let receiver = receiver_runtime(); + let mut wrong = spec.clone(); + wrong.destination = SessionId::from_bytes([128; 16]); + assert!(matches!( + prepare(&receiver, &wrong, &handle, &fixture), + Err(cellule_runtime::Error::Fenced) + )); + let mut underdeclared = spec.clone(); + underdeclared.cost.disk_bytes -= 1; + assert!(matches!( + prepare(&receiver, &underdeclared, &handle, &fixture), + Err(cellule_runtime::Error::Capacity(_)) + )); + let prepared = prepare(&receiver, &spec, &handle, &fixture).unwrap(); + // Another attempt cannot steal the exact worker while its prepared credit + // is retained, even though node memory, disk and Cell slots remain free. + let mut competing = spec.clone(); + competing.id.sequence = 2; + assert!(matches!( + prepare(&receiver, &competing, &handle, &fixture), + Err(cellule_runtime::Error::Capacity("incoming Cell worker")) + )); + assert_eq!( + receiver.stats().local_disk_reserved_bytes(), + spec.cost.disk_bytes + ); + assert_eq!(receiver.stats().active_cells(), 1); + assert_eq!(receiver.stats().retained_bytes(), MAX_RECORD_BYTES as usize); + receiver.cancel_prepared_receiver(&prepared).unwrap(); + retire(&receiver, &prepared).await; + receiver.shutdown().await.unwrap(); + handle.drain().await.unwrap(); + source.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn mismatched_replica_scope_is_refused_before_any_receiver_admission() { + let fixture = fixture_for(b"prepared-receiver-scope"); + let (source, handle, _) = activate_runtime(&fixture, 64 << 20).await; + let spec = attempt(&source, &fixture, 1).await; + let receiver = receiver_runtime(); + for (cell, incarnation) in [ + ([129; 32], *spec.incarnation.as_bytes()), + (*spec.target.cell_id().as_bytes(), [130; 16]), + ([129; 32], [130; 16]), + ] { + let replica = CellReplica::new( + fixture.layout.clone(), + cell, + incarnation, + fixture.replica.limits(), + ) + .unwrap(); + assert_eq!(replica.scope(), (cell, incarnation)); + assert!(matches!( + receiver.prepare_receiver( + spec.clone(), + handle.catalog().clone(), + replica, + fixture._directory.path().join("receiver.sqlite"), + spec.deadline_ms, + now_ms(), + ), + Err(cellule_runtime::Error::Fenced) + )); + assert_eq!(receiver.stats().active_cells(), 0); + assert_eq!(receiver.stats().worker_jobs(), 0); + assert_eq!(receiver.stats().resident_bytes(), 0); + assert_eq!(receiver.stats().file_descriptors(), 0); + assert_eq!(receiver.stats().retained_bytes(), 0); + assert_eq!(receiver.stats().local_disk_reserved_bytes(), 0); + assert!(receiver.prepared_receiver(spec.id).unwrap().is_none()); + } + assert_eq!(counter(&handle).await, 0); + receiver.shutdown().await.unwrap(); + source.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn unused_preparation_closes_on_shutdown_despite_retained_caller_handles() { + let fixture = fixture_for(b"prepared-receiver-shutdown"); + let (source, handle, _) = activate_runtime(&fixture, 64 << 20).await; + let spec = attempt(&source, &fixture, 1).await; + let receiver = receiver_runtime(); + let prepared = prepare(&receiver, &spec, &handle, &fixture).unwrap(); + let retained = prepared.clone(); + tokio::time::timeout(std::time::Duration::from_secs(5), receiver.shutdown()) + .await + .unwrap() + .unwrap(); + let stats = receiver.stats(); + assert_eq!(stats.active_cells(), 0); + assert_eq!(stats.resident_bytes(), 0); + assert_eq!(stats.file_descriptors(), 0); + assert_eq!(stats.worker_jobs(), 0); + assert_eq!(stats.local_disk_reserved_bytes(), 0); + assert_eq!(stats.retained_bytes(), 0); + assert!(matches!( + retained.state(), + Err(cellule_runtime::Error::RuntimeClosed) + )); + assert_eq!(counter(&handle).await, 0); + handle.drain().await.unwrap(); + source.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn failed_restore_returns_credit_and_canonically_releases_its_acquired_epoch() { + let fixture = fixture_for(b"prepared-receiver-restore-failure"); + let (source, handle, _) = activate_runtime(&fixture, 64 << 20).await; + let spec = attempt(&source, &fixture, 1).await; + let receiver = receiver_runtime(); + let destination = fixture + ._directory + .path() + .join("directory-is-not-a-database"); + std::fs::create_dir(&destination).unwrap(); + let prepared = receiver + .prepare_receiver( + spec.clone(), + handle.catalog().clone(), + fixture.replica.clone(), + destination, + spec.deadline_ms, + now_ms(), + ) + .unwrap(); + handle.drain().await.unwrap(); + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + let root = idle.value().root.clone(); + assert!(matches!( + receiver + .activate_prepared_receiver(&prepared, authority.clone(), idle, owner(), now_ms(),) + .await, + Err(cellule_runtime::Error::Ltx(_)) + )); + assert_eq!(prepared.state().unwrap(), ReceiverState::Failed); + let after = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(after.value().epoch, spec.source_epoch + 1); + assert_eq!(after.value().state, ControlState::Idle); + assert!(after.value().owner.is_none()); + assert_eq!(after.value().root, root); + assert_eq!(receiver.stats().active_cells(), 0); + assert_eq!(receiver.stats().resident_bytes(), 0); + assert_eq!(receiver.stats().file_descriptors(), 0); + assert_eq!(receiver.stats().worker_jobs(), 0); + assert_eq!(receiver.stats().local_disk_reserved_bytes(), 0); + retire(&receiver, &prepared).await; + receiver.shutdown().await.unwrap(); + source.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn prepared_receive_uses_current_root_after_an_intervening_canonical_owner() { + let fixture = fixture_for(b"prepared-receiver-current-idle"); + let (source, handle, _) = activate_runtime(&fixture, 64 << 20).await; + let spec = attempt(&source, &fixture, 1).await; + let receiver = receiver_runtime(); + let prepared = prepare(&receiver, &spec, &handle, &fixture).unwrap(); + handle.drain().await.unwrap(); + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + let other_session = SessionId::from_bytes([129; 16]); + let other = CellRuntime::new_with_replica_host( + SqlWorkerPool::new(1, 2).unwrap(), + 64 << 20, + other_session, + ReplicaHost::default().with_local_disk_budget(DiskBudget::new(8 << 30)), + ) + .unwrap(); + let other_handle = other + .acquire_idle_restored( + handle.catalog().clone(), + fixture.replica.clone(), + authority.clone(), + idle, + fixture._directory.path().join("other.sqlite"), + Owner { + session: other_session, + endpoint: "https://other.internal:8081".into(), + }, + ) + .await + .unwrap(); + let clock = now_ms(); + other_handle + .execute( + mutation_identity_window(130, clock, clock + 60_000), + Digest::from_bytes([130; 32]), + clock, + 64, + 64, + |transaction| { + transaction.execute("UPDATE counter SET value = 98", [])?; + Ok(HandlerOutcome::Success(Vec::new())) + }, + ) + .await + .unwrap(); + other_handle.drain().await.unwrap(); + let current = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(current.value().epoch, spec.source_epoch + 1); + let latest_root = current.value().root.clone(); + let activated = receiver + .activate_prepared_receiver(&prepared, authority.clone(), current, owner(), now_ms()) + .await + .unwrap(); + assert_eq!(counter(&activated).await, 98); + let serving = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(serving.value().root, latest_root); + assert_eq!(serving.value().epoch, spec.source_epoch + 2); + receiver.shutdown().await.unwrap(); + other.shutdown().await.unwrap(); + source.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn lost_activation_waiter_and_cas_reply_leave_one_inspectable_serving_actor() { + let store = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); + let fixture = fixture_with_limits_and_store( + b"prepared-receiver-lost-reply", + Limits::default(), + Store::new(store.clone()), + ); + let (source, handle, _) = activate_runtime(&fixture, 64 << 20).await; + let spec = attempt(&source, &fixture, 1).await; + let receiver = receiver_runtime(); + let prepared = prepare(&receiver, &spec, &handle, &fixture).unwrap(); + handle.drain().await.unwrap(); + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + let original_input = idle.value().clone(); + store.arm_next_update(); + store.lose_next_update_response(); + store.acquisition_fault.store(2, Ordering::Release); + let activation = { + let receiver = receiver.clone(); + let prepared = prepared.clone(); + let authority = authority.clone(); + let idle = idle.clone(); + tokio::spawn(async move { + receiver + .activate_prepared_receiver(&prepared, authority, idle, owner(), now_ms()) + .await + }) + }; + tokio::time::timeout( + std::time::Duration::from_secs(5), + store.wait_until_blocked(), + ) + .await + .unwrap(); + activation.abort(); + assert!(activation.await.err().unwrap().is_cancelled()); + assert_eq!(prepared.state().unwrap(), ReceiverState::Activating); + assert!( + matches!(receiver.cancel_prepared_receiver(&prepared), Err(cellule_runtime::Error::FleetOperation(error)) if matches!(*error, OperationError::Busy)) + ); + assert!( + matches!(receiver.activate_prepared_receiver(&prepared, authority.clone(), idle, owner(), now_ms()).await, Err(cellule_runtime::Error::FleetOperation(error)) if matches!(*error, OperationError::Busy)) + ); + assert_eq!( + receiver.stats().local_disk_reserved_bytes(), + spec.cost.disk_bytes + ); + store.release(); + wait_state(&prepared, ReceiverState::Activated).await; + assert!(store.lost_update_response_consumed()); + let serving = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + let activated = receiver + .local_handle(handle.catalog().clone(), &serving) + .await + .unwrap() + .unwrap(); + assert_eq!(counter(&activated).await, 0); + assert_eq!(serving.value().owner.as_ref().unwrap().session, RECEIVER); + assert_eq!(store.acquisition_fault.load(Ordering::Acquire), 0); + let record = CellAuthority::new(fixture.layout.clone()) + .acquisition_record( + original_input.cell, + original_input.incarnation, + serving.value().epoch, + ) + .await + .unwrap() + .unwrap(); + assert_eq!(record.input(), &original_input); + assert_eq!(record.materialized().root, serving.value().root); + assert_eq!(record.materialized().state, ControlState::Recovering); + assert_eq!(receiver.stats().active_cells(), 1); + receiver.shutdown().await.unwrap(); + assert_eq!(receiver.stats().local_disk_reserved_bytes(), 0); + source.shutdown().await.unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn shutdown_joins_accepted_claim_and_rolls_it_back_before_worker_close() { + let store = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); + let fixture = fixture_with_limits_and_store( + b"prepared-receiver-join", + Limits::default(), + Store::new(store.clone()), + ); + let (source, handle, _) = activate_runtime(&fixture, 64 << 20).await; + let spec = attempt(&source, &fixture, 1).await; + let receiver = receiver_runtime(); + let prepared = prepare(&receiver, &spec, &handle, &fixture).unwrap(); + handle.drain().await.unwrap(); + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + let root = idle.value().root.clone(); + store.arm_next_update(); + let activation = { + let receiver = receiver.clone(); + let prepared = prepared.clone(); + let authority = authority.clone(); + tokio::spawn(async move { + receiver + .activate_prepared_receiver(&prepared, authority, idle, owner(), now_ms()) + .await + }) + }; + tokio::time::timeout( + std::time::Duration::from_secs(5), + store.wait_until_blocked(), + ) + .await + .unwrap(); + let shutdown = { + let receiver = receiver.clone(); + tokio::spawn(async move { receiver.shutdown().await }) + }; + tokio::time::timeout(std::time::Duration::from_secs(5), async { + while receiver.is_acquiring() { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + assert!(!shutdown.is_finished()); + assert_eq!(receiver.stats().active_cells(), 1); + store.release(); + assert!(matches!( + activation.await.unwrap(), + Err(cellule_runtime::Error::RuntimeClosed | cellule_runtime::Error::CellDraining) + )); + tokio::time::timeout(std::time::Duration::from_secs(10), shutdown) + .await + .unwrap() + .unwrap() + .unwrap(); + let current = authority + .load(fixture.target.cell_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(current.value().state, ControlState::Idle); + assert!(current.value().owner.is_none()); + assert_eq!(current.value().root, root); + assert_eq!(receiver.stats().active_cells(), 0); + assert_eq!(receiver.stats().worker_jobs(), 0); + assert_eq!(receiver.stats().local_disk_reserved_bytes(), 0); + assert_eq!(receiver.stats().retained_bytes(), 0); + source.shutdown().await.unwrap(); +} diff --git a/crates/cellule-runtime/tests/runtime/migration.rs b/crates/cellule-runtime/tests/runtime/migration.rs index 3d3f1bc5..971e3476 100644 --- a/crates/cellule-runtime/tests/runtime/migration.rs +++ b/crates/cellule-runtime/tests/runtime/migration.rs @@ -23,7 +23,7 @@ use cellule_runtime::identity::{ use cellule_runtime::identity::{IncarnationId, NodeId}; use cellule_runtime::node::durability::{NodeDurability, NodeLogAuthority}; use cellule_runtime::node::lease::NodeLeaseGuard; -use cellule_runtime::node::log::{DurabilityGate, NodeLogRotationBarrier}; +use cellule_runtime::node::log::DurabilityGate; use cellule_runtime::node::log_shipper::NodeLogShipper; use cellule_runtime::node::log_transport::{LocalFollowerTransport, NodeLogTransport}; use cellule_runtime::peer::{ @@ -70,7 +70,7 @@ impl NodeLogAuthority for MigrationNodeAuthority { fn close<'a>( &'a self, - _barrier: &'a NodeLogRotationBarrier, + _retirement: &'a cellule_runtime::node::log::NodeLogRetirementObservation, ) -> futures_util::future::BoxFuture<'a, cellule_runtime::Result<()>> { Box::pin(async { Ok(()) }) } diff --git a/docs/fleet-control-plane-audit.md b/docs/fleet-control-plane-audit.md new file mode 100644 index 00000000..5a0abb4b --- /dev/null +++ b/docs/fleet-control-plane-audit.md @@ -0,0 +1,87 @@ +# Cellule control plane production design audit + +Audit date: October 1, 2026, America/Vancouver. +Framework baseline inspected: `58721227e56dac3ebcda6b74b4ed2a514f8cc42b`. +Original plan SHA256: `4e31ac8b2b76aff1a454c1ba43d09c599d947e8f7346b4d67e74e44a0ab5816b`. + +The original design provided a sound direction for fenced execution, but did +not yet specify a complete production product with low application-team effort +or a scalable execution path. The [revised plan](fleet-control-plane-plan.md) +resolves the identified design omissions with explicit contracts, source work, +dependencies, supported deployment boundaries and measurable release gates. +These are design resolutions. Implementation and production qualification +remain outstanding. + +The review covered architecture, correctness of proposed coordination, +application integration, scale and resource bounds, lifecycle automation, +identity, operational recovery, API/UI behavior, rollout and release evidence. +It inspected the existing journal, roster, profile and reconciliation contracts. +It did not execute runtime tests, benchmark providers, certify security of +unimplemented endpoints, or audit unrelated application code. Concurrent native +reader work was treated as unqualified until its own evidence is recorded. + +## Findings and required resolution + +Evidence citing an original section refers to the plan fingerprint above; +those sections have now been revised. Framework links identify inspected source. +P0 means a safety prerequisite for the proposed production capability; P1 means +a production delivery or scale blocker. Effort describes implementation, not +the documentation edit. All findings below have high confidence as design gaps. + +| ID | Priority | Original evidence and impact | Resolution and package | Effort and implementation risk | +| --- | --- | --- | --- | --- | +| A01 | P1 | Delivery contract and application layout offered a standalone reference example. No maintained distribution or support envelope completed the requested production product. | Supported binary/image, CLI, UI, worker SDK, deployment profiles and compatibility manifest; CP1/CP13. | L; medium, new product packaging and compatibility surface. | +| A02 | P1 | Architecture assigned enrollment, signing, observer, transport, journal and endpoint construction to embedding applications. Every team would repeat substantial fleet engineering. | Shipped SDK owns those facilities; teams supply existing providers and business hooks. One-engineer-day onboarding gate; CP1/CP12. | L; high, lifecycle integration must preserve one runtime and accepted work. | +| A03 | P1 | Controller ownership used one lease per fleet. [FleetProfile](../crates/cellule-runtime/src/fleet/operations/records.rs) and [operation limits](../crates/cellule-runtime/src/fleet/operations/mod.rs) cap the whole current scope at two moves and 8 GiB. This supplied no path to large-fleet progress. | Versioned execution partitions, per-partition leases and bounded fair scheduling under atomic aggregate permits; CP0/CP4/CP5. | L; high, partitioning changes authorization and budgeting. | +| A04 | P1 | Observation retained all terminal enrollments under the [10,000-row roster cap](../crates/cellule-host/src/fleet/roster/mod.rs). Repeated boot/role churn can eventually prevent collection regardless of current live fleet size. | Active closure indexes, immutable terminal archive/exclusion roots and exact replay lookup; ten-million-history qualification; CP3/CP12. | L; high, compaction must never recreate Pending from absence. | +| A05 | P1 | Whole-head/registry comparison and full page traversal in [roster collection](../crates/cellule-host/src/fleet/roster/scan.rs) were the only observer design. Continuous unrelated changes can repeatedly invalidate a fleet scan. | Consistent immutable snapshots, transactional deltas, checkpoint reconciliation and action-local revalidation; CP3. | L; high, equivalence of completeness and finalization barriers must be proved. | +| A06 | P1 | Management Cells and events specified one fleet read-model Cell and fleet-wide ordered sequence. These introduce serial hot paths and duplicate polling pressure without sizing evidence. | Bucketed Cell projections, per-partition ordering/cursors, shared inventory and bounded merge/read views; CP3/CP10. | M; medium, snapshot/split/replay correctness. | +| A07 | P0 | Partition parallelism was absent, so the original plan contained no shared budget transaction or cross-partition movement/reassignment contract. Adding more controllers without these would break global resource bounds. | One source-owned attempt, atomic crossing receiver reservation, assignment epochs and aggregate permits; CP4. | L; high, crash and stale-envelope races. | +| A08 | P1 | Overload/no-capacity handling ended with a request to add capacity. No supplied capacity reconciliation or safe provider removal protocol reduced that operator burden. | Desired pool capacity, qualified Kubernetes adapter, enrollment-based readiness and exact-instance drain/removal; CP8. | L; high, provider actions can remove the wrong instance without preconditions. | +| A09 | P0 | Controller maintenance had a one-node rule, but there was no fleet/failure-domain unavailable-capacity ledger covering concurrent drains and unexpected failures. | Atomic disruption permits with minimum service/role redundancy and actual unavailable capacity; CP4/CP8/CP9. | L; high, stale redundancy evidence or independent drains could violate availability. | +| A10 | P1 | API exposed individual requests but lacked a declarative fleet specification, bulk selector/cursor, rolling upgrade, retirement and precise post-acceptance cancellation workflow. | FleetSpec, dry-run/explain, durable parent operations, canary rollout and safe control commands; CP6/CP8/CP9. | L; medium, request replay and partial completion semantics. | +| A11 | P0 | Journal/node transport ownership did not specify how workers access authoritative transactions without broad database credentials. Making the new gateway depend on the controller's own completed enrollment would also create a bootstrap cycle. | Scoped gateway on all replicas, SDK remote journal adapter and independent authenticated bootstrap service; own controller boot uses the same local journal implementation; CP1/CP2. | L; high, identity and acceptance boundaries. | +| A12 | P1 | Mutual authentication and roles were stated, but identity issuance, renewal, revocation, returning stale peers, queued-user revocation and tenant resource isolation were unspecified. | Supplied workload identity/OIDC integration, exact boot credentials, reauthorization on adoption, quotas and endpoint policy; CP5/CP6/CP11. | L; high, trust lifecycle and scope isolation. | +| A13 | P0 | Backup restore only said to reconcile and fence old sessions. Restoring an old journal can also restore old controller epochs and omit accepted actions. | Independently anchored control-plane generation, exclusion of old writers/gateways, recovery mode and scope-local unknown-obligation barriers; CP0/CP11. | L; high, stale restore must not revive authorization or erase native effects. | +| A14 | P1 | Acceptance had no quantified large-fleet size/load/SLO/soak envelope; the initial 128-stream limit was unrelated to operator demand. | Small and large profiles, including 10,000 nodes/one million Cells, latency/freshness/fairness/takeover targets and 72-hour soak; CP12. | L; medium, targets require measured hardware and realistic native behavior. | +| A15 | P1 | Runbooks listed topics but supplied no automated preflight, dependency diagnosis, grouped alerts, credential/backup supervision or redacted support bundle. | Preflight/doctor/explain commands, service objectives, bounded traces, actionable alerts and automated housekeeping; CP6/CP11/CP13. | M; medium, diagnostics must be truthful and avoid leaking application data. | +| A16 | P1 | Partial capability stages could be delivered indefinitely while the product was called complete. Missing identity/migration/support artifacts had no release blocker. | Explicit baseline feature list, signed release/compatibility artifacts and machine-checked evidence ledger that fails on missing/skipped native gates; CP0/CP12/CP13. | M; medium, release enforcement across source and provider versions. | +| A17 | P0 | Health receiver exclusions were planner inputs only. Ordinary writer/read/follower acquisition could bypass the intended operational exclusion. | SDK installs canonical revisioned receive admission gates across every new-role path, independently of sticky maintenance and pressure; CP7. | L; high, must preserve existing owners and canonical recovery safety. | +| A18 | P0 | The single-runtime controller model did not explain how a three-node pool could manage many application FleetScopes without mixing boot identity, journal permissions or management Cell authority. | Dedicated management application/runtime and boot scope, explicit per-fleet service grants, worker scope preserved in every action, cross-fleet moves rejected; CP0/CP1/CP2/CP11. | M; high, identity isolation across applications. | + +## Decisions preserved after review + +- PostgreSQL remains the supplied production journal backend. This is an + explicit tradeoff to keep atomic application transactions and independent + controller bootstrap. A managed service and maintained adapter reduce team + effort; a Cellule journal would require additional atomicity/bootstrap proof. +- Cellule hosts real management Cells with ordinary durability and recovery. + They are rebuildable projections so their loss cannot block authoritative + journal takeover. Their partitioning prevents a single projection writer from + becoming the fleet's throughput ceiling. +- Runtime authority, leases and canonical recovery remain the data safety path. + Health suspicion does not prove writer failure, and finalization still needs + all native obligations and exact session withdrawal. +- The baseline has one journal transaction domain per installation. Regions + have independent installations; cross-database/region Cell migration remains + an explicit unsupported operation. Large fleet support within one supported + installation is a mandatory release gate. +- Kubernetes is the fully supplied initial capacity/lifecycle platform. VM + hosting uses the same binary, while automated provisioning is advertised only + for qualified built-in providers. Application teams should not implement these + adapters to obtain the baseline feature set. + +## Remaining release evidence + +All eighteen identified gaps now have a design decision, implementation owner +package and acceptance requirement. That does not prove that every possible +failure has been discovered. The revised plan requires canonical protocol models, +native fault tests, database and platform qualification, compatibility/restore +drills, measured full-scale behavior and independent operator onboarding before +production readiness can be claimed. + +Record the final design fingerprint alongside each future implementation run. +Changes to partitioning, archive/closure evidence, restore fencing or disruption +accounting reopen their corresponding audit finding until the revised protocol +and qualification evidence are accepted. Do not close a production finding +solely because its documentation was updated. diff --git a/docs/fleet-control-plane-plan.md b/docs/fleet-control-plane-plan.md new file mode 100644 index 00000000..09cbb6b3 --- /dev/null +++ b/docs/fleet-control-plane-plan.md @@ -0,0 +1,1438 @@ +# Cellule fleet control plane design and implementation plan + +Status: proposed implementation handoff. This document delivers the design; +the controller application, HTTP API, UI, and production journal described here +are implementation targets. + +Prepared: October 1, 2026, America/Vancouver. +Inspected committed baseline: `58721227e56dac3ebcda6b74b4ed2a514f8cc42b`. +Concurrent reader enrollment work was present during inspection. This design +does not certify that work or depend on its uncommitted API names. + +Build a highly available fleet management application on Cellule. Several +controller capable nodes serve the management API and UI. One fenced owner per +execution partition drives canonical reconciliation under shared fleet budgets. +The application identifies +slow, pressured, and unhealthy nodes, records operator intent, and automates +bounded Cell relocation and maintenance through canonical runtime mechanisms. + +The audience is implementers and operators. The decisions, interfaces, ordered +changes, commands, and acceptance criteria below are sufficient to start work +without earlier conversation context. Sections explicitly marked proposed do +not describe shipped APIs. Release acceptance includes the production and scale +gates in this document. + +The [production audit](fleet-control-plane-audit.md) records the gaps found in +the first design and maps each to a required implementation and release gate. + +| Read first | Purpose | +| --- | --- | +| [Product and integration](#production-product-and-application-integration) | What Cellule supplies and what an application team configures. | +| [Fleet scale](#fleet-scale-and-partitioned-execution) | Partition safety, complete incremental inventory, history lifecycle and measurable scale targets. | +| [Lifecycle automation](#capacity-control-and-lifecycle-automation) | Capacity, disruption budgets, upgrades and low-intervention operation. | +| [Implementation packages](#ordered-implementation-packages) | Ordered source changes, dependencies and exit evidence. | +| [Release acceptance](#acceptance-matrix-and-required-evidence) | Native faults, performance, operational usability and required artifacts. | + +## Delivery contract + +The deliverable is a versioned controller binary/container with bundled UI and +CLI, a maintained worker SDK, a supported production journal adapter, standard +identity and platform integrations, automated lifecycle workflows, fault +scenarios, and deployment runbooks. Three controller processes must adopt +durable operations across failure while actual CellNodes preserve acknowledged +application state and the single writer contract. + +| Decision | Initial implementation | +| --- | --- | +| Controller topology | Three controller capable nodes in distinct failure domains; API service on all three, fenced reconcilers per execution partition with shared fleet budgets. | +| Controller runtime | One `CellNode` and one runtime per controller process; ordinary Cellule hosting and lifecycle. | +| Deployment isolation | Dedicated controller pool by default; mixed worker/controller nodes are supported only with reserved management resources. | +| Journal | PostgreSQL application adapter implements all existing fleet journal traits in one transaction domain. The current SQLite example remains a local reference. | +| Cellule management state | Real management Cells provide partitioned, rebuildable read models and audit mirrors. Canonical operation state remains in the journal for this release. | +| Fleet execution | Reuse `FleetReconciler`, runtime planner/reducer, node action executor, and canonical Cell authority. | +| Admission and safety | Local pressure protection, lease fencing and publication continue without a controller; new enrollment and recovery-role recruitment can depend on the journal gateway. | +| API and UI ownership | Ship the controller application and optional worker SDK with HTTP, identity integration, UI, and deployment defaults; framework core remains provider neutral. | +| Initial bounds | Retain two unresolved moves and 8 GiB per partition; add atomic fleet and installation caps and disruption permits before enabling partition parallelism. | +| Automation defaults | Observation only; enable relief, maintenance, slow-node relocation, and balancing through individually qualified policy stages. Production release must qualify every advertised baseline feature. | +| Bootstrap | Journal, catalog, object storage, node identities, and canonical authority are available independently of controller management Cells. | + +Three controller processes provide application redundancy. Journal availability +requires a separately qualified database deployment with fenced primary +promotion and preservation of acknowledged commits. Controller election does +not implement database consensus. + +### Relationship to the existing fleet plan + +The [fleet operations plan](fleet-operations-plan.md) remains authoritative for +movement, enrollment, role evacuation, primitive quiescence, recovery, and +finalization. This document supplies the application and deployment around it. +It does not introduce another Cell authority or weaken its acceptance gates. +Partitioning, snapshot/archive closure and management-generation fencing require +explicit versioned amendments to the fleet plan in CP0. Until those protocols +and migrations are implemented and qualified, current global bounds and strict +roster contracts remain authoritative; this document cannot be used to bypass +them by configuration. + +| Present foundation | Use here | Remaining dependency | +| --- | --- | --- | +| [FleetReconciler](../crates/cellule-host/src/fleet/reconciler/mod.rs) | One bounded reconciliation pass; adapter interfaces and progress report. | Complete observation and maintenance barriers from the fleet plan. | +| [FleetJournal](../crates/cellule-host/src/fleet/controller.rs) | Controller claims, revision checks, permits, intents, scheduling, history. | Production adapter and atomic application request/policy transactions. | +| [FleetRoster](../crates/cellule-host/src/fleet/roster/mod.rs) | Traverse retained intents and enrollments, including failed boots and Pending work. | Match complete native writer, reader, producer, and follower evidence. | +| [Fleet action contracts](../crates/cellule-host/src/fleet/actions.rs) | Journal exact node effects and preserve original accepted inputs/results. | Full maintenance actions and finalization qualification. | +| [Pressure classifier](../crates/cellule-runtime/src/fleet/pressure.rs) | Use actual locally classified, signed pressure. | Application telemetry and independent health/slowness evaluation. | +| [Fleet operations example](../crates/cellule-host/minion/README.md) | Journal contracts and real movement/restart scenario patterns. | Remote HTTP, multiple processes, sustained convergence, production authentication and providers. | + +The current ownership-only example explicitly reports incomplete role coverage. +Its overload and controller-restart scenarios do not establish complete node +maintenance or production availability. Track completion against the exact +source revision and [fleet execution evidence](fleet-operations-progress.md). + +## Architecture and responsibilities + +```mermaid +flowchart TD + Operator[Operator UI and CLI] --> API[Controller API replicas] + API --> Journal[Transactional fleet journal] + Active[Partition owner loops] <--> Journal + Standby[Eligible takeover loops] <--> Journal + Active --> Observer[Authenticated fleet observer] + Observer --> Nodes[Worker and controller CellNodes] + Active --> Driver[Existing FleetReconciler] + Driver --> Transport[Authenticated node transport] + Transport --> Nodes + Nodes --> Runtime[Canonical actor and lifecycle paths] + Runtime --> Authority[Existing Cell and node authority] + Journal --> Projector[Replayable projection worker] + Projector --> Views[Cellule management Cells] + API --> Views +``` + +Every controller process exposes HTTP routes and hosts real Cellule management +Cells when admitted. Only the journal lease holder reconciles its assigned +execution partition. The management Cells use the ordinary fenced writer, durable response gate, and +recovery paths; their writer may live on a different controller than a partition +lease holder. These two ownership concepts are independent. + +Controllers host one dedicated management application and runtime, with +management Cell namespaces keyed by authorized tenant/fleet. A controller's boot +enrolls in that management scope; it is not advertised as a worker boot in every +managed application. Its service identity receives explicit grants for the +managed fleet scopes. Partition claim validation binds that identity to its +management boot, deployment and granted scope. Worker actions always retain the +worker application's own FleetScope, NodeId and SessionId. A cross-fleet move is +rejected; operating many fleets does not merge their data authority or identity. +Management pool maintenance is coordinated through its own scope by surviving +controllers, and its disruption policy protects all managed fleets' service. + +| Layer | Responsibility | +| --- | --- | +| Runtime | Pure placement, operation transitions, authority, recovery, local admission, and resource accounting. | +| Host | Fleet driver, exact node actions, inventory, lifecycle, retained work, and one drain lane. | +| Controller application | Supervised loop, health evaluation, policy, journal adapter, API, node transport, projections, UI, and audit presentation. | +| Worker SDK supplied with Cellule | Install fleet facilities, enroll boots, publish observations, bind native action owners, expose standard management transport, renew identity, and supervise intent. | +| Embedding worker application | Supply its existing catalog/providers, application identity and business-specific workload constraints; configure the SDK once. | +| Deployment | Database failover, object storage, identity and certificate provisioning, load balancing, failure domains, process restart, and backups. | + +Controller role is application deployment metadata associated with a stable +physical `NodeId` and fresh boot `SessionId`. It is not a new Cell ownership +kind. Keep capability metadata in the journal application schema; do not +silently extend signed advertisements or reinterpret their persisted fields. +Metadata alone cannot authorize enrollment, receive capacity, or takeover. + +### Proposed application layout + +Create `apps/fleet-controller/` as a supported application workspace with its +own lockfile, using path dependencies on the framework during development. Ship +versioned binary/container, CLI, UI, and a publishable optional `cellule-fleet-agent` +SDK from this workspace. Production artifacts pin compatible framework versions. +HTTP, identity, database and platform dependencies stay in these application +packages. The reference fixture exercises the same release binaries. + +```text +apps/fleet-controller/ + Cargo.toml # application workspace, fleet-controller and cellule-fleetctl + Cargo.lock + AGENTS.md # application boundary and verification rules + src/main.rs + src/bin/cellule-fleetctl.rs + agent/ # optional cellule-fleet-agent SDK + src/bootstrap/mod.rs # one CellNode, enrollment, management Cells + src/controller/mod.rs # partition leases, fairness and takeover + src/partitions/mod.rs # assignment epochs and cross-partition reservations + src/capacity/mod.rs # desired node pools and disruption budgets + src/journal/mod.rs # existing fleet traits plus application transactions + src/journal/postgres.rs + src/inventory/mod.rs # snapshot manifests, deltas and evidence archive + src/platform/mod.rs # built-in Kubernetes lifecycle adapter + src/requests/mod.rs # durable request and policy processing + src/observer/mod.rs # complete retained roster and native role matching + src/health/mod.rs # bounded health and slow-node evaluation + src/transport/mod.rs # authenticated exact-boot node calls + src/http/mod.rs # public and internal management routes + src/projection/mod.rs # Cellule read model and audit mirror + src/management_cells/mod.rs + migrations/ + openapi.yaml + ui/ # static TypeScript UI assets and build lockfile + deploy/ # supported Helm release, VM templates and profiles + release/ # signed artifacts, SBOM and compatibility manifest + qualification/ # subprocess/provider scenarios and evidence runner + README.md +``` + +Use an application HTTP server, a PostgreSQL driver with bounded connection +pooling, and a small TypeScript UI with pinned dependencies and reproducible +asset builds. Select and pin concrete dependencies in CP1; record the selected +versions and license review in the application manifest rather than copying +unverified versions into this plan. Avoid importing Rust files from another +example with `#[path]`. Share test contracts by fixtures or a deliberate +application support module, keeping one production execution path. + +## Production product and application integration + +The production deliverable is maintained by Cellule and distributed as one +controller image with API, UI, scheduler and journal gateway, plus a CLI and an +optional worker SDK. A managed PostgreSQL service and the application's existing +Cellule object/authority storage are the required persistence dependencies. No +separate message broker, bespoke operator, time-series service or coordination +cluster is required for the baseline. External metrics and identity systems can +be connected through supplied integrations. + +### Application team contract + +The team supplies its existing `CellNodeBuilder`, trusted catalog and storage +providers, application scope, an identity binding, and optional business-specific +workload constraints. The SDK supplies enrollment, intent watching, node +observations, signatures, bounded management transport, action wiring, accepted +work ownership, certificate renewal, metrics and graceful drain integration. +A team using supported deployment providers writes no journal, observer, +transport, election, health classifier, UI or maintenance state machine. + +The proposed `FleetAgent::install(builder, config, application_hooks)` returns a +validated builder and owned management service. It binds to the single existing +CellNode/runtime, uses one lifecycle drain lane, and fails before runtime start +when required facilities or identity are missing. The SDK can mount its router +in an application HTTP server or own a dedicated management listener. Both modes +use identical authentication, bounds and native contracts. The release supplies +a complete compiling integration example and upgrade guide. + +Business hooks are explicit: catalog lookup, primitive-specific readiness that +cannot be inferred by the framework, workload-class SLOs, and optional placement +constraints. Every supported built-in primitive gets a shipped readiness +adapter. Unknown/custom primitives block their affected moves with a named +capability error; they cannot block unrelated healthy Cells or be treated as +safe by omission. Preflight reports the exact missing hook before automation +is enabled. + +Worker agents do not receive database credentials. Their journal trait adapter +calls the controller's authenticated journal gateway, available on every +controller replica. The gateway executes the same PostgreSQL transactions and +binds each request to the enrolled physical node, exact boot and allowed role. +Controller-to-node dispatch followed by node-to-gateway acceptance creates no +open database transaction across the network. The native effect starts only +after acceptance is confirmed; an ambiguous gateway reply retains the original +request for reconciliation. Controller leadership is unnecessary for gateway +availability, while mutations still validate the current partition epoch. + +Already serving Cells retain canonical local protection during a controller +outage. New fleet effects and enrollment require the gateway and may block new +boots or recovery-role recruitment. Publish this dependency in availability +status and deployment planning; do not promise that an arbitrarily long control +plane outage is invisible to a restarting fleet. + +### Included operator workflows + +Ship `cellule-fleetctl` and matching UI/API workflows for fleet creation/import, +node-pool enrollment, preflight, dry-run placement, maintenance scheduling, +rolling upgrades, capacity scaling, cancellation of unstarted requests, stopping +new scheduling, safe retirement, replacement, and return to service. A bulk +operation stores a selector snapshot and bounded child cursor. Re-evaluate +current safety constraints for each child; nodes enrolled later do not silently +join the operation. Resume from durable progress after process failure. + +One declarative FleetSpec contains identity bindings, node pools, failure +domains, storage references, workload constraints, capacity limits, disruption +budgets, maintenance windows and automation policy. A revision-checked apply +operation reports drift, validates capability compatibility and records a plan. +Applying that same specification again creates no duplicate work. Exported +configuration contains secret references only. Bootstrap/import and fleet +retirement are resumable operations with evidence and visible blockers. + +The supported baseline deployment is Kubernetes: ship a Helm chart for the +controller pool, worker SDK deployment templates, the PostgreSQL connection +profile, probes, resources, disruption policy, identity integration, dashboards +and alerts. A VM/systemd deployment uses the same binary and protocol; automatic +machine provisioning is advertised only for built-in qualified providers. +Application teams can use an existing managed database and cluster identity. +Provider configuration is owned once by the platform team. Each node pool +binds a `cluster_ref`, namespace/workload UID and instance UID; one regional +installation may span several supported Kubernetes clusters. The 10,000-node +target does not imply one Kubernetes cluster exceeds its own qualified limits. + +## Fleet scale and partitioned execution + +The current implementation has one controller head per FleetScope, two active +attempts, 8 GiB restore credit, full roster traversal and a 10,000-row bound. +These are current foundations, not large-fleet qualification. The production +release must add the versioned contracts below before claiming scale support. +Raising constants or running several unfenced copies of the current driver is +not sufficient. + +### Execution partition contract + +Introduce `ExecutionPartitionId` and `assignment_epoch` in management journal +namespaces and authorization envelopes. Keep canonical Cell IDs, application +identity, fleet identity and Cell authority unchanged. Each admitted worker +boot belongs to exactly one execution partition, normally selected by node pool +and failure domain. Target at most 256 live nodes per partition; provision more +partitions before reaching that target. A partition can have a larger retained +obligation set, which is paged and indexed rather than loaded into one vector. + +Each partition owns a head, registry, lease, pending work queue and disruption +permit ledger. Each head initially retains the current two-attempt and 8 GiB +bounds. One controller owns a partition lease, while all three controllers may +own different partitions. Deterministic assignment plus journal CAS distributes +leases; bounded work stealing reassigns eligible partitions after failure. +Maintain one canonical reconciler, instantiated with a partition-aware journal +view. A short fleet metadata record contains policy, partition assignments and +shared caps; it is not updated for every sample or ordinary partition pass. + +Use a proposed default of 64 unresolved moves and 256 GiB restore credit per +fleet, and 128 moves/512 GiB per installation, subject to lower admission and +operator limits. These new aggregate defaults require measured qualification; +existing profile limits remain unchanged within each partition. Charge global, +fleet, partition, source/receiver and failure-domain permits atomically before +first dispatch. Unknown work retains every charge. Telemetry writes never take +these budget locks. Reserve capacity for resolving accepted work so discovery, +UI traffic and low-priority balancing cannot exhaust it. + +A cross-partition move has one immutable attempt owned by its source partition +and a destination reservation reference. One PostgreSQL transaction compares +both assignment epochs and intents, locks the relevant budget and head rows in +stable order, reserves the receiver and charges all required ledgers. Only the +source partition's fenced owner advances the attempt; destination actions +validate the same journal attempt and exact destination boot. Receiver credit +and both partitions' references are retired together only after canonical +settlement. Count a crossing move once in fleet/installation attempt totals while charging +the applicable resources in both partitions. The destination reference cannot +allocate a second attempt or independently free the original permit. Limit the production baseline +to one journal transaction domain per installation; cross-database or cross-region +Cell migration is an explicit unsupported capability until a separate protocol +is qualified. + +Node repartitioning is a durable operation: stop new placement and enrollment +for that boot, retain its old partition's effects, settle accepted work and +cross-partition references, prove a complete responsibility handoff, then CAS a +new assignment epoch. Stale envelopes cannot authorize work after the switch. +If no safe handoff exists, keep the assignment and add another partition for new +nodes. An automatic rebalancer cannot rewrite membership while work is unknown. +Partition maintenance preserves the initial one-drain-per-partition limit and +also consumes the fleet/failure-domain disruption permits below. + +### Complete inventory without repeated full scans + +Replace each-pass all-history traversal with a canonical versioned inventory +snapshot and a durable change log. Every enrollment producer updates its role +record, exact boot index, role/Cell adjacency index, per-partition count/digest +and monotonic change sequence in the same journal transaction. Snapshot manifests +name immutable page roots, membership revision and sequence watermarks. Capture +native pages against the manifest; retain each page's actual boot, generation, +capture interval, signed count and completeness flags. + +The observer bootstraps from a complete manifest, then consumes ordered deltas. +It detects sequence gaps, compares digest/count checkpoints, and rebuilds the +affected scope after a gap or incompatibility. Background full reconciliation +runs incrementally at least every 15 minutes, with jitter and bounded bandwidth; +it never blocks renewal or accepted-work settlement. Hot mutation authorization +rechecks current intent, membership, policy and exact native source/receiver +facts. A cached complete manifest authorizes no effect by itself. + +Unrelated head renewal or new work in another partition must not restart the +whole scan. A valid MVCC/immutable snapshot remains readable for its bounded +lifetime; current authorization independently detects changes that matter to +an action. Scope closure evidence includes every unresolved crossing reference +and dirty producer. Finalization atomically closes the enrollment gate, pins a +barrier, drains all pre-barrier accepted jobs, and compares the resulting exact +role/dependency set before committing. It cannot use a partial incremental cache. +The existing strict roster contract remains in force until this replacement +has equivalent completeness and race proofs in public host tests and models. + +Cache lightweight node summaries every ten seconds with jitter. Fetch detailed +Cell/role pages on change, candidate selection, operator demand and background +reconciliation. Controller replicas share persisted inventory; standby controllers +do not each poll every node. Bound all caches and page buffers by bytes, use +indexed candidate queues, and batch cost/cooldown lookups. A pass's expected +cost is proportional to changed nodes and candidate pages, not total history. +Maintain an index for greatest committed movement time per incarnation and +partition scope in the same retirement transaction; historical linked-list traversal +cannot remain on the production planning path. + +### History and terminal evidence lifecycle + +Separate unresolved responsibility indexes from terminal history. Preserve +Pending, Unknown, current enrollment, foreign tails, accepted jobs and all +referenced proofs in the active closure set. Compaction may archive a terminal +record only after terminal native evidence, durable result and all dependent +references are settled. Publish an immutable archive manifest/digest and a +lookup/exclusion index in the same transaction that removes it from active scans. +A reader can prove active-set completeness across active and archived roots. + +Exact request replay retrieves the original archived inputs and outcome. +Late acceptance for an archived terminal identity is rejected or returns that +original terminal outcome; it can never recreate Pending from absence. Retain +session tombstones and terminal exclusions under the canonical authority rules. +Archive unavailability produces an explicit Unknown/blocker, never permission +to repeat an effect. Compaction, index rebuild and backup restore use the same +proof rules. Run them automatically with space forecasts and bounded IO. + +Qualify more than ten million terminal enrollments and repeated boot churn +without increasing steady-state active scan cost with historical row count. +The current 10,000-row collector cannot satisfy this; changing only its numeric +cap is not an acceptable implementation. Page byte limits, backpressure and +explicit over-limit status remain mandatory even after the collector changes. + +### Qualification scale and service objectives + +These are proposed release targets, not results. Publish measured limits and a +resource profile with each release. Small installations use one partition and +need no manual partition configuration. Large installations add partitions and +controllers through the supplied reconciler and declarative specification. + +| Profile or objective | Required qualification target | +| --- | --- | +| Small supported installation | Up to 100 nodes, 10,000 Cells and 10 fleets; three controllers with 4 vCPU/8 GiB each; journal sizing profile recorded. | +| Large supported installation | Up to 10,000 nodes, 1,000,000 Cells and 100 fleets in aggregate, with one fleet allowed to use the full node/Cell total; three controllers with 16 vCPU/32 GiB each and initial journal profile 16 vCPU/64 GiB. | +| Large workload | 1,000 node summaries/s, 10,000 bounded inventory deltas/s, 200 operator requests/s, 500 API reads/s, and 1,000 concurrent event streams; specified burst/queue tests included. | +| Read API | p95 below 250 ms and p99 below one second for indexed status and bounded pages under the published load and network profile. | +| Durable request admission | p95 below 500 ms and p99 below two seconds, excluding client retry time; saturation returns bounded explicit rejection. | +| Freshness | p99 node summary age below 30 seconds for reachable nodes; role coverage and detailed-page age are separately exposed. | +| Takeover | p99 partition owner replacement within 60 seconds with healthy journal and peers; control plane can lose one controller while meeting the supported load. | +| Fairness | At least one eligible reconciliation opportunity per active partition within 30 seconds under the profile; no quiet fleet starvation from a noisy fleet. | +| Soak | 72 hours with repeated boot churn, backlog, partition reassignment, history compaction and one-controller-loss periods; memory, connections, queues and storage-growth slopes remain within published bounds. | + +Per-controller connection and stream limits are sized from this profile; replace +the initial 128-stream prototype limit with a qualified default of 1,024 streams +per controller, still enforcing per-tenant and byte caps. Demand beyond published +limits returns a capacity diagnosis and installation scaling recommendation. +A claimed large-fleet release requires real native inventory/role behavior and +full-topology control-plane measurements. Synthetic generator throughput alone +cannot establish the large profile; record simulator and native results separately. + +## Controller startup and bootstrap + +The controller cannot require its own running reconciliation loop to create +the facilities necessary to start that loop. Initial journal provisioning and +roster import are explicit deployment operations. + +1. Provision the database, canonical catalog, authority/object storage, stable + physical identities, and application signing keys through deployment tooling. +2. Create the fleet journal with movement stopped and bootstrap incomplete. + Import the controlled node roster and retained obligations. Commit bootstrap + only after the existing enrollment barrier is satisfied; an empty live + directory is insufficient. +3. Start the controller database adapter and authenticated bootstrap/gateway + service independently of management Cell readiness. A controller enrolls its + own runtime through that same local journal implementation; it need not call + a gateway that depends on its own runtime having started. Platform identity + and certificate issuance must also be available independently of management + Cells. Start each controller with a fresh session. Read its exact physical intent + and follow the existing [fleet boot admission](../crates/cellule-host/docs/lifecycle.md#fleet-boot-admission) + sequence, including Pending, canonical advertisement, Established, atomic + confirmation, required probes, and ordinary host start. +4. Start the remaining API, observer, projector, and reconciliation tasks under explicit + ownership. Management readiness and Cell serving readiness remain separate. +5. Create or recover the declared management Cells through ordinary catalog and + authority paths. Their absence delays projections, not journal reconstruction + or already authorized fleet work. +6. A controller observes assigned partition leases, then attempts a claim if + eligible under the current assignment epoch. Bootstrap does not require becoming leader first. +7. Begin observation only. Enable new movement through a durable revision + checked policy command after qualification. + +The controller pool is excluded from ordinary automatic workload balancing in +the initial deployment policy. Management Cells are placed only within that +pool through trusted application placement inputs. Pool discovery and recovery +use ordinary canonical directory/authority facilities. If no controller is +available, data nodes retain local protection and ordinary recovery, but no +optional fleet operation starts. + +CP1 must enforce management Cell eligibility at ordinary acquisition as well as +fleet receiver preparation. Declare management code/schema support only on the +controller pool and validate it through the trusted catalog/provider path; route +management clients only to compatible signed boots. Test an attempted ordinary +acquisition on a worker. Controller labels alone cannot enforce this constraint. + +An entirely Cellule backed journal is a future adapter, not a prerequisite for +this delivery. It must prove atomicity across head, registry, accepted actions, +requests, policy, and audit plus independent bootstrap and failover. Hosting +an ordinary SQL Cell alone does not establish those contracts. + +## Durable journal and application transactions + +Implement `FleetJournal`, `FleetEnrollmentJournal`, and `FleetActionJournal` +against one PostgreSQL primary and transaction domain. Never validate a permit +from a cache or follower and then accept an action in another database. + +### Proposed record model + +Canonical records use existing bounded codecs. Application records have their +own explicit schema version, length limits, checksums, and migration rules. +All primary keys include application and fleet scope. + +| Record | Essential fields and rules | +| --- | --- | +| Partition head and registry | Canonical encoded records, partition/assignment epoch, exact head/registry revisions, validated immutable per-partition `FleetProfile`, bootstrap and scheduling state. | +| Fleet and installation budgets | Aggregate move/restore/IO/disruption permits, policy revision and referenced immutable attempts; updated atomically with partition allocation/retirement. | +| Intents and enrollments | Existing exact physical identities, boot sessions, original specs, statuses, evidence, and retained revisions. Preserve failed and Pending obligations. | +| Actions, basis, results, history | Existing immutable exact inputs/proofs; retain unknown results and permits; atomically publish history with permit retirement. | +| Controller capabilities | NodeId, enrolled session, eligibility, failure domain, build/schema compatibility, bounded lease diagnostics. | +| Operator request | Request ID, principal, kind, target, idempotency key, canonical body digest, expected policy/intent revisions, submitted time, expiry, state, linked canonical operation/attempt IDs. | +| Policy | Revision, automation stage, health thresholds, target exclusions, controller pool, risk limits, and policy digest. No credentials. | +| Decision | Decision ID, policy revision, observation/evidence digest, rationale, target, proposed cost, and resulting request/attempt IDs. | +| Audit and event outbox | Per-partition ordered sequence, request/actor identity, transition type, referenced canonical evidence, previous/new revision, commit time. Append in the same transaction as the change. | +| Health checkpoint | Exact boot, signal/window identifiers, original sample range, state, reason, expiry. Advisory; cannot substitute for current role evidence. | +| Projection cursor | Last applied journal event sequence, model schema, and Cell identity/incarnation. A projection cursor never authorizes mutation. | + +Operator requests and decision records are bounded independently of the small +fleet head. Start with at most 128 active child requests per execution partition and 1,024 accepted +bulk parents per fleet; a parent keeps a bounded cursor rather than materializing +an unbounded queue. Cap queued interactive requests at 128 per partition and a 64 KiB public +request body. Excess requests return capacity errors. Terminal audit/history +retention is controlled by referenced proof obligations and documented archival +policy; never delete evidence because a UI retention period elapsed. + +### Transaction recipe + +Use serializable transactions and a consistent lock order: installation and fleet budget rows when required, sorted partition heads, +partition registries, application policy/request metadata, then sorted +intent/enrollment/action rows. +Lock the affected partition heads for canonical mutations; lock shared budget +rows only for permit changes. Telemetry, list reads and projection updates do +not lock scheduling heads. Policy revision and scope suspension checks remain +in each authorization transaction. Run the existing pure reducer inside the +transaction; insert referenced rows and events before committing the resulting +head. Recheck exact expected versions and complete immutable replay inputs. + +PostgreSQL serialization failures require retrying the whole transaction. +Use the same immutable request ID, bounded retries, and the remaining deadline; +do not repeat external node effects inside a database retry. Row locks end at +transaction completion. These mechanisms are described in the official +[transaction isolation](https://www.postgresql.org/docs/current/transaction-iso.html) +and [locking documentation](https://www.postgresql.org/docs/current/explicit-locking.html). + +Production database deployment must preserve acknowledged commits during +primary promotion, reject writes on the old primary, and route authorization +reads to the current primary. Qualify synchronous WAL durability and the +promotion rule together. Synchronous replication can wait for standby WAL +durability or application, depending on configuration; it does not supply +primary fencing by itself. See the official +[synchronous replication contract](https://www.postgresql.org/docs/current/warm-standby.html#SYNCHRONOUS-REPLICATION). +Record the selected database release, replica configuration, backup/restore +process, and failure proof in CP2 and CP12. + +An ambiguous commit response is Unknown. Retry lookup using the original +idempotency identity. Never reply with a definite rejection if the transaction +could have committed. HTTP clients may safely resubmit the same request. + +### Proposed request processing contract + +`submit_request` performs authorization, canonical validation, scope/revision +checks, idempotency comparison, request insertion, and audit insertion in one +transaction. Equal key and equal body return the original result; equal key +and different body return conflict. This contract is application owned and +does not yet exist in `FleetJournal`. + +A `202 Accepted` response means the request is durably queued. It does not mean +the node is cordoned, a move has begun, or maintenance is complete. Expose +`Queued`, `Running`, `Blocked`, `Completed`, `Rejected`, and `ExpiredBeforeStart` +as application status views with explicit canonical phase and evidence fields. +Do not map an unknown accepted effect to a terminal rejection or expiration. + +The active controller adopts a queued maintenance request by atomically linking +its request to `BeginMaintenance` and the committed node intent. Only one node +maintenance operation may be active per execution partition initially; fleet +and failure-domain disruption permits additionally constrain concurrent drains. +Another request remains visibly queued with `MaintenanceBusy`. Recheck its expected target revisions and expiry +before adoption. Requests that expired before any canonical action may become +`ExpiredBeforeStart`; started requests preserve their intent and obligations. + +Policy updates and scheduling stop/resume commit atomically with their audit +event and registry gate. Existing charged attempts continue settling after +stop. Acceptance and Allocate recheck the policy revision in the same +transaction. `FleetProfile` resource bounds are immutable for a journal scope +in this release; changing them requires a separate qualified migration. + +### Proposed application interfaces + +Implement these logical methods on the same journal adapter, with original +source errors, canonical immutable inputs, bounded pages, and retained jobs: + +```text +submit_request(scope, authorized_principal, immutable_request, deadline) + -> OriginalOrNewRequest +adopt_request(expected_head_and_registry, controller_epoch, request_id, now) + -> RequestLinkedToCanonicalTransition +replace_policy(scope, expected_policy_and_registry, immutable_request, now) + -> CommittedPolicyAndAudit +load_operation(scope, request_or_operation_id) + -> CanonicalStatusWithEvidenceReferences +events_page(scope, partition_cursor_vector, limit) + -> BoundedCommittedEvents +load_controller_eligibility(scope, physical_node, exact_session) + -> RevisionedEligibility +``` + +Authorization is verified by the application adapter before transaction entry; +persist its principal and scope with the accepted immutable request. Fleet +action acceptance independently verifies authenticated node scope and current +journal authorization. Request adoption checks expiry, compatibility, target +revision, policy revision, current epoch and shared budget together. No method +calls a node or performs Cell activation inside its database transaction. + +## Partition ownership and process lifecycle + +Each process has a fresh claimant `SessionId`. No static leader setting, API +load balancer choice, or management Cell writer replaces the existing +`ControllerLease` epoch. Every transition and node acceptance verifies current +journal authorization. Already accepted work remains owned by its node executor +and must be adopted after takeover. + +Use the current timing/profile defaults per partition initially: 30 second partition lease, +15 second periodic reconciliation, two unresolved attempts, and 8 GiB restore +budget per partition. The profile validates lease duration at no more than 30 seconds. Use +a proposed 10 second pass budget so a slow pass leaves renewal headroom. +Progress events can wake the loop sooner, but coalesce them and allow only one +pass per owned partition. A controller admits at most 16 concurrent passes, +128 outstanding node RPCs and 32 database transactions by default; lease renewal +and accepted-work settlement have reserved capacity. Qualification sizes these +limits within the controller resource envelope. The fixture starts with one +fleet and two execution partitions to exercise shared budgets. + +The journal and nodes need one qualified time domain for authorization. The +application clock supplies nonnegative time that cannot regress during a pass; +database transactions validate deadlines and lease claims against trusted +primary time and reject excessive skew. Configure a proposed one second maximum +skew and fail closed when it cannot be maintained. Test clock jumps, primary +changes, and lost time synchronization. Browser timestamps never authorize +effects. Process monotonic time bounds waiters separately from journal time. + +```text +while supervised application task is running: + fairly select a due partition within bounded local capacity + load its committed lease, assignment epoch and compatibility/eligibility + if another live claimant owns it: + wait for lease change or bounded periodic wake with jitter + else: + reconcile_once(trusted_clock, pass_deadline) + treat claim conflicts/fencing as a return to standby + preserve source errors, back off journal failures, publish diagnostics + adopt queued requests through the same fenced transaction contract + join/check owned tasks; coalesce bounded event notifications +``` + +Lease lookup is advisory scheduling information. Only CAS authorizes leadership. +Do not claim a lease twice around each pass: `reconcile_once` already claims or +renews it. New application request transitions must use the current epoch and +the same head transaction. Hold no database transaction over an HTTP call. + +Liveness reports whether the process and supervisor are running. API readiness +reports usable authorization/journal access. A separate controller status +reports each owned partition, claimant, epoch, last completed pass and lease expiry. Standby is a +healthy state. Unavailable projections produce explicit degraded reads rather +than a false complete fleet view. + +### Controller maintenance and shutdown + +For controller maintenance, require another eligible, compatible controller +outside the same maintenance target and a healthy qualified journal. Initially +permit at most one controller maintenance operation at a time, preserving two +eligible controllers from the three-node pool. This is an application safety +policy, not a claim that controllers form a quorum. + +If the target owns partitions, stop its new passes and let those leases expire; +there is no implied lease-transfer API today. Eligible controllers claim the next +epoch for each partition and prove adoption of every charged attempt before the +target drains. This +step retains ordinary node lease maintenance and management access. Recheck +controller eligibility and policy when claiming; an enrolled maintenance node +cannot reacquire controller eligibility merely by rebooting. + +For graceful process shutdown, close new API mutation admission, stop starting +passes, join the current bounded pass, join application transaction/projection +owners, and invoke the canonical host drain. Keep node lease renewal and +management endpoints for as long as accepted work and role settlement require +them. Controller lease expiry does not cancel node effects or release permits. +Retain task handles across cancelled drain waiters; forced process death relies +on durable adoption and recovery and is tested separately. + +Use the host task group for long-running application tasks. Finite journal jobs, +native capture jobs, and node effects require retained completion owners with +bounded admission; do not report normal finite-task completion as supervisor +failure or assume aborting a waiter cancelled native work. + +## Observation and health evaluation + +Capture authenticated, exact-boot observations for the complete active closure +set plus its committed terminal archive/exclusion roots. Preserve busy/transitioning writers, managed readers and accepted reader +jobs, leader enrollment producers, local and cold follower lanes, authoritative +foreign node-log obligations, enrollment state, and current lifecycle evidence. +Revalidate the pinned inventory basis and action-relevant current revisions +after collection and after dependent actions; finalize only under a closed +enrollment barrier. + +Page traversal is bounded and revision aware. Retain canonical page bounds of 128 entries and 1 MiB. Replace the current +10,000-row whole-roster materialization with the snapshot/delta/archive protocol +above before large-scale release. Reject mixed snapshots, gaps and explicit +overflow; never declare completeness from truncation. Controllers cache advisory displays +with original capture times; actions require their existing fresh inspections. + +### Independent status dimensions + +| Dimension | Proposed values | Meaning | +| --- | --- | --- | +| Reachability | Reachable, Suspect, Unreachable, Unknown | Recent authenticated communication; absence does not prove writer failure. | +| Service health | Healthy, Degraded, FailedProbe, Unknown | Required component/runtime probes and independent application availability signals. | +| Performance | Normal, Slow, InsufficientSamples | Workload-aware latency and queue-delay evaluation. | +| Pressure | Existing signed node pressure tier | Actual local classifier result; controller cannot fabricate or re-sign it as node telemetry. | +| Desired mode | Existing retained node intent | Active, maintenance/evacuation, or later lifecycle intent, independent of pressure recovery. | +| Observation coverage | Complete, Partial, Stale, Incompatible | What the collector proved for the displayed roster revision. | + +Do not treat a failed management endpoint as proof that application serving is +dead. Health evaluation never bypasses canonical failed-session proof, +authority fencing, enrollment barriers, or exact recovery. + +### Proposed initial evaluation policy + +Keep sample times, sample counts, workload class, histogram schema, error class, +and authenticated producer identity. Evaluate separate per-window histograms; +do not average node p99 values into a fleet p99. Aggregate compatible histogram +buckets or expose separate quantiles. Retain at most twelve ten-second windows +per boot and at most 32 configured workload classes; excess cardinality becomes +an explicit missing-data diagnostic. Detailed time series live in the embedding +telemetry backend, not the fleet journal. + +| Rule | Proposed starting value | Response | +| --- | --- | --- | +| Missing observations | Three missed ten-second capture windows | Suspect and exclude from proactive receiver selection; retain previous ownership as unknown. | +| Slow node | At least 100 comparable completed operations/window; p95 above configured class SLO and twice the median of at least two healthy peer p95 values for six consecutive windows | Mark Slow with the exact class and sample range. Without two peers or enough samples, report InsufficientSamples. | +| Slow recovery | Six valid windows below class SLO and below 1.5 times peer median | Clear Slow; never clear maintenance intent. | +| Failed local probe | Three consecutive fresh failed probe samples | Mark FailedProbe and exclude proactive receiving; distinguish application failure from management reachability. | +| Pressure relief | Existing signed sustained Shedding/Critical tier | Use the current planner, demand, receiver admission, and fleet permits. | +| Repeated ineffective relocation | Two trial moves without class latency improvement during the next valid two-minute windows | Suspend slow-node trials and expose a capacity/workload blocker. | + +These values are proposed qualification defaults, not measured production SLOs. +Record policy revision and complete evidence for each decision. Restart restores +unexpired checkpoints and original window ranges; if evidence is missing, +collect a full new dwell before acting. A missing or stale sample cannot declare +recovery. Operator thresholds must be validated and revised durably. + +## Automation and operation safety + +Automatic decisions and operator requests use the same journal, resource +budgets, action executor, and evidence. Process charged/unknown work first, +then planned maintenance, sustained pressure relief, qualified slow-node trials, +and ordinary balancing. Local actor pressure protection remains independent. + +### Proposed automation stages + +| Stage | Allocations permitted | Prerequisite | +| --- | --- | --- | +| Observe | None; settle already accepted work when policy stops new allocations. | Authenticated API and explicit completeness/freshness diagnostics. | +| Relief | Current pressure-driven movement. | Trusted pressure, admissible receiver, qualified observer and movement fault gates. | +| Maintenance | Operator requested full role evacuation and finalization. | Fleet W6–W8 native barriers and controller maintenance adoption proof. | +| Slow trials | One eligible trial movement for a slow-node decision, within shared permits. | Explicit policy eligibility extension, comparable telemetry, durable trial/cooldown history, and fault qualification. | +| Balance | Existing count/resource balancing. | Complete fresh membership, role coverage, stable post-batch evidence, and measured convergence. | + +The current driver has no general application API for forced Cell moves or +slow-node targeting. CP7 must extend the existing planner/driver contracts with +explicit policy exclusions and typed relocation reasons, then use canonical +attempt allocation. Specify compatibility and codec changes before adding +persisted fields. Do not falsify signed pressure, remove missing nodes from the +roster, or translate Slow into a permanent maintenance cordon to obtain a move. + +Receiver exclusions for Slow, Suspect, or FailedProbe are application policy +inputs with expiry and revision. They supplement actual signed capacity, +compatibility, and intent checks. They cannot increase advertised capacity or +authorize a stale boot. A policy revision race must fail Allocate or first effect +acceptance in the same transaction; already accepted effects remain adopted. + +The proposed host constructor accepts one `FleetPolicyProvider` supplying a +revisioned policy basis for the pass. Extend the existing planner with pure +`PlannerPolicy` inputs for exact boot exclusions and requested settled Cell +relocations; retain authenticated placement observations unchanged. Include +policy digest and relocation reason in the canonical planning/attempt evidence +through an explicitly versioned format change. The journal adapter validates +that basis again at Allocate and effect acceptance. Keep the existing driver +as the sole executor; these names and constructor changes require CP7 API and +producer/consumer review before implementation. + +Requested Cell moves specify exact Cell incarnation and source boot, with an +optional preferred receiver. They preserve admission, generation, authority, +residence, cooldown and shared resource bounds. Explicit operator intent may +replace the optimizer's gain threshold after policy validation, while runtime +safety gates remain unchanged. An unavailable preferred receiver produces a +blocker instead of silently changing the requested destination. + +For a slow-node trial, select a settled eligible Cell with a qualified cost, +compatible receiver, complete input evidence, and existing residence/cooldown +requirements. Record hypothesis and pre-move class latency. At most one trial +per affected node may be unresolved, while total attempts still obey the +existing fleet maximum. Validate improvement before another trial. Hot/oversized +Cells, fleet-wide backend latency, and no-capacity conditions become blockers +with guidance to partition demand or add capacity. + +Receive eligibility must also reach ordinary acquisition paths. The shipped +SDK installs a revisioned receive gate through canonical host/runtime admission, +covering ordinary writer acquisition, reader/follower recruitment, recovery +receiver selection and prepared movement. A health-policy exclusion cannot be +only a controller planner filter. Keep it separate from sticky maintenance +intent and locally measured pressure; clearing it cannot reopen an operator +cordon. Existing valid owners continue serving. A missing required policy or +unknown startup intent blocks new role admission with a typed reason. CP7 must +prove these paths through public runtime/host behavior before exposing automatic +health exclusion as complete. + +### Maintenance completion + +Writers leaving a node is one milestone. Completion requires canonical evidence +for primitive quiescence, reader closure/replacement where policy requires it, +leader and foreign follower obligations, accepted job settlement, native host +shutdown, `Stopped`, exact session withdrawal, and retained enrollment +retirement under the exact closed enrollment barrier, current assignment and +management generation. Unrelated inventory writes cannot substitute for or +invalidate the target-specific closure proof. + +`safe_to_take_offline` is false or unknown until the committed finalization +proof is verified. A deadline, zero writer count, missing advertisement, expired +lease, or successful cordon is insufficient. Provider shutdown/reboot is an +embedding deployment action following this proof; no generic controller route +deletes leases or powers off machines. Return to service requires a newer +authorized Active intent and a new boot through normal admission. + +Initial Cell maintenance consists of inspection and settled relocation. Busy +Cell quiescence is authorized through node maintenance and its existing exact +intent barrier. Exposing an independent busy-Cell pause, repair or deletion +requires an additional typed operation with primitive-specific contracts; +CP9 does not imply a generic destructive Cell maintenance endpoint. + +## Capacity control and lifecycle automation + +The controller owns desired worker-pool capacity and operation scheduling within +operator-approved limits. Ship a `CapacityPolicy` with minimum/maximum replicas, +per-pool resource shape, failure-domain requirements, restore headroom, cost +ceiling, scale-up/down dwell and maintenance windows. Forecast from admitted +Cell demand and measured restore peaks; CPU or average Cell count alone is +insufficient. Unknown cost prevents unsafe contraction and creates a named +measurement blocker. + +The default platform adapter scales an application-owned Kubernetes StatefulSet +worker pool, with stable physical identities and explicit partition ownership +of its replica count. Use the existing cluster provisioning system to supply +physical compute. A Pending pod, cloud quota or unavailable machine shape stays +`CapacityPending`; do not claim capacity from a desired replica count. Competing +HPA/GitOps writers produce a visible ownership conflict until the user selects +one authority. Ship configuration that prevents accidental competing ownership. + +Scale-up records an idempotent `EnsureCapacity` operation before provider calls, +then advances a retained retired node intent through authorized return-to-service +when reusing its physical identity, and waits for exact new instances, enrollment, +compatible capabilities, probes +and fresh signed receive headroom. Scale-down selects the exact instance the +provider will remove, acquires a disruption permit, drains it through canonical +maintenance, and verifies offline proof for the current instance UID, boot and +intent revision before reducing replicas. In the StatefulSet baseline choose +the next removable ordinal and use a resource-version precondition. Generic +replica decrement that may kill a different, unprepared node is forbidden. The +implementation must pin the supported [StatefulSet lifecycle semantics](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) +and test them for the released Kubernetes versions. A Kubernetes disruption +budget does not replace the fleet role barrier; voluntary and involuntary +disruptions differ as described in the [Kubernetes disruption contract](https://kubernetes.io/docs/concepts/workloads/pods/disruptions/). + +Provider actions have immutable request IDs, target instance UIDs, revision +preconditions, deadlines, accepted/unknown/result state and bounded retries. +Persist provider acceptance before reporting success; ambiguous responses are +resolved through provider lookup and original identities. Restart or replacement +is limited to policy-authorized cases after canonical fencing/closure permits +it. An unreachable worker cannot be destroyed solely to make a drain appear +complete. The Kubernetes baseline removes the exact managed pod, not its shared +host machine. Whole-machine removal is supported only when the adapter proves +closure for every resident managed runtime/tenant; otherwise it is refused. No opaque provider action bypasses Cell durability or protected tails. + +### Disruption policy and placement constraints + +Define pools and workload classes with minimum available capacity, minimum +reader/follower redundancy, allowed code/schema versions, required labels, +anti-affinity/failure domains and maintenance windows. Reserve at least one +node of restore headroom per pool and a proposed 20 percent resource reserve, +subject to measured Cell peak requirements. A smaller authorized deployment +must show its reduced failure envelope explicitly. + +Initial disruption policy permits at most one planned unavailable worker per +failure domain and at most five percent of ready workers per fleet, rounded up +for nonempty fleets, while always preserving declared minimum availability and +redundancy. Controller maintenance permits only one of three controllers. +Compute disruption against actual unavailable nodes plus reserved future +outages; an unexpected node failure consumes the budget and stops new planned +drains. Take the permit atomically with the corresponding intent transition. +Reference counts and original identities survive controller failure. + +Prioritize accepted work, failed-node recovery obligations, maintenance already +in progress, sustained pressure and then optional balancing. Apply deficit-based +fair queuing across fleets/partitions and age waiting requests. Per-tenant limits +cover API calls, pending requests, DB time, inventory bandwidth, event streams +and movement IO. Reserve a proposed 25 percent of control-worker capacity for +lease/gateway/accepted-work progress; noisy discovery or UI traffic cannot use it. +Track throughput by successful settlement, not by attempted dispatch count. + +### Reduced operator intervention + +Ship automatic cooldowns, bounded exponential backoff with jitter, quarantine +for incompatible or repeatedly failing endpoints, and scope-local circuit +breakers. A stalled operation retains its original evidence and retries when +its dependency changes. It cannot generate an endless stream of new requests. +Repeated ineffective movement or widespread infrastructure latency pauses new +optional movement in the affected scope and emits one actionable incident. +Resolved transient failures clear automatically after their recovery window. + +Expose `ExplainPlacement`, `ExplainBlocker` and `PlanOperation` as read-only API +and CLI commands. They return exact missing capacity, redundancy, incompatible +capability, oldest unknown action, proof requirement and next permitted action. +Dry-run plans record their policy/snapshot basis and expire; execution always +revalidates. Operator controls include cancel-before-start, stop-new-children, +extend the same deadline, and resume. Cancellation after native acceptance means +stop new work and settle the original effects, never rollback by assumption. + +Deploy rolling upgrades as durable parent operations: preflight compatibility, +select a canary, reserve disruption/headroom, maintain it, replace the exact +instance with the desired build, verify new enrollment and a healthy soak, then +advance. Stop automatically on budget exhaustion, error-rate regression or +failed receipt readback. Rollback selects a binary that can read current +persisted records; it does not erase migrated state. Fleet retirement stops +new enrollment, settles all responsibilities, fences/withdraws sessions, and +archives evidence before deleting deployment resources. Application data deletion +remains a separate authorized operation with the Blob cross-Cell proof rules. + +## HTTP API and transport contracts + +Public routes are versioned under `/api/v1/fleets/{fleet}`. Authorization binds +principal, application, fleet, resource, and action. Define Viewer, Operator, +and Administrator roles. Operator can request operations and stop scheduling; +Administrator can alter policy, enrollment bootstrap, and controller eligibility. +Deploy browser identity through the embedding identity provider. For cookie +sessions enforce CSRF protection and bounded session lifetime. Never expose +cloud credentials or local storage paths in requests or responses. + +### Proposed public routes + +| Route | Contract | +| --- | --- | +| `GET /api/v1/fleets/{fleet}` | Committed revisions, partition ownership/epochs, health/capacity summary, coverage and capture age. | +| `GET /api/v1/fleets/{fleet}/nodes` | Paginated exact physical/boot identities, capabilities, observed health, desired intent, obligations and blockers. | +| `GET /api/v1/fleets/{fleet}/cells` | Paginated Cell identity/incarnation, observed writer, cost, residence, eligibility and movement evidence. | +| `GET /api/v1/fleets/{fleet}/operations` | Paginated requests and canonical operation/attempt links. | +| `POST /api/v1/fleets/{fleet}/operations` | Idempotent durable maintenance, CellMove, ReturnToService, ExtendDeadline, StopScheduling, ResumeScheduling, EnsureCapacity, RollingUpgrade or RetireFleet request. Advertise only qualified kinds. | +| `GET /api/v1/fleets/{fleet}/operations/{id}` | Durable request state, canonical phase, independent release/activation/recovery counts, unknown effects, obligations and offline proof. | +| `GET /api/v1/fleets/{fleet}/policy` | Revision and supported automation capabilities. | +| `PUT /api/v1/fleets/{fleet}/policy` | Validated complete policy replacement with expected revision and idempotency key. | +| `GET /api/v1/fleets/{fleet}/events` | Authorized server-sent events from committed event outbox. | +| `GET/PUT /api/v1/fleets/{fleet}/spec` | Revision-checked declarative fleet/pool/disruption specification with drift and preflight. | +| `POST /api/v1/fleets/{fleet}/plans` | Bounded dry-run maintenance, capacity or rollout plan with expiring evidence basis. | +| `GET /api/v1/fleets/{fleet}/explanations` | Authorized placement/blocker explanation for one exact target and observation basis. | +| `POST /api/v1/fleets/{fleet}/operations/{id}/control` | Idempotent cancel-unstarted, stop-new-children, extend or resume command; native effects retain their settlement rules. | +| `GET /livez` and `GET /readyz` | Process liveness and API readiness; partition ownership is a separate diagnostic. | + +Require `Idempotency-Key` for mutations and an expected revision in their body. +Document `202` queued, `200` replay/read, `400` invalid, `401` unauthenticated, +`403` unauthorized, `404` absent within authorized scope, `409` revision/key +conflict, `413` body too large, `422` unsupported operation/stage, `429` bounded +capacity, and `503` unavailable/unknown commit outcome. Error bodies carry a +stable code, request ID, retryability, and sanitized source diagnostic. Unknown +commit responses instruct clients to retry the same identity. + +Proposed maintenance request and acceptance, shown as schema examples: + +```json +{ + "kind": "NodeMaintenance", + "node_id": "canonical-node-id", + "expected_intent_revision": 17, + "expected_policy_revision": 4, + "deadline_ms": 1790900000000, + "reason": "planned hardware service" +} +``` + +```json +{ + "request_id": "generated-request-id", + "state": "Queued", + "operation_id": null, + "intent_committed": false, + "safe_to_take_offline": false, + "status_url": "/api/v1/fleets/example/operations/generated-request-id" +} +``` + +Identifiers and time in these examples are placeholders, not executable valid +requests. `openapi.yaml` must define canonical ID encoding, numeric ranges, +supported request variants, response versions, nullability, and error schemas. + +### Paging and event delivery + +Limit public pages to 128 rows and encoded responses to 1 MiB. Cursors bind +scope, filters, sort key, snapshot/projection revision, and schema version. Pin a bounded-lived immutable snapshot for paging. Unrelated current writes +do not invalidate it. Expiry or unsupported snapshot versions require an explicit +restart response; do not silently combine pages. Show capture time separately from response time and journal revision. +Strong operation reads come from the primary journal, while projected fleet +views disclose their event watermark and missing coverage. + +SSE event IDs bind fleet, feed generation and partition cursor vector. +Ordering is guaranteed within a partition and within each operation; a merged +fleet feed does not claim global transaction order. Replay from the retained +outbox; duplicates are permitted and clients deduplicate. An expired cursor +returns an explicit reset requirement followed by full refresh. Start with 1,024 +streams per controller, 64 queued events per stream, and a 1 MiB queue byte cap; +disconnect lagging clients rather than growing memory. Notifications wake readers +but are not the durable event record. + +### Proposed internal node routes + +Serve fleet observation capture/pages, effect dispatch, and fresh inspection +under `/internal/fleet/v1/`. Use mutual authentication binding a peer to enrolled +scope and exact boot session. Check version, payload size, digest, deadline, +registry revision, controller authorization and intent before the existing host +method. Keep viewer credentials separate from controller-to-node permissions. + +Observation capture is read-only and request-bound; capture nonce, exact boot, +roster revision, original time interval, and signature remain attached to pages. +Do not cache an effect result as a fresh observation. Effect dispatch carries +the canonical bounded `FleetAction` encoding. Inspection carries its canonical +request and validates the returned observation against it. Requests carry Cell +identity and trusted catalog references, never receiver filesystem paths. + +Application node endpoints must work during maintenance while new role admission +is closed. Accepted native effects are owned by host facilities across HTTP +disconnects. Dropping the HTTP future must not release credit, cancel native +work, or erase journal acceptance. Remote capture ownership and deadline behavior +are part of the observer gate, including unresolved producer jobs. + +## Management Cells and UI + +Define a management application namespace with a declared catalog and schema. +Use a sparse set of management Cells keyed by fleet, projection generation and +stable bucket. Start with 16 buckets and split through a versioned manifest up +to 256 before a bucket exceeds its qualified size; never funnel all fleet events +through a single Cell. Keep audit mirror segments bounded. +Store event sequence, canonical record references/digests, model schema and +capture metadata. Large telemetry stays in the application metrics backend. + +Projection applies each event through an idempotent Cell command whose outcome +and model mutation commit together. Use `(partition, sequence, projection generation)` as the command identity and +track independent watermarks for source partitions. A projector acknowledges a +journal event only after every affected bucket has a durable receipt; partial +application replays idempotently after failure. Resolve ambiguous +command outcomes through the ordinary receipt/outcome path. Only persist a +journal projection cursor after durable Cell acknowledgement. On rebuild or +incarnation change, replay retained events or rebuild from a consistent journal +snapshot and resume from its watermark vector. Publish a new immutable page +manifest only after all affected buckets reach the declared cut. Public paging +pins that manifest; it does not combine mutable bucket heads into a claimed +consistent fleet snapshot. Canonical operation reads continue to use the journal. + +Projection failure delays display but cannot acknowledge an operator mutation, +authorize a node action, or block journal takeover. Audit source records remain +in the journal; management Cells are rebuildable mirrors. This avoids a +distributed transaction between a Cell command and the fleet journal. + +### Required UI views + +| View | Information and controls | +| --- | --- | +| Fleet overview | Controllers and owned/takeover-eligible partitions, epochs/expiry, compatible builds, coverage, capacity, pressure, slow/unhealthy counts and blocked operations. | +| Nodes | Physical node and boot, role/failure domain, independent health/pressure/intent, sample age, writer/reader/follower responsibilities, maintenance request. | +| Cells | Exact incarnation/owner, resource cost, class performance, eligibility, blockers, historical move outcomes, qualified move request. | +| Operation detail | Durable request timeline, canonical phases, release versus activation/recovery, pending/unknown actions, remaining role obligations, deadline extension and offline proof. | +| Policy and audit | Current revision/stage, fleet specification and drift, validated edits, decision reasons, principal/request identity and immutable transition history. | +| Pools and lifecycle | Actual versus desired capacity, quota/headroom/disruption budget, bulk maintenance/upgrade progress, canary results and exact blocked child. | +| Service health | Control plane dependencies, inventory lag, backup/restore status, certificate expiry, compatibility and redacted diagnostics. | + +Display stale/partial/unknown values visibly. A missing node is not shown as +zero load; released is not rendered as serving elsewhere. Disable unsupported +actions using server capability metadata and explain their blockers. Reconnect +SSE by cursor, recover with full refresh after reset, and render API revision +conflicts with an explicit reload/retry workflow. Include keyboard navigation, +accessible status text, mobile layouts, and pagination/virtualization for large +fleets. The operation page reads canonical journal status even if projections +are unavailable. + +## Ordered implementation packages + +Each package produces reviewable commits and updates a release evidence ledger. +The changed production scope supersedes the earlier reference-only package +sequence. Implement within these module boundaries and preserve one canonical +runtime path. Framework API/codec changes require reading each nearest crate +guide and every producer/consumer before editing. All new names below are +proposed; actual framework gaps remain prerequisites, not completed features. + +| Package | Concrete commit sequence and source boundaries | Required exit evidence | +| --- | --- | --- | +| CP0 Contracts and release ledger | Freeze current codecs and inventory required framework gaps; define partition/generation/snapshot V2 envelopes and migration matrix; add the machine-readable release profile and scenario ledger. | Every advertised capability maps to an implemented path, qualification scenario and supported version; no silent V1 reinterpretation. | +| CP1 Product workspace and worker SDK | Create `apps/fleet-controller` and `agent/`; implement configuration/identity validation; integrate the single CellNode, standard facility binding and management listener; add management catalog and code eligibility. Complete native enrollment after CP2 gateway core. | A clean sample application integrates with supplied SDK, receives/drains a real Cell, preserves receipts and needs no custom fleet adapter. Worker acquisition of management code is refused. | +| CP2 Journal gateway and atomic records | Implement versioned PostgreSQL migrations and all three journal traits plus request/policy/outbox records; add primary-time validation, node-scoped gateway, retained transaction jobs and bounded pools; qualify database promotion. | Independent-client/process races, lost replies and original replay; unauthorized node cannot read/write another scope; no lost acknowledged records under supported primary failure. | +| CP3 Scalable complete inventory | Extend host roster/observer and canonical enrollment contracts with immutable snapshots, transactional delta/index updates and closure barriers; implement native role capture; add terminal archive lookup/exclusion and indexed cooldowns. | Native role completeness including failed owners; no false completion under producer races; more than ten million terminal records with bounded active scan work; gaps trigger bounded scope rebuild. | +| CP4 Partition leases and shared permits | Extend runtime fleet records/reducer, host journal view and reconciler with assignment epochs and per-partition leases; add global/fleet/node/disruption permits, crossing reservation transaction and safe repartitioning. | Two owners racing, cross-partition loss at every boundary, no overcommit, late envelope fenced, one canonical attempt, healthy partitions progress independently. | +| CP5 Controller availability and fairness | Implement assignment, bounded passes, reserved settlement/renewal capacity, fair queues, jitter and controller maintenance adoption; wire supervisor and canonical shutdown. | Three real controllers distribute leases; one failure remains within takeover/load target; noisy tenant cannot starve another; dropped shutdown waiter retains accepted native work. | +| CP6 API and declarative workflows | Commit OpenAPI, authentication/scope middleware, FleetSpec/preflight/dry-run/explanation routes; implement durable requests/bulk parent cursor/control operations, snapshot paging and partition-cursor SSE. | Idempotency, revision conflict, stale selector, cancel/stop semantics, scope isolation, stream overflow/reset and ambiguous commit are correctly reported. | +| CP7 Health and placement automation | Implement workload-aware windows/checkpoints; explicit planner policy basis, receiver exclusions, settled CellMove and slow trials; add effectiveness/cooldown/circuit-breaker rules. | Sustained relief, no fabricated pressure/failure proof, no oscillation, real receiver admission, policy race fenced, fleet-wide dependency slowdown does not create relocation storms. | +| CP8 Capacity and disruption control | Implement durable pool capacity/provider requests and Kubernetes StatefulSet adapter; add failure-domain/redundancy/headroom policy, exact-instance removal, canary upgrade and rolling/bulk cursors. | Capacity becomes usable only after enrollment; quota/backlog explained; stale resource-version cannot delete another boot; unexpected failure consumes disruption budget; rollout resumes safely. | +| CP9 Full maintenance and recovery | Complete fleet W6–W8 native primitives, reader/follower/producer settlement and finalization; consume through SDK/API; implement return-to-service, replacement and fleet retirement. | All supported primitives and foreign tails covered; traffic cannot starve drain; offline proof matches current boot/instance/intent; controller handover precedes its drain. | +| CP10 Management Cells and UI | Implement bucket manifest, idempotent projector/split/rebuild and audit mirrors; deliver fleet/nodes/Cells/pools/operations/policy/service-health screens and accessible workflows. | Projection crash/split cannot lose canonical events; canonical operation fallback; large lists and streams remain bounded; operator can complete baseline scenarios through UI/CLI. | +| CP11 Identity upgrade and disaster recovery | Implement workload identity bootstrap/rotation/revocation, tenant quotas, endpoint validation and diagnostics; add external recovery-generation fencing, restore reconciliation, schema migration and mixed-version rollout. | Revoked/untrusted peers fail closed; old controller generation rejected after stale restore; unknown obligations block only affected scopes; old/new binaries cannot weaken evidence. | +| CP12 Native fault and scale qualification | Deliver subprocess/provider runner and artifacts; run all fault, compaction, scale, noisy-fleet, migration, restore and 72-hour soak profiles; retain unchanged existing qualification gates. | Full matrix below, published measured resource/SLO envelope, zero unresolved safety failures, explicit disposition for performance/availability failures. | +| CP13 Packaged production release | Ship signed binary/image/SDK/CLI and UI, Helm/VM profiles, preflight/doctor, backup automation, dashboards/alerts, runbooks and compatibility contract; run independent operator usability exercise. | Fresh supported deployment and routine lifecycle operations use shipped workflows; release capability manifest enables all baseline features only after CP12 evidence. | + +Dependency order: CP0; CP1 scaffold and CP2 core; finish CP1 native integration; +CP3 and CP4; CP5/CP6; CP7/CP8/CP10; CP9/CP11; CP12; CP13. Source work that completes +fleet W1–W10 occurs in the canonical host/runtime modules and is consumed by +these packages. CP8 provider actions remain disabled until CP9 can prove their +required native settlement. CP11 recovery-generation fencing is mandatory before +any production deployment, including the small profile. + +### Framework changes that must be explicit + +| Existing source | Required change | +| --- | --- | +| `crates/cellule-runtime/src/fleet/operations/` | Versioned partition/generation authorization, shared-permit references, closure/snapshot/archive contracts and pure transition rules. | +| `crates/cellule-runtime/src/fleet/placement/` | Explicit policy basis, cross-partition candidates, workload/failure-domain exclusions and reasoned operator moves; preserve trusted observations. | +| `crates/cellule-host/src/fleet/controller.rs` | Partition-aware journal view and atomic authorization/snapshot semantics; preserve source errors and immutable replays. | +| `crates/cellule-host/src/fleet/roster/` | Replace whole-history vectors with bounded canonical manifest/delta traversal and complete closure proofs. | +| `crates/cellule-host/src/fleet/reconciler/` | Reuse one bounded driver under partition leases; consume shared permits, current policy and incremental observation evidence. | +| Existing host/runtime inventory and enrollment producers | Atomically publish every native role's pending/established/terminal changes and close the exact barrier for finalization. | +| `apps/fleet-controller/` and `agent/` | All HTTP, identity, PostgreSQL, platform, UI and deployment integration; library core stays provider neutral. | + +### Milestones and production definition + +M1 delivers a runnable single-partition product with SDK, durable API and truthful +partial status. M2 adds complete inventory, partition safety and qualified +movement. M3 completes capacity, all baseline maintenance, identity, recovery +and UI. M4 passes the small and large release profiles, fault matrix, soak and +operator acceptance. M1–M3 are development milestones; they cannot be labeled +fully functional production control plane releases. + +Production baseline includes overload relief, unhealthy-node exclusion and +canonical recovery coordination, qualified slow-node trials, balancing, Cell +moves, worker/controller maintenance, capacity control, rolling upgrade, durable +bulk workflow, identity rotation, backup recovery and diagnostics. A provider +may be unsupported, but a baseline capability cannot be deferred behind a +permanent feature flag while the release claims completion. + +## Commands and executable acceptance + +Commands in this subsection are implementation targets until their packages +land. Every command must be implemented and documented, return nonzero on +failure, and write a bounded evidence summary. Invocations below assume the +workspace root and an isolated verification checkout for process/provider work. +Do not run broad process suites in the active development checkout. + +CP1 installs the application manifest. CP5 supplies the initial `dev-up` harness; +CP13 packages it to provision a local PostgreSQL fixture, canonical +object/authority provider, development identities and three controllers plus +six worker processes across two execution partitions. Fixture credentials +are local only; real-provider qualification uses its documented environment. + +```sh +export CARGO_INCREMENTAL=0 +export CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/cellule-controller-verification" +cargo build --manifest-path apps/fleet-controller/Cargo.toml --workspace --locked +cargo run --manifest-path apps/fleet-controller/Cargo.toml --locked --bin fleet-controller -- dev-up --state-dir /tmp/cellule-controller-demo +``` + +The fixture prints API/UI addresses, fleet ID, enrolled sessions and supported +capabilities. CP6 supplies an authenticated client using a credential file with +owner-only permissions; do not print tokens or place them in command arguments. + +```sh +cargo run --manifest-path apps/fleet-controller/Cargo.toml --locked --bin cellule-fleetctl -- --config /tmp/cellule-controller-demo/client.toml fleet-status +cargo run --manifest-path apps/fleet-controller/Cargo.toml --locked --bin cellule-fleetctl -- --config /tmp/cellule-controller-demo/client.toml maintenance --node worker-1 --idempotency-key maintenance-worker-1 +cargo run --manifest-path apps/fleet-controller/Cargo.toml --locked --bin cellule-fleetctl -- --config /tmp/cellule-controller-demo/client.toml operations +``` + +`worker-1` is a fixture alias resolved to its canonical NodeId; the client fetches +and displays expected intent/policy revisions on first submission, and retains +that immutable body for retries using the same key. Identical input returns +the same request without a new deadline or refreshed expected revision. An +incomplete maintenance implementation reports its +unsupported capability or retained blocker and cannot pass the full scenario. + +CP12 delivers these named scenarios, all using actual subprocesses and the same +production application routes/adapters. The runner uses a disposable isolated +state directory, cleans up only its owned processes/resources, and preserves +failure artifacts. + +```sh +cargo run --manifest-path apps/fleet-controller/Cargo.toml --locked --bin fleet-controller -- qualify --scenario controller-failover +cargo run --manifest-path apps/fleet-controller/Cargo.toml --locked --bin fleet-controller -- qualify --scenario pressure-convergence +cargo run --manifest-path apps/fleet-controller/Cargo.toml --locked --bin fleet-controller -- qualify --scenario slow-node-trial +cargo run --manifest-path apps/fleet-controller/Cargo.toml --locked --bin fleet-controller -- qualify --scenario worker-maintenance +cargo run --manifest-path apps/fleet-controller/Cargo.toml --locked --bin fleet-controller -- qualify --scenario controller-maintenance +cargo run --manifest-path apps/fleet-controller/Cargo.toml --locked --bin fleet-controller -- qualify --scenario journal-failover +cargo run --manifest-path apps/fleet-controller/Cargo.toml --locked --bin fleet-controller -- qualify --scenario receiver-loss +cargo run --manifest-path apps/fleet-controller/Cargo.toml --locked --bin fleet-controller -- qualify --scenario projection-rebuild +cargo run --manifest-path apps/fleet-controller/Cargo.toml --locked --bin fleet-controller -- dev-down --state-dir /tmp/cellule-controller-demo +``` + +`pressure-convergence` sustains reproducible load across multiple batches; +it does not stop the producer after the first two moves. Record time to relief, +admission pressure and actual latency separately. A permanently saturated fleet +may end BlockedCapacity; it cannot claim convergence or continue unsafe moves. +`slow-node-trial` injects node-specific latency as well as a separate fleet-wide +backend slowdown and proves the latter does not trigger relocation storms. + +The production runner also implements the following release suite. It uses +fixed manifests for `small` and `large`, and requires all baseline capabilities; +unsupported or skipped required scenarios cause a nonzero result. + +```sh +cargo run --manifest-path apps/fleet-controller/Cargo.toml --locked --bin fleet-controller -- qualify --profile large --all-required --artifacts /tmp/cellule-control-plane-evidence +cargo run --manifest-path apps/fleet-controller/Cargo.toml --locked --bin fleet-controller -- verify-evidence --profile large --require-baseline --artifacts /tmp/cellule-control-plane-evidence +``` + +The suite includes `inventory-history-churn`, `continuous-enrollment`, +`cross-partition-move`, `repartition-race`, `noisy-fleet-isolation`, +`bulk-rolling-upgrade`, `scale-down-race`, `identity-rotation`, +`stale-backup-recovery`, `mixed-version-upgrade`, `sdk-onboarding`, +`large-native-profile` and `72-hour-soak`, in addition to the named fault cases. +Each artifact records profile/schema version, native/synthetic evidence type, +source and binary fingerprints, platform/database configuration, scenario seed, +measured distributions, exact invariant outcomes and cleanup status. +`verify-evidence` rejects mismatched revisions, missing runs, unsupported required +capabilities, synthetic substitution for native gates and failed thresholds. +The fixed hardware/network/Cell-size distribution is part of the versioned +profile; changing it creates a reviewed profile revision, not a silent pass. + +### Scoped verification routes + +Run checks only against an inventoried snapshot. Application manifests and +lockfiles are outside the root workspace, so verify them explicitly. CP10 adds +UI lockfile/build scripts and an API contract gate; CI must invoke both. + +```sh +cargo fmt --manifest-path apps/fleet-controller/Cargo.toml --all --check +cargo check --manifest-path apps/fleet-controller/Cargo.toml --workspace --all-targets --all-features --locked +cargo test --manifest-path apps/fleet-controller/Cargo.toml --workspace --all-features --locked +cargo clippy --manifest-path apps/fleet-controller/Cargo.toml --workspace --all-targets --all-features --locked -- -D warnings +RUSTDOCFLAGS='-D warnings' cargo doc --manifest-path apps/fleet-controller/Cargo.toml --workspace --all-features --no-deps --locked +python3 scripts/check-boundaries.py +python3 scripts/check-module-layout.py +python3 scripts/check-doc-rust-fences.py +python3 scripts/check-doc-links.py +``` + +Update static gates in CP1 so the standalone application sources, documentation, +Cargo manifest/lockfile and UI assets are covered; current workspace scans must +not silently omit the new application. Framework changes run the scoped host +and runtime suites plus root CI routes from `AGENTS.md`. Provider, process, +cloud and broad qualification require their existing documented environments. +Do not weaken latency, throughput, fault, or compatibility profiles to pass. + +## Acceptance matrix and required evidence + +| Scenario | Required result | +| --- | --- | +| Three controllers race | One current journal claimant per partition; no duplicate permit allocation; all replicas can return committed request/status. | +| Kill active controller during accepted move | Successor adopts original action, permits and evidence; receipt-bound state survives; no second release or writer. | +| Pause old controller then resume after expiry | New epoch remains authoritative; stale new allocations/effects refused; already accepted work may complete and is adopted. | +| Journal unavailable or promoted | New effects stop without current authorization; no acknowledged request/acceptance disappears under the qualified database failure envelope. | +| Request response lost after commit | Same identity returns original request/operation; unknown response never manufactures rollback. | +| Receiver lost before or after release | Source stays authoritative before confirmed release; after release, exact root is recoverable through canonical paths; unused credit is joined. | +| Missing or forged node sample | No false completeness or proactive receiver; original retained ownership/obligations remain visible. | +| Slow management endpoint with serving worker | Report reachability suspicion; no fabricated failed-session proof, writer takeover, or maintenance success. | +| Slow hot Cell or fleet-wide backend | Effectiveness gate stops trials; expose workload/capacity diagnosis; no repeated relocation storm. | +| Oscillating pressure/performance | Dwell, minimum residence, cooldown, shared budgets and post-batch freshness bound movement. | +| Sustained overload | Qualified multi-batch relief or explicit no-capacity blocker; actual workload state remains readable through receipts. | +| Policy update races allocation/acceptance | Exact revision checked in the mutation transaction; new work obeys policy; accepted work is preserved. | +| Busy node maintenance | Accepted foreground work settles; native reader/follower/producer obligations close; Stopped and withdrawal evidence precede offline flag. | +| Dead owner with follower tail | Tail remains protected until object coverage or canonical recovery and exact retirement; zero local writers is insufficient. | +| Controller pool maintenance | Another eligible controller adopts before target drain; target cannot regain eligibility from stale metadata or reboot. | +| Controller majority lost | Optional automation may be unavailable; canonical data safety holds; no application quorum or automatic database promotion is inferred. | +| Projection unavailable or rebuilt | Canonical status/mutations still use journal; Cells rebuild receipt-bound views without duplicate command effects or unsafe mutation authority. | +| SSE overflow or expired cursor | Bounded memory, explicit disconnect/reset, successful consistent refresh; no event loss presented as complete history. | +| Mixed versions and rollback | Unsupported codecs/capabilities fail closed; new state is not read by an incompatible binary; rollback cannot delete accepted work. | +| Clock jumps and shutdown cancellation | Authorization fails closed, no stale lease allocation, retained finite owners remain joinable, no leaked native handles/credit/tasks. | +| More than ten million terminal records | Active scans and candidate/cooldown queries stay within the same bounded work envelope; archived replay never recreates a terminal enrollment. | +| Continuous enrollment during observation | Immutable snapshot remains usable; relevant deltas are revalidated; no endless whole-fleet restart and no false closure. | +| Cross-partition movement and repartition | One attempt, atomic aggregate charges, exact receiver credit, assignment fencing, and safe adoption through crashes. | +| Noisy fleet at quota | Other fleets meet the fairness/availability target; DB, RPC, memory, stream and inventory limits hold. | +| Bulk upgrade with mid-run failure | Original selector and cursor preserved; only qualified children start; disruption and canary stop rules hold. | +| Stale provider scale-down | Exact instance/boot/intent/resource-version mismatch blocks destructive action; no healthy successor is removed. | +| Identity expiry, revocation and key rotation | New unauthorized work denied; renewal works through a rolling change; existing obligations remain visible and safely settled. | +| Fleet import and retirement | Bootstrap cannot omit unknown responsibilities; retirement cannot erase data/evidence through deployment cleanup. | +| Independent operator exercise | Integrate one sample app within one engineer-day using only shipped SDK/docs; install the supported profile and complete normal lifecycle work without custom fleet logic or direct DB edits. | +| Database backup restore | Refuse normal automation until restored journal and authority are reconciled; old snapshot cannot erase newer action obligations or revive stale epochs. | + +Evidence for each run includes source commit and dirty-source fingerprint, +manifests/lockfiles, binary and UI digests, provider/database versions and +configuration, scenario seed/load, actual enrolled boot identities, journal +epoch/revision timeline, action/authority evidence, acknowledged command receipts +and readback, resource maxima, cleanup result, raw logs, and terminal failures. +Record functional, performance, availability and safety results separately. +Unexplained existing failures remain tracked; a later pass does not establish +their cause. No test count by itself certifies complete role coverage. + +Proposed fixture availability target: under a healthy journal, bounded skew, +and reachable peers, controller takeover is visible within 60 seconds of active +process loss. This target includes the 30 second lease plus detection/claim/pass +time and must be measured. It is not a bound on recovery when the journal or +required role evidence is unavailable. Publish measured fleet-size and API/load +limits; this plan makes no unmeasured scalability claim. + +## Security recovery and release operations + +### Identity and tenant isolation + +Ship an OIDC verifier for operator identity and a standard workload-identity to +short-lived mTLS certificate exchange for nodes. Bootstrap enrollment binds the +platform instance UID, tenant/application/fleet, physical NodeId, exact boot, +role and declared capabilities. Bootstrap tokens are scoped, short-lived and +single-use. Renewal happens before expiry with overlapping trusted keys; +revocation fences future management authorization while retaining existing +native obligations. Identity renewal and trust-bundle rollout have fault tests, +including a disconnected node returning with an obsolete key. + +Use tenant-scoped keys and authorization on every route, journal gateway call, +object lookup, event subscription and archive retrieval. Controller capability +is not an operator administrator credential. Provide Viewer, Operator, +Administrator and narrowly scoped platform-service permissions, with actor and +reason recorded for changes. Revalidate that a queued actor/request remains +authorized at adoption; revocation stops unstarted work without erasing accepted +native effects. Permission changes and evidence reads are audited. + +Controllers discover peer endpoints from authenticated enrollment bound to +platform network policy; users cannot provide arbitrary URLs or local paths in +move requests. Validate URI scheme, network allowlist and peer identity before +connection, and bound redirects or disallow them. Protect browser sessions, +CORS, CSRF, response headers and UI rendering; serve a restrictive CSP and no +raw backend error strings. Encrypt credentials at rest through the deployment +secret manager. The diagnostic bundle redacts secrets and application payloads +and reports only metadata needed to trace a specific operation. + +### Backups and recovery generations + +Automate database backups, WAL retention, archive verification and restore +exercises using the qualified provider +[continuous archive and recovery contract](https://www.postgresql.org/docs/current/continuous-archiving.html). Publish the data included, encryption/key dependencies, restore size, +RPO and RTO in the deployment profile. Single-controller or database-primary +failure targets zero lost acknowledged control records under the synchronous +replication/fenced-promotion contract. Regional disaster has a separate proposed +journal RPO of five minutes and operator recovery target of 60 minutes at the +published backup size. Application Cell data has its own storage durability +contract; control-journal RPO is not an application-data guarantee. + +A restored older journal cannot safely reuse the prior controller epochs. +Introduce a `control_plane_generation` anchored by monotonic CAS in independent +canonical fleet authority storage, outside the restored journal backup. Bind +all management envelopes, gateway authorization and node agent admission to it. +Recovery obtains a newer generation only after old controller identities are +fenced by the same authority. First effect acceptance requires a fresh authority-generation check, never a +UI or telemetry cache. Recovery must prove all old journal writers/gateways are +fenced and invalidate outstanding unaccepted authorization channels before +activating a new generation. If this exclusion cannot be proved, keep mutation +suspended. Accepted old effects remain obligations through this barrier; the +protocol must model the race between generation change and first acceptance. +Nodes fail closed when current generation cannot be established. This is +management fencing and does not replace Cell writer authority. + +Restore starts in suspended recovery mode. Enumerate enrolled boots, native +accepted work, pending enrollments, canonical writers and foreign log obligations +against the recovered journal and archive roots. Missing historical acceptance +or result remains Unknown, with conservative permits and its affected scope +quarantined. Never manufacture a clean release from a later authority root. +Resume a partition only after every relevant pre-restore obligation is proved +settled or adopted under canonical recovery. If evidence cannot be recovered, +keep that scope blocked and expose the exact proof needed. The RTO target applies +to restoring management service; unresolved native safety obligations may take +longer and must be reported separately. + +### Software and schema lifecycle + +Publish signed container/binary/SDK artifacts, SBOM, provenance, pinned lockfiles, +protocol compatibility matrix and upgrade/rollback instructions per release. +Support the current and previous minor protocol versions during a documented +rolling upgrade window. Persisted-format compatibility is separately declared; +unknown critical fields fail closed. Capability negotiation cannot silently +downgrade evidence requirements. Apply expand/backfill/verify/activate/contract +schema migrations with durable progress, bounded batches and restart safety. +Activation waits until every affected producer and consumer is compatible. + +Each release includes reproducible install/upgrade commands, automated preflight, +health checks, database/archive migrations, dashboards and actionable alerts. +A `doctor` command checks configuration, identities, network, storage, inventory +coverage, queue pressure and the longest blocked obligations, then produces a +redacted support bundle. It performs no mutation unless an explicit supported +repair operation is requested. Repairs use the ordinary journal/evidence path. + +### Service operation objectives + +Measure public API availability separately from optional scheduling and actual +Cell serving. Set a proposed 99.9 percent monthly management API availability +target under the supported deployment; expose all dependency outages in reports. +Monitor fresh observation coverage, lease renewal margin, time since useful +reconciliation, unknown-action age, drain completion time, restore headroom, +queue wait, inventory gaps, projection lag, archive/backups and key expiry. +Use bounded metric labels; exact Cell/request identities belong in event/log +records. Supply traces that link request, partition, attempt, node acceptance, +canonical publication and final evidence. + +Alerts group by cause and affected scope, include a runbook and next permitted +action, and clear when evidence proves recovery. Avoid paging for every retry or +for healthy standby controllers. Notify on sustained inability to preserve +service objectives, stalled critical maintenance, threatened durability, +credential/backup expiry and capacity exhaustion. No automatic repair may +weaken a safety or qualification gate to suppress an alert. + +## Deployment and operator runbooks + +Deploy controllers separately from application workload capacity, reserve host +resources for API/reconciliation and management Cells, and spread the three +instances across failure domains. Set explicit pools, connections, scan/page +bounds, queue limits, timeouts, and telemetry cardinality. Startup validates +configuration, capabilities, identity scope, immutable profile, and schema +versions before opening mutation readiness. + +Runbooks must cover bootstrap/import, capacity addition, pressure with no +receiver, hot Cell diagnosis, stuck/unknown operation, worker and controller +maintenance, journal outage/promotion, certificate rotation, incomplete role +coverage, failed-owner follower obligations, deadline extension, backup restore, +upgrade, rollback and return to service. Each states the API status/evidence +required and the action permitted. A journal restore is an incident requiring +reconciliation with current authority and retained node acceptances before +automation resumes; deployment must fence old controller sessions. + +Roll out observation-only API/UI, then pressure canaries, maintenance canaries, +slow trials, and full balancing. Deploy compatible readers and node adapters +before writing new persisted or signed formats. Disable new allocations during +incompatible transitions, while retaining resolution paths. Rollback preserves +committed intents, accepted actions, and immutable history; use the last binary +that understands the current record formats. + +## Completion checklist + +- [ ] Supported controller/CLI/UI/worker SDK artifacts build reproducibly and host partitioned management Cells. +- [ ] Application onboarding and normal operations require no custom fleet adapters or direct journal edits. +- [ ] Partition leases, crossing reservations and aggregate disruption/resource permits pass race and takeover gates. +- [ ] Incremental complete inventory and terminal archive replay remain bounded through historical churn. +- [ ] Three controller processes serve authorized API/UI and prove fenced adoption. +- [ ] Production journal implements all fleet/application transactions with qualified promotion and restore behavior. +- [ ] Complete fresh observation covers retained boots and all native role obligations. +- [ ] Slow, overloaded, unhealthy, unknown and maintenance states are independently visible. +- [ ] Pressure relief, explicit Cell moves, slow trials and balancing share canonical actions and budgets. +- [ ] Worker and controller maintenance finish only with complete native and withdrawal evidence. +- [ ] OpenAPI, bounded pagination/events, management Cell rebuild and accessible UI are delivered. +- [ ] All named process/provider scenarios and unchanged qualification gates have recorded results. +- [ ] Capacity, bulk maintenance, upgrades, identity rotation, generation-fenced recovery and retirement are complete. +- [ ] Small and large scale/SLO/soak profiles have measured passing evidence, including one-controller loss. +- [ ] Deployment, compatibility, incident and rollback runbooks pass the independent operator exercise. + +Production release requires every baseline gate above. A completed design, +local fixture, partial native implementation or synthetic scale test cannot +establish production readiness. diff --git a/docs/fleet-operations-plan.md b/docs/fleet-operations-plan.md new file mode 100644 index 00000000..d3386d13 --- /dev/null +++ b/docs/fleet-operations-plan.md @@ -0,0 +1,1744 @@ +# Cellule fleet operations design and implementation plan + +Status: design and execution plan; partial foundations exist, fleet execution +and qualification remain incomplete. +Source baseline: `e07670e2348231ed401cc7280a47e3ab97596ffe`. +Prepared: September 30, 2026, America/Vancouver. +Updated: October 3, 2026, America/Vancouver. + +Build one reusable fleet reconciliation path for automatic ownership balancing, +sustained pressure relief, and planned node maintenance. Reuse Cellule's fenced +writer, actor, publication, recovery, and node lifecycle. Applications supply +the recurring loop, operator authorization, peer endpoints, deployment policy, +and a durable operation journal adapter. + +The implementation is complete only when a reference fleet demonstrates an +overloaded node shedding eligible Cells, a maintenance node evacuating its +writers and durability obligations, and a restarted controller continuing an +operation without losing acknowledged state or creating a second writer. + +This document supplies the decisions, proposed interfaces, failure semantics, +work packages, and verification routes needed to implement that behavior. +Names identified as proposed below do not exist at the baseline revision. + +## Implementation brief + +The deliverable is a library orchestration path, a runnable three-node example, +and an application integration recipe. An embedding application installs the +node adapters once, runs a supervised reconciliation loop, and exposes durable +maintenance requests and paginated status. The framework owns the safety checks +and finite accepted work. Automatic balancing and operator maintenance use the +same action and evidence contracts. + +| Implementer question | Decision | +| --- | --- | +| What ships first? | Settled-Cell movement through W1–W5 and the first overload example. Full maintenance follows W6–W8; deployed automation follows W9–W10. | +| Which algorithm? | Reuse the current weighted placement and pressure classifier; add execution and evidence around them. | +| Which external facilities are required? | A linearizable journal, complete enrollment registry, authenticated management transport, and trusted local Cell inputs. The reference example supplies these behind the same contracts. | +| What keeps operations simple? | One operation ID, one status contract, one application loop, bounded defaults, and one canonical host drain lane. | +| What proves success? | Receipt-preserving actor activation on another node. Maintenance additionally requires settled role obligations, host Stopped, and session withdrawal. | +| What is the next concrete change? | Resolve the measured routing command throughput regression under the unchanged profiles. Then aggregate exact original-writer/suffix successor proofs and complete authenticated observations, and connect reader/follower policy barriers to SettleRoles/Finalize. Keep cross-session recovery and the remaining failure/inspection gates in scope. | + +Start with the [first slice commit sequence](#first-slice-commit-sequence). +Every increment must expose a reviewable public behavior and retain its test +evidence. The remaining sections specify the complete handoff; implementation +does not depend on earlier chat messages or an external embedding codebase. + +Here, executable means an implementation handoff with ordered changes, +commands, and observable exit conditions. The complete balancer and maintenance +workflow are delivery targets. Commands for a proposed scenario become runnable +when its package lands; the journal inspection command is available now. + +### Immediate execution checklist + +Use this checklist for the next end-to-end increment in the current working +tree. The detailed contracts later in this document govern each step. + +1. **Establish the checkpoint.** Read the root and affected crate guides, + inventory the current diff, and record the source fingerprint. Compare + existing code with W1–W4 before adding another implementation of a contract. +2. **Finish observation coverage.** Reuse the reference boot producer and host + startup barrier and managed reader enrollment binding; wire follower producers to + `FleetEnrollmentJournal`. + Compose bounded ownership and role + pages into `FleetObserver`; compare the membership and registry revisions + before and after collection. An incomplete scan returns a typed blocker. +3. **Connect transport and driver.** Implement the application's `FleetTransport` + against the existing `FleetReconciler` under host `fleet/`. Route effect dispatch to + `apply_fleet_action` and fresh capture to `inspect_fleet_action`. Consume + the existing planner and reducer, with one journal-backed count/byte budget. +4. **Connect the durable example.** Use the existing `SqliteJournal`, three + independently leased `CellNode`s, signed local observations, and a trusted + Cell provider. Invoke the exported driver; the example contains no second + placement or movement state machine. +5. **Prove the first movement.** Write acknowledged SQL state, make one node + a sustained donor, reserve an eligible receiver, release and activate, then + verify receipt-bound readback. Drop an action reply and reconstruct the + controller from the same journal. Assert one writer and retained permits + until serving and resource settlement are proved. +6. **Deliver and record.** Add the `overload` example command, run the relevant + focused gates below, and retain selected test counts and source evidence. + Proceed to busy maintenance and role evacuation only with this public path + available as a regression scenario. + +The first increment is done when an implementer can run `overload`, observe +separate release and activation progress, reconstruct the controller, verify +acknowledged state, and close every owned task and reservation. Full maintenance +additionally requires W6–W8; deployment requires W9–W10. + +For execution from this working tree, begin by inventorying the current +implementations and pending changes listed below. Reuse the W1 registry contracts and strict local journal +adapter, connect the complete observer and W5 driver next, then deliver the +settled-movement example. Busy maintenance and safe node finalization follow +in W6–W8. The [commit sequence](#first-slice-commit-sequence) names each handoff +and the public behavior that must pass before proceeding. + +## Starting state for the implementer + +Inspection of this working tree through October 1, 2026 found the following +foundations. Inspect the current revision and diff before implementing each +package, reuse matching work, and run its gates before marking it complete. + +| Package | Implemented foundation | Remaining exit evidence | +| --- | --- | --- | +| W1 | Pure controller/maintenance/movement transitions; bounded versioned journal codecs; unknown-outcome permit retention; action envelopes, physical-node intent, progress pages, immutable accepted-action records, and distinct recovery basis/evidence/outcomes with complete replay comparisons. Registry records carry boot-bound enrollment, retained intent pages, a bootstrap revision, stop/resume state, and transactional allocation gates. The embedding example supplies one SQLite transaction domain for all three journal contracts, with independent-client races, lost replies and reconstruction tests. Fresh inspection requests/observations bind nonce, complete action, registry revision, endpoint and original capture interval. | Complete fleet/role observation envelopes and enrollment producer wiring; consume fresh inspections and the durable adapter through the public reconciler. Process/fault/provider qualification remains required. | +| W2 | Schema 3 operational decoding/signing; schema 2 bridge serializer and signing checks; canonical byte comparison after one decoder verification pass; monotonic sample checks; shared sticky cordon/reversible pressure gate; writer/read/follower/recovery candidate filtering; host follower gate installation; public pressure/cordon lifecycle test. | Complete signed-snapshot production, broader public node admission scenarios, and the mixed-binary rollout campaign. | +| W3 | Bounded actor ownership pages; worker measurements and conservative costs; residence and stable samples; mutation revision checks; persisted follower-lane pages; expired/fenced log-reference discovery; host managed-reader pages with canonical accepted-work lifetimes, local closure observations and complete bounded native-state continuation fingerprints. | Complete the observer's membership/enrollment barrier and host action consumption, including accepted queries and retained peer views. Run all W3 gates against the current diff before marking the package complete. | +| W4 | Runtime receiver preparation reserves actual Cell, memory, descriptor, affine SQL-job, and scoped LTX disk credit. The host journals source/receiver effects, confirms acquisition/recovery input before CAS and recovery position before actor admission, checks current serving, and owns work across dropped waiters. CleaningReceiver preserves release evidence and charged permits; unused credit can retire before ordinary admitted activation resumes. Public local tests cover receipt preservation, refusal, duplicates, basis faults, post-release cleanup, source loss, and ordinary-acquisition races. A separate fresh inspection path bypasses historical Inspect results and cannot start recovery; its finite jobs share the existing action bound and drain. Canonical runtime acquisition retains immutable exact claim input/materialization before admission. Native prepared-root lineage proves an exact per-movement released/recovered prefix across publication and compaction; complete bounded origin verification and fresh actor/authority checks gate serving evidence. Full original-writer/suffix aggregation remains required. | Complete recovery across receiver sessions and refusal/unknown reconciliation, complete observer consumption, maintenance role actions, durable production adapters, and all process/provider and W4 exit assertions. | +| W5 | Public caller-driven reconciler and observation/transport contracts; production calls to the existing planner and reducer; phase CAS before dispatch, fresh serving checks, cooldown/post-batch journal reads, and permit projections. SQLite-backed sequencing tests exercise simulated effects, lost replies, deadlines and competing drivers. The overload executable moves two real Cells across three leased runtimes; controller-restart changes claimant after actual lease expiry and adopts lost releases. Maintenance now dispatches exact journal-bound Cordon, commits Cordoned then BeginEvacuation, and applies retained intent before donor selection. Partial fresh observations can evacuate settled normal-pressure donors; operation deadlines bound new moves. | Complete producer/observer wiring and the full W5 failure/concurrency evidence; source/receiver failure adoption, cross-session recovery, busy maintenance and all W5 exit evidence. | +| W6 | Node-owned monotonic Cordon closes the shared role gate. Driver retries/adopts lost results and dispatches explicit busy maintenance release using the peak receiver envelope. Canonical quiescence retains accepted foreground and native primitive completion. Public SQL, Queue, Effect, Activity and Workflow cases cover selected receipt/lease/expiry/waiter faults. Configured host startup holds all new roles until atomic Established boot/current intent confirmation and required probes; retained drain exposes management without serving. The example wires actual signed canonical boot enrollment and joined withdrawal/retirement. The existing reader loop now repairs retained producer requests and fenced views without another hint; metadata collection survives pre-lease startup and local fencing while native admission remains lease checked. | Complete primitive/fault matrix, Cron and Blob external owners, failed-owner producer reconciliation; complete reader replacement/failed-process evidence, ongoing intent supervision, sustained traffic, and all W6 assertions. | +| W7–W10 | Prepared follower ensembles retain signed boots before their original-token CAS; conditional refusal competes with that same write, while absence remains unknown. Confirmed member retirement retains original fences across ambiguous authority closure and shares canonical object coverage and authority closure. Epoch-bound host requests wake the existing supervisor, retain strict retries, reject stale/foreign replacement bindings and preserve accepted cleanup across cancellation and host deadlines. Bounded weak inspection handles expose local completion without certifying fleet settlement. The managed follower producer accepts every original member Pending before its one CAS, retains unknown results in the existing supervisor, and publishes original establishment/confirmed retirement events before releasing inventory. Atomic nonexecution exclusions fence delayed reader/follower acceptance. | Complete role observation, replacement-policy and failed-owner evidence, role evacuation/finalization, complete maintenance and failure examples, fault qualification, and runbooks. | + +This planning pass does not certify the Rust implementation or provider/process +behavior. Record verification against the exact revision and diff that ran; +earlier test results do not certify subsequent changes. The commands and +required assertions below define the implementation gates. + +Focused implementation checkpoints are recorded in +[execution evidence](fleet-operations-progress.md). Each checkpoint states +its source fingerprint, selected commands, and limits; it does not mark an +entire work package complete. + +The [fleet journal example](../crates/cellule-host/minion/README.md) +currently supports `inspect-journal `, `overload`, and +`controller-restart`. The first reopens the local durable reference journal. +The latter commands exercise real bounded movement, including a new controller +epoch after lost source replies and actual lease expiry. Maintenance and +receiver-loss scenarios, complete production observations, and the full W8 +exit requirements remain deliverables. + +| Start here | Contents | +| --- | --- | +| [Current implementation](#current-implementation-and-concrete-gaps) | What can be reused and what is missing. | +| [Architecture](#architecture-and-ownership) and [interfaces](#proposed-interfaces-and-data-contracts) | Module ownership and implementation targets. | +| [Movement](#movement-protocol) and [controller state](#durable-controller-state-and-bounded-execution) | Safety, retries, recovery, and resource bounds. | +| [Maintenance](#maintenance-lifecycle-and-busy-cells) and [follower evacuation](#evacuating-readers-and-follower-responsibilities) | Busy work and safe node shutdown. | +| [Operator controls and runbooks](#operator-controls-and-runbooks) | The application contract for maintenance, status, and incident response. | +| [Work packages](#implementation-work-packages) | Ordered changes with concrete exit conditions. | +| [Source map](#source-map-and-first-implementation-slice) | Files to read, proposed modules, and the first end-to-end change. | +| [Execution checkpoints](#execution-checkpoints) | Milestones, focused commands, and evidence to retain. | +| [Acceptance matrix](#acceptance-matrix) and [commands](#verification-commands) | Required implementation and qualification evidence. | + +## Reading and execution order + +1. Read the current behavior and ownership split below. +2. Implement work packages W1 through W10 in dependency order. +3. Use the acceptance matrix to verify each behavior through its public path. +4. Complete the reference application integration and measured qualification + before enabling automatic movement in a deployed fleet. + +Treat this document as the handoff: the executor needs only this repository, +the listed source files, and the declared qualification environment. Proposed +APIs and example commands become deliverables during implementation; their +presence in this plan does not mean they are callable today. + +Read the root and nearest crate `AGENTS.md` before each code change. The +[runtime guide](../crates/cellule-runtime/docs/README.md), +[host lifecycle guide](../crates/cellule-host/docs/lifecycle.md), and +[framework integration guide](framework.md) remain the existing contracts. + +## Design decisions + +| Decision | Implementation consequence | +| --- | --- | +| One caller-driven reconciler | The application supervises one recurring task; the framework exposes a bounded step and creates no background scheduler. | +| One canonical release and acquisition path | Movement uses ordinary fencing, durable publication, close, and exact-root recovery. No writable database handoff or new Cell authority record. | +| Reserve before release | Lack of receiver capacity leaves the source serving. Reservation success is separate from activation success. | +| Durable fleet intent and attempts | Controller replacement adopts unresolved work; reboot respects physical-node cordon; unknown outcomes retain permits. | +| Explicit eligibility | Pressure and maintenance mode affect local admission and signed remote placement consistently. | +| Three delivery milestones | Ship settled movement first, complete maintenance next, and enable deployed automation only after qualification. | + +The reconciler needs three application adapters: a strongly consistent journal, +a complete fleet observer, and an authenticated action transport. Each node +also needs a trusted `FleetCellProvider` for catalog, canonical storage, local +destination, and owner inputs. The local action journal and controller journal +must share the same committed authorization and permit state. W8 delivers +reference implementations in the runnable example. Applications with an +existing CAS store and node registry can reuse those facilities; applications +must satisfy all adapter contracts before enabling movement. + +## Terms and desired behavior + +| Term | Meaning | +| --- | --- | +| Cordon | Exclude a physical node from new writer, reader, and follower enrollment while existing obligations remain serviceable. | +| Drain | Cordon, settle accepted work, evacuate ownership and follower obligations, then complete normal shutdown. | +| Reconcile | Compare durable intent with current observations and perform a bounded amount of work toward that intent. | +| Move attempt | One advisory proposal, pinned to a source session and actor generation, to release and reacquire one Cell. | +| Released | Source ownership release is confirmed against Cell authority. This does not establish destination readiness. | +| Recovered | Canonical failed-session proof, retained recovery input/position, and current successor serving are proved. It remains distinct from clean source release. | +| Serving elsewhere | A current destination session owns the Cell and its actor has activated the required state. | +| Operation journal | Application-owned CAS records for operator intent, controller ownership, movement permits, and progress. These records never confer Cell ownership. | +| Blocker | A typed reason progress cannot currently continue; the underlying operation remains resumable. | + +| Trigger | Required behavior | Completion | +| --- | --- | --- | +| Fleet ownership skew | Move settled Cells toward the existing weighted targets. | No eligible donation remains outside the planner's deadband. | +| Sustained node pressure | Stop new acquisition, preserve durability work, and move eligible Cells to receivers with usable capacity. | Pressure clears through the existing classifier's recovery rules. | +| Planned maintenance or scale-down | Persist a node cordon, move owned Cells, settle follower responsibilities, and shut down. | Verified relocation plus successful host shutdown and session withdrawal. | +| Insufficient fleet capacity | Refuse additional planned releases that cannot be admitted elsewhere; report the capacity shortfall. | Capacity or policy changes permit progress. | +| Node failure | Continue ordinary lease fencing and exact recovery independently of this controller. | Existing recovery establishes a valid successor. | + +Moving one hot Cell relocates its single writer; it does not increase that +writer's throughput. Sustained demand beyond one node's capacity requires +application partitioning, admission control, or additional read replicas where +the application's consistency policy permits them. The controller reports this +condition instead of repeatedly moving the same Cell. + +## Current implementation and concrete gaps + +| Implementation at the source baseline | Evidence and limitation | +| --- | --- | +| Deterministic placement and transfer planning | [PlacementPlanner](../crates/cellule-runtime/src/fleet/placement/mod.rs) ranks nodes, balances weighted Cell counts, projects receiver demand, and bounds a batch. At the baseline, `fleet_balance` and `plan_transfers` have test callers but no production executor in this workspace. | +| Signed placement observations | [NodePlacementCapacity](../crates/cellule-runtime/src/node/capacity.rs) carries totals, Cell/job counts, and three backlog counters. `PlacementObservation::from_signed_advertisement` derives pressure and draining from zero free memory or disk; it does not carry the local hysteretic tier. | +| Local pressure protection | [PressureClassifier](../crates/cellule-runtime/src/fleet/pressure.rs) and the [actor pressure path](../crates/cellule-runtime/src/cell/actor/task.rs) already classify sustained pressure and start bounded idle eviction. This protection must keep working without the fleet controller. | +| Exact idle release | [Runtime movement methods](../crates/cellule-runtime/src/cell/actor/runtime.rs) list candidates and release a session/generation through fresh actor checks. Candidate tuples contain last-use time, not a complete measured Cell demand or residence history. | +| Conservative executable-work checks | [TransferWorkInventory](../crates/cellule-runtime/src/primitives/maintenance.rs) blocks ready/leased Queue work, running Workflows, pending Workflow work, and due Effects/Cron work. A continuously busy node can therefore remain undrained. | +| Local scale-down | [CellNode scale-down](../crates/cellule-host/src/node/scale_down.rs) stops acquisition, attempts local releases, and returns aggregate progress. It does not reserve a destination or verify remote activation. | +| Ordered shutdown | [Host lifecycle](../crates/cellule-host/src/node/lifecycle.rs) retains lease maintenance through runtime/log drain and sets `Stopped` only on success. `ScaleDownStatus::ready_to_stop` alone is not proof of fleet relocation or foreign follower-tail safety. | +| Node-log rotation and retirement | [Host durability](../crates/cellule-host/src/durability/mod.rs), [directory log authority](../crates/cellule-runtime/src/node/directory/log.rs), and [FollowerStore](../crates/cellule-runtime/src/follower/mod.rs) contain the mechanisms to cover and retire old epochs. Planned evacuation of a follower node needs orchestration around them. | + +The [earlier scaling design](../crates/cellule-runtime/docs/canonical-ltx-scaling.md#balance-ownership-under-live-pressure) +describes a controller in the historical Crab embedding application. Treat +those references as design context; do not count that external implementation +as a production caller in Cellule or assume its endpoints exist here. + +## Architecture and ownership + +The caller-driven `FleetReconciler` facade in `cellule-host` performs one +bounded step when called; it starts no scheduler and constructs no runtime. +The embedding application owns the recurring loop and supplies the policy, +journal, observation, and transport adapters. This makes the orchestration +reusable while retaining application control of deployment decisions. + +```mermaid +flowchart TD + App[Application loop and authorized operator intent] --> Driver[Host FleetReconciler] + Driver --> Journal[Application journal and controller lease adapter] + Driver --> Observe[Application fleet observation adapter] + Observe --> Ads[Signed node observations] + Driver --> Plan[Runtime pure placement and operation decisions] + Plan --> Transport[Application authenticated peer adapter] + Transport --> Host[Source and destination CellNode] + Host --> Actor[Existing actor and coordination kernel] + Actor --> Authority[Existing publication and Cell authority CAS] +``` + +| Location | Responsibility | +| --- | --- | +| `cellule-runtime::fleet::placement` | Receiver eligibility, resource projections, weighted targets, movement policy bounds. | +| `cellule-runtime::fleet::operations` foundations and remaining records | Pure operation transitions, typed observations/actions/results, retry decisions, and fleet permit accounting. No I/O, clocks, tasks, or product identity. | +| Existing `cellule-runtime::coordination` and actor | Per-Cell admission, generation checks, accepted-work drain, publication, close, and release. | +| `cellule-host::fleet` | Bounded `reconcile_once` driver and adapter interfaces; delegate decisions to the runtime planner and operation state machine. | +| `cellule-host::CellNode` | Local action execution, actual resource reservations, role evacuation, and the existing single drain lane. | +| Embedding application | Timer/task ownership, authenticated management routes, authorization, durable journal backend and namespace, membership registry, signing, peer serialization, readiness mapping, process termination, and rollout. | + +Keep `cellule-app` dependency-light and free of fleet lifecycle policy. No new +crate, second scheduler, alternative authority record, direct writer handoff, +or writable SQLite file transfer is needed. Keep all Cell-control mutation on +the current runtime path. + +## Proposed interfaces and data contracts + +Use the current APIs listed below and implement the remaining orchestration +targets. Follow the host's existing boxed-future adapter style; do not +introduce a new async abstraction dependency. + +The following distinction applies to this working tree, rather than the +baseline revision. Use the source signatures when implementing against a +present API; the pseudocode below specifies the remaining orchestration. + +| Surface | Current state | Implementation instruction | +| --- | --- | --- | +| `CellNodeBuilder::with_fleet_startup_intent` and `CellNode::confirm_fleet_startup` | Present host startup barrier. | Load retained physical intent before build, journal canonical boot enrollment, then confirm its exact key. Confirmation keeps admission held until `start()` validates required facilities; Maintenance exposes management while serving/new roles stay closed. | +| `CellNode::refresh_fleet_intent` | Present caller-driven live intent refresh. | The supervised application membership/lease loop reads current intent and the original Established boot atomically before renewal. Sticky admission closure preserves existing owners; stale replies, changed boot evidence and shutdown races are refused. The example invokes it before canonical heartbeat CAS. Production supervision and complete maintenance qualification remain required. | +| `CellNode::install_fleet_actions(scope, node, journal, cells)` | Present in host `node/fleet.rs`. | Install once during startup, before readiness. Supply the shared atomic action journal and trusted local Cell inputs. | +| `ReadReplicaManager::prepare_source` / `activate_source` and `CellNode::install_fleet_reader_enrollment` | Present exact source bridge and owned durable reader producer. | Install the binding before start/first activation. Ordinary hints and prepared activation journal Pending before native opening, retain completion after waiter cancellation, replay original publication, and retire after joined closure. Configured fleet hosts require the binding. Complete observer and failed-process/replacement evidence remain required. | +| `ReadReplicaManager::fleet_reader_enrollments_page` | Present bounded advisory producer pages. | Scan original requests and native/publication progress even during paused journal/open calls; account accepted preparation, unobserved and joining jobs. Also collect native view pages, current authority and durable roster revisions. Unbound/closed/unavailable inventories cannot prove empty coverage. | +| `CellNode::fleet_follower_enrollments_page` | Present bounded advisory epoch pages. | Preserve every original selected member and unknown acceptance during paused provider/journal work; retain original retirement/error Arcs. Busy protocol includes preparation before a row exists. Also collect supervisor completion, inbound lanes, fresh directory authority and replacement policy; zero rows or idle protocol cannot prove settlement. | +| `CellNode::fleet_durability_supervisor` | Present fixed-metadata advisory capture of the retained native supervisor and its bounded rotation bank. | Preserve distinct unstarted/running/returned/joined states and separate original supervisor/request-stop failures, including capture during drain after byte admission closes. Missing or failed capture supplies no coverage; joined state and local rotation completion still require producer/native lanes, current authority and replacement-policy evidence. | +| `NodeDirectory::prepare_log_enrollment` / `prepare_log_enrollment_attempt` / `commit_log_enrollment` | Present opaque follower selection and CAS bridge. | Revalidate exact boots before acceptance; journal all selected members Pending before dispatch, retain the original attempt, and reconcile that same scope. `fence_log_enrollment` proves only its original-token refusal; absence cannot settle unknown work. Managed host producer and journal retirement remain required. | +| `CellNode::apply_fleet_action(action, now_ms)` | Present; returns a retained `FleetActionCompletion`. | Movement actions use this path. Check `committed`, retain unknown outcomes, and obtain fresh serving evidence. Maintenance integration remains required. | +| `CellNode::inspect_fleet_action(request)` | Present; returns `FleetInspectionObservation`. | Use a fresh nonce, exact head/registry and endpoint, and exclusive capture deadline. Authenticate and validate the full request plus original capture interval; recheck head/registry when committing dependent decisions. Inspection starts no acquisition or cleanup effect; the shared lane can settle prior completed work. | +| `CellNode::fleet_snapshot(request)` | Present request-bound native page capture; aggregate observation remains incomplete. | Pin the full head/registry, physical boot, nonce, category, original continuation, limit and deadline. Authorize before and after capture through the same finite action bank. Retain original page charges and errors; authenticate and combine every required producer/native page with fresh authority and replacement-policy evidence. Unbound is never empty coverage. | +| `FleetNodeInventoryScan` and `FleetNodeInventory::recheck` | Present complete native traversal and all-category rechecks at a full roster barrier. | Preserve original requests, native lifetimes, transitions and shared errors. Validate enrollment matching, then recheck every category after fleet-wide authority/policy discovery and reconfirm the journal. Native interval evidence cannot settle failed processes or grant finalization. | +| `FleetFollowerReferences` | Present complete foreign-log traversal and exact authority recheck. | Include expired/fenced leaders, retain native continuations and compare all original epoch/phase/ensemble/coverage/liveness fields after native collection. Match original roster requests and reconfirm that full barrier. Missing references cannot settle Pending work; retired local lanes can still have foreign authority obligations. Replacement policy and canonical recovery/retirement remain required. | +| `FleetRoleCoverage` | Present cross-node original producer, native lane and current foreign authority matching. | Supply every required original boot and physical reference scan; finish initial collection, global native rechecks, then exact foreign rechecks and full journal confirmation. Observe enrolled lanes before their first append through the complete original producer/ensemble witness. Authentication, unexpected advertisements, current Cell authority, replacement policy and failed-process closure remain required; coverage does not grant settlement or finalization. | +| `ReaderEvacuationRecord` / `FleetReaderEvacuationJournal` / `FleetReaderEvacuationVerifier` | Present bounded durable reader policy history, complete replacement pages, atomic latest pointers and independent native revalidation/refresh. | Commit through the shared registry transaction; confirm exact policy, authority, enrolled selected boots and ready prefixes after reconstruction. Refresh from the original committed retirement after policy/boot/owner changes, including Closing. Failed-owner follower policy, failed receiver integration, complete fleet observation and aggregate maintenance actions remain required. | +| `FollowerEvacuationRecord` / `FleetFollowerEvacuationJournal` / `FleetFollowerEvacuationVerifier` | Present bounded durable live-owner ensemble history, revisioned application policy, atomic latest pointers and canonical/native revalidation/refresh. | Bind every original Retired member and replacement request, confirm current canonical ensemble plus the actual source/member native owners through `FleetSnapshotTransport`. Refresh after policy/epoch/operation changes without repeating original retirement. Complete failed-owner successor policy, full observer and SettleRoles/Finalize integration; this per-owner evidence cannot finish maintenance. | +| `FleetObservation::with_role_coverage` | Present retained graph attachment and planner-input binding. | Preserve the original interval and reject replacement or scope/registry mismatches. The reconciler rechecks the graph's exact head/registry and full roster digest; unchanged registry alone cannot admit an earlier controller head. Attachment never upgrades partial adapter coverage. | +| `CellNode::follower_evacuation` | Present per-live-owner replacement and retirement check after the original requested rotation. | Require the declared member minimum, donor exclusion, complete Established replacement rows, pinned signed boots, current authority and full journal rechecks. Retain the original completion/error history and publish/revalidate its interval. This does not settle failed owners, Pending producers or the physical node's other obligations. | +| Recovered-log authorization, member retirement and canonical Retired CAS | Present runtime tail closure after canonical recovery pinning. | Use `RecoveredNodeLogTransport`, receiver-side `authorize_recovered_log_retire` and `FollowerStore::retire_recovered`; confirm every original member before `retire_recovered_log`. Adopt exact committed closure with `retired_recovered_log` before repeating effects. Publish original enrollments through the host capsule, then complete failed-process barriers and maintenance orchestration; native closure alone cannot finish W7. | +| `RecoveryManifestStore::load_manifest` | Complete digest-verified recovered-suffix metadata across every original application, retained after successor materialization. | Read the identity from canonical sealed-log authority, then verify each bundle through its application's ordinary store. Retain the complete original writer set separately, including object-covered Cells without a suffix; check current successor authority, acknowledged prefixes and serving. This metadata read cannot settle roles or finalize a node. | +| `CellCatalog::scan_all` / `CatalogScanReceipt` | Complete bounded traversal captures all 256 scoped heads before page reads, uses ordinary page verification and refuses partial/failed scans or changed final heads. | Authenticate the complete application/tenant source set, join original process/accepted work, retain every delivered owner history under the operation and recheck the catalog receipt. Sequential heads are not a global transaction or durable object pin; catalog completion alone cannot settle roles or finalize a node. | +| `FleetOriginalWriterCapture` / `FleetOriginalWriterJournal` | Complete authenticated catalog traversal and original process joining retain every matching full owner Control, including rootless originals removed by takeover. Bounded manifest/pages publish atomically in minion's existing SQLite owner and preserve original bytes on replay/reconstruction. | Authenticate complete original backend mappings; verify every acknowledged prefix, exact dependency availability and current successor serving. Integrate complete role/native policies into SettleRoles/Finalize and qualify real process/provider failures. Original metadata grants no root pin or settlement rights. | +| `CellAuthority::owner_history` | Canonical owner departure retains full original controls before the existing owner CAS; complete per-incarnation epoch reads refuse missing or changing history. | Use the original capture/journal path to retain the complete authenticated operation set, then verify current successor prefix/serving. Missing legacy, restored or older-binary history is a typed blocker. These rows grant no authority or root retention pin and cannot settle roles or finalize a node. | +| `FleetRecoveredFollowerRetirement` | Present checked publication of recovered ensemble enrollment closure. | Capture the exact canonically Retired physical leader/epoch against the full roster, publish all original member requests through the existing journal, then confirm the complete new roster and canonical authority. Retain every original response/error. Fresh recapture adopts lost replies without native RPCs or timestamp refresh; failed boots, replacement policy and operation integration remain required. | +| `FleetFailedBootRetirement` / `FleetFailedBootProcesses` | Present original failed-boot closure against canonical terminal fencing and durable application process evidence. | Complete every original related request first. The read-only provider proves original process/accepted external work joining and session nonexecution; the capsule publishes through the existing journal and rechecks full roster, authority, physical references and process evidence. Actual provider qualification, replacement policy and maintenance actions remain required. | +| `NodeDirectory::fenced_session` / `FleetFailedBootProcessRequest::capture_fenced` / `confirm` | Present process confirmation before leader-log recovery/retirement, using the existing application provider. | Authenticate and join the exact original process and all accepted work, then retain every affected Cell before recovery effects. The v2 identity survives recovery and claim adoption; `capture_retained` reuses it only after strict terminal boot/role checks. Existing terminal v1 identities remain unchanged. Process evidence cannot substitute for original writer inventory or current successor/policy evidence. | +| `FleetFailedReaderRetirement` / `FleetFailedBootProcessRequest::capture` | Present original failed receiver reader publication, including Pending and Established histories, through the existing journal/provider boundary. | Capture the original boot/process request before role settlement, then publish each exact reader only under joined receiver lifetime evidence. Source failure cannot close a live receiver. Persist replacement-policy evidence, handle live receivers through ordinary closure, and qualify actual OS/process/external-job providers before completing W7. | +| `FleetActionJournal` | Present in host `fleet/journal.rs`. | Implement acceptance, result publication, original-action lookup, and acquisition/recovery basis and evidence recording/lookup with the specified atomic and durable semantics. | +| `FleetCellProvider::cell_inputs(spec)` | Present in host `fleet/cells.rs`. | Resolve metadata and canonical local inputs without performing an effect. | +| `FleetCellProvider::recovery_inputs(spec)` | Present in host `fleet/cells.rs`. | Supply existing canonical failed-session proof and manifest access. Ordinary recovery establishes fencing and tail sealing independently. | +| `CellRuntime::fleet_cells_page`, `prepare_receiver`, and `activate_prepared_receiver` | Present in runtime actor modules. | Reuse generation-bound inventory and actual resource-token ownership. Do not replace them with host counters. | +| `AcquisitionObserver` and observed Idle/takeover methods | Present in runtime actor modules. | Confirm exact input before CAS and actual recovery position before admission; use canonical rollback on failure. | +| `FleetJournal` and `FleetEnrollmentJournal` | Present transaction contracts in host `fleet/controller.rs` and `fleet/enrollment.rs`. | Implement both with the action journal in one durable transaction domain. Persist retained rows and operations, exact registry versions, scheduling policy, and history. | +| Example `SqliteJournal` | Present in host `minion/journal/`; implements all three journal contracts. | Reuse for local reference execution. Independent SQLite clients and reconstruction have focused evidence; complete observer coverage and process/provider qualification remain required. | +| `FleetReconciler`, `FleetObserver`, and `FleetTransport` | Present settled-movement driver and adapter contracts under host `fleet/reconciler/`. | Wire real observation and management adapters. Existing evidence uses SQLite and simulated transport effects; it does not establish complete W5 or three-node restoration. | +| `CellRootLineage`, `VerifiedRootPrefix`, runtime `verify_root_prefix` | Present native verified-preparation metadata and exact prefix/origin proof. | Canonical publisher and recovered-overlay materialization retain additive links before root CAS; host serving checks require the exact released/recovered root, complete current origin bytes and fresh native/authority checks. Shared runtime memory admission and bounded inventories refuse excess. Authentication, original physical-boot scope and every sealed recovery suffix still need aggregate producer/observer consumption; a per-movement proof cannot enable SettleRoles/Finalize. | +| `VerifiedRecoveryPrefix`, runtime `verify_recovered_prefix` | Present exact original sealed-suffix materialization and current origin proof, including interrupted claims at later acquisition epochs. | Obtain each row from the original digest-verified manifest; bind its retained closed owner and identical canonical acquisition overlay, then verify exact materialized prefix/current graph and unchanged selected authority. Host recovered serving compares full journal/canonical inputs and results before consuming this proof. Complete authenticated original writer/suffix aggregation, native roles and accepted-work barriers remain required. | +| Complete aggregate observation and node finalization actions | Remaining aggregate/maintenance surfaces. | Compose complete role inventory and implement W6–W7 barriers through the existing drain lane. | + +The current local action API receives supplied time. Its action envelope holds +the deadline, and the transport owns its waiter timeout. A proposed facade's +deadline argument must preserve those boundaries; timing out a waiter cannot +cancel accepted work or free its permit. + +```text +FleetReconciler::new(scope, claimant, profile, journal, observer, transport) +FleetReconciler::reconcile_once(clock, deadline) -> FleetReconcileReport + // clock: application callback returning Result in the node clock domain. + // Read directly at each boundary; reject regression. Deadline is monotonic. + +CellNode::fleet_snapshot(FleetSnapshotRequest) -> retained FleetNodeSnapshot +CellNode::apply_fleet_action(action, now_ms) -> retained FleetActionCompletion + +FleetJournal: + load_snapshot(scope) -> head and registry from one transaction + claim_controller(scope, expected_revision, claimant, now_ms) + compare_exchange(expected_snapshot, controller_epoch, now_ms, transition) + load_operation(scope, operation_id) + load_progress(scope, digest) + last_moved_at(expected_snapshot, cell, incarnation) + last_movement_at(expected_snapshot) + intents_page(exact_registry_version, cursor, limit) + enrollments_page(exact_registry_version, cursor, limit) + set_scheduling(expected_registry_version, enabled) + // Inherits FleetActionJournal and FleetEnrollmentJournal. + +FleetObserver: + observe(expected_snapshot, now_ms, deadline) -> FleetObservation + // Application collector uses bounded membership, ownership and role pages. + // Prove completeness and enrolled signing keys; retain original sample times. + +FleetTransport: + dispatch(authorized_action, deadline) -> FleetActionCompletion + inspect(exact_request, deadline) -> FleetInspectionObservation + // Effects return committed evidence or a retained publication obligation. + // Inspection returns current request-bound evidence, never cached success. + +FleetCellProvider: + cell_inputs(immutable_move_spec) -> FleetCellInputs + // Metadata lookup only; no hydration, takeover, or writable SQL handle. +``` + +Operation creation and status endpoints belong to the application. They write +and read the same journal contract used by the reconciler; they do not invoke +unrecorded release operations. `reconcile_once` returns aggregate progress, +bounded events, the next wake deadline, typed blockers, and original endpoint +errors per charged attempt. Each attempt has a bounded share of the pass +deadline so an unavailable endpoint leaves time for healthy siblings. Ambiguous +timeouts require a fresh journal/epoch check before continuing; journal failures +stop the pass. An event can wake +the application loop before its periodic interval. + +Keep format sizes concrete for the initial implementation: a fleet head is at +most 64 KiB, a paginated observation/progress response at most 1 MiB and 128 +entries, and an individual action/result envelope at most 64 KiB. Large error +details stay in correlated logs; return a bounded classification and source +error reference. Reject oversized inputs before allocation or dispatch. These +limits supplement existing peer body and decoder limits; they never widen them. + +`authorized_action` is constructed only after the application's receiver has +validated peer identity, fleet/application scope, journal permit, controller +epoch, deadline, and exact target session. Rust visibility cannot substitute +for remote authentication. Each action also repeats the runtime's local +admission and authority checks. + +| Proposed value | Required contents and semantics | +| --- | --- | +| `FleetSnapshot` | Scope, membership revision, capture start/end, sorted live sessions, each signed sample sequence/time, completeness, and digest of the canonical bounded representation. The digest identifies inputs; it does not make a scan atomic. | +| `OwnedCellObservation` | Cell target, incarnation, source session, actor generation, authority epoch/root/sequence if known, actual residence start, last-use time, measured or conservative memory/disk/job cost, stable-sample count, and blocker classes. | +| `FleetAction` | Operation and attempt IDs, journal revision/controller epoch, exact source and destination sessions, Cell identity/incarnation/generation, deadline, snapshot digest, cost, and action kind. | +| Action kinds | Cordon, prepare receiver reservation, quiesce/release Cell, activate reserved Cell, settle follower obligations, and finalize node drain. Duplicate kinds for one attempt return or reconstruct the same result. | +| `FleetActionOutcome` | Reserved, blocked, released with authoritative position, activated with current session/owner evidence, recovered with canonical recovery evidence, already completed, rejected, or outcome unknown. Preserve source errors separately from the bounded status classification. | +| `DrainBlocker` | Incomplete/stale observation, incompatible release/schema, no receiver capacity, busy execution, live external lease, pending publication, follower obligation, unknown inventory, movement budget, action outcome unknown, deadline, or facility failure. | + +Bound pages to 128 Cell/progress entries and remote dispatch concurrency to +the active fleet permits. Use cursors bound to a snapshot generation; restart a +scan when its generation changes. Do not copy every Cell into one journal +record, spawn one task per Cell, or expose Cell IDs as metric labels. + +### Application startup and adapter handoff + +The host startup intent barrier and reference boot producer implement the order +below. Complete observer/role production, ongoing supervision and the full W8 +maintenance/deployment evidence remain required. See the +[host startup contract](../crates/cellule-host/docs/lifecycle.md#fleet-boot-admission) +and [recorded evidence](fleet-operations-progress.md). + +1. Load a stable physical NodeId, create a fresh boot session, and read its + committed intent. Missing or ambiguous intent cannot open acquisition. + Pass it to `with_fleet_startup_intent` before build. The shared gate holds + writer/reader/follower admission, including under an Active row. +2. Construct canonical catalog/storage inputs and the shared resource ledger. + Build the CellNode with required owned components, including fleet actions. + Install its lease through the startup path that keeps readiness closed. +3. Install the action journal and trusted Cell provider, then the required + durability, reader, follower, and primitive facilities. The Cell provider + resolves exact identity and local paths; remote requests carry no credentials + or filesystem paths. Lookup performs no authority mutation or activation. +4. Accept Pending in the shared registry before canonical boot advertisement, + publish checked Established evidence, and call `confirm_fleet_startup` with + the exact original enrollment key. Its journal read includes current intent + in the same transaction. Reconcile retained obligations, + and publish signed observations from actual local samples. Complete ordinary + startup probes. Open serving readiness only under the committed intent and + local admission state. A draining reboot keeps its management/recovery path + available via `is_management_ready()` while rejecting new role enrollment. + Confirmation leaves the startup hold until `start()` checks required + components. Continued intent changes require the authorized action path. +5. Compose one reconciler with the journal, observer, transport, and validated + profile. Supervise its caller-driven loop; wake on progress and otherwise + use the proposed 15-second interval. A dropped reconciliation waiter leaves + accepted node work owned by the node executor. +6. Route authorized maintenance/status commands to the same committed journal. + Expose `safe_to_take_offline` from the finalization proof. Preserve peer + endpoints and lease maintenance until the canonical host drain permits + withdrawal; join the application loop during process shutdown. + +| Adapter | Deliverable required before enabling movement | +| --- | --- | +| Journal and enrollment registry | Conditional-write/transaction proof; idempotent request acceptance; durable action/basis lookup; head, permits, intent, and history reconstruction; racing-controller and lost-reply tests. | +| Observer | Complete revisioned roster and role pages; signature/freshness checks; generation-bound demand; explicit incomplete views and post-batch barrier. | +| Transport | Application authentication and scope authorization; bounded decoding; exact session routing; duplicate/ambiguous response handling; no timeout converted into a definite refusal. | +| Cell provider | Exact catalog/Cell/incarnation binding, release/schema compatibility, leased local owner, private destination, and canonical authority/replica construction. | + +## Signed pressure and admission + +Introduce placement schema 3 with an explicit node mode, pressure tier, and +sample sequence/time. Node mode is `Active`, `Cordoned`, or `Draining` and is +independent of free capacity. Preserve actual capacity measurements on draining +nodes so they remain visible as donors and diagnosable by operators. + +Publish the stable result of the existing `PressureClassifier`. Its +`Recovering` variant is an internal transition marker, not a wire state. Do +not reconstruct the tier on a peer, sign a controller guess as a node +measurement, or advance sample time when merely republishing an old sample. +The sample sequence is scoped to the node session and covered by the placement +signature; the directory's CAS heartbeat generation is unsuitable because the +current signing path deliberately excludes it. + +Use effective container/process limits and the runtime's resource ledger when +constructing usable headroom. Unknown measurements remain explicitly unknown. +Reserve CPU/queue-age signals for a later measured extension: the first +implementation uses memory, disk, job saturation, and existing backlog data, +and must not claim to identify every kind of CPU overload. + +The local acquisition gate becomes reasoned state: pressure is reversible; +cordon and shutdown are sticky until an authorized lifecycle transition. +Clearing pressure cannot clear a maintenance cordon. Do not implement pressure +relief by calling today's one-way `stop_acquiring` and then reopening admission +with an unrelated atomic flag. Gate writer acquisition, new reader activation, +and new follower enrollment at their local admission points as well as in +remote placement. + +| Observation | New proactive receive | Existing work and planned movement | +| --- | --- | --- | +| Active and Normal | Eligible after capacity/release checks. | Ordinary cooldown and gain/balance gates. | +| Constrained or stale | No new proactive receive. | Preserve accepted work and publication; obtain fresh evidence before new planned releases. | +| Shedding or Critical | Ineligible. | Keep local bounded protection; prioritize safe relief moves. | +| Cordoned or Draining | Ineligible regardless of pressure. | Existing responsibilities continue until explicitly settled. | +| Unknown placement schema | Ineligible. | Ordinary authority routing remains possible only if identity/lease decoding and verification succeed. | + +Apply this eligibility consistently to direct placement, transfer planning, +reader recruitment, follower selection, and the receiving node. The current +planner rejects Critical generally and Shedding on some paths; unify the new +rules so a relief path cannot select a node that balancing would reject. + +## Snapshot and demand collection + +The observation adapter provides a revisioned enrollment roster. Read the +roster, fetch every required signed advertisement, then recheck the roster +revision. Missing members, duplicates, schema mismatch, regressions, or a +changed roster make count balancing unavailable for that pass. Never compute +weighted fleet totals from a filtered subset and label it complete. + +`NodeDirectory::live` alone does not supply that roster revision. If an +embedding application has no suitable registry, its journal adapter must add +an application-owned enrollment roster: register before serving readiness, +retire after authoritative withdrawal, and reconcile unexpected live records +as an incomplete view. The reference adapter must exercise this protocol. +Fresh per-source/per-receiver observations can still support pressure or drain +planning under their own gates, without claiming complete fleet counts. + +The current host reconciler implements durable roster traversal through +`fleet/roster/`: canonical bounded pages, strict continuations, full head/registry +rechecks, both original endpoints of unresolved responsibilities, and retained +terminal rows. It supplies `FleetRoster` to `FleetObserver::observe` and binds all +rows into the planner digest. Count balancing additionally requires established +boot coverage, no Pending enrollment and writer rows matching signed counts. +The host also provides `FleetNodeInventoryScan` for canonical all-category +traversal, retained native producer/job state, exact roster matching and fresh +all-category revalidation after fleet-wide discovery. The reference observer +consumes it with canonical signed heartbeats, fresh Cell authority and repeated +native/registry checks, preserving independently verified pressure rows when +count coverage is invalidated. It can +establish complete counts for its closed writer-only construction profile; +unexpected boots, role installation/enrollment or missing obligations invalidate +that coverage. The `balance` command exercises actual residence and repeated +batches. Complete reader/follower role/authority envelopes and the finalization +transaction described below remain required; writer-profile coverage does not +discharge those work packages. + +### Enrollment barrier for finalization + +The observer needs more than a stable list of live nodes. Implement a durable +enrollment registry in the application adapter, covering node sessions and +reader/follower enrollment that can create a responsibility on a node. Its +records are advisory coordination; ordinary node-log authority still decides +membership and retirement. + +1. Before starting an enrollment side effect, CAS a pending entry naming its + exact participating sessions, physical nodes, epoch, and request identity. + In that same transaction, check the nodes' intent revisions. A Cordoned or + Draining target cannot receive a new permit. +2. Perform the existing enrollment protocol. Record authoritative completion + or a definite refusal. A lost reply leaves the entry pending; expiry alone + cannot remove it. +3. A maintenance request closes new permits for its physical node in the same + committed intent transition. Install the local cordon and join or reconcile + every enrollment admitted before that transition. +4. Capture a registry revision and collect local roles plus the exact authority + for every registered obligation, including failed sessions. Recheck the + revision and intent before committing the finalization proof. Changed or + incomplete observations restart this step. +5. Retire registry obligations only after the ordinary reader close or + follower coverage/recovery and retirement proof. Withdraw a node's registry + session only after its authoritative directory withdrawal. + +All participating enrollment producers must use this protocol before fleet +maintenance is enabled. Bootstrap existing obligations during a controlled +enrollment pause, including cold local lanes and failed-owner records. Record +the bootstrap revision and reject unknown live records afterward. An adapter +that cannot establish this initial coverage reports an incomplete inventory +and cannot finalize maintenance. The runnable example must demonstrate a +racing enrollment and a lost enrollment reply at the maintenance boundary. + +After a dispatched batch, count balancing waits for a complete set of samples +taken after the batch barrier. Keep the existing single-donor rule, weighted +Cell targets, and receiver deadband. Revalidate release compatibility and Cell +schema support before reserving the destination. + +Collect Cell costs from actor/worker accounting. Today candidate tuples expose +`last_used_ms`; do not relabel it as residence start. Add generation-bound +residence tracking and real or conservative disk restore demand. Unknown cost +blocks proactive movement until an upper bound can be established. Receiver +admission must charge memory, disk, descriptors, worker jobs, and restoration +resources through the existing ledgers even when the planner uses fewer +dimensions for ranking. + +Measure logical SQLite bytes as `page_count * page_size` through the serialized +SQL worker, alongside primitive inventory and commit sequence. Do not use a +sparse file's allocated bytes as full restore demand. Bind the result to actor +generation, inventory revision, and published sequence; a mutation invalidates +the sample, and an older probe cannot restore its validity. Repeated page reads +do not create new samples. Require two independently captured, unchanged +samples for ordinary movement and reject unknown, regressing, or stale samples. + +Use validated LTX limits for the conservative disk reservation: twice the +database ceiling, plus the plan-input ceiling, plus canonical restore headroom. +Use checked arithmetic and reject a Cell outside those bounds. Memory, +descriptor, and job demand come from existing resource accounting. These +values are admission estimates; process RSS and actual restore peaks still +need the measured qualification campaign. W4 must reconcile this estimate +with the canonical restore path before claiming that a reservation is enough. + +Keep cooldown history for completed moves in the bounded/paged journal and +associate it with Cell incarnation. A controller restart must not erase the +anti-oscillation evidence. Urgent movement may bypass ordinary residence and +gain rules; it never bypasses durability, receiver admission, or known costs. + +## Movement protocol + +```mermaid +sequenceDiagram + participant R as Reconciler + participant J as Operation journal + participant D as Destination + participant S as Source CellNode + participant C as Cell authority + R->>J: CAS attempt and fleet permit + R->>D: Prepare admitted receiver reservation + D-->>R: Reservation bound to session and attempt + R->>J: Record reservation and release intent + R->>S: Release exact Cell generation + S->>S: Close admission, settle work, publish, close + S->>C: Existing exact owner release CAS + S-->>R: Confirmed release and position + R->>J: Record release + R->>D: Activate using ordinary authority acquisition + D->>C: Acquire and verify current authoritative state + D-->>R: Serving evidence or typed failure + R->>J: Confirm result and retire permit +``` + +1. Journal the attempt before dispatch. Its identity includes operation, Cell + incarnation, exact source session/generation, and a monotonically allocated + attempt sequence. It is never reused for a newer owner generation. +2. The receiver reserves conservative resources locally and binds the + reservation to the attempt/session. Initial implementation does no + speculative full database hydration. A reservation confers no ownership. +3. The source checks the valid permit and its own session/generation, then + follows the actor's release path. For ordinary balancing and pressure, + retain the existing settled-work predicate. Planned busy-Cell maintenance + uses the explicit extension described below. +4. Planned release waits until every accepted acknowledged mutation is covered + by the authoritative object root and local close completes. Failure to + obtain that proof blocks graceful movement. Owner failure still uses the + existing follower-log recovery path; never manufacture a clean release. +5. The destination rereads control and acquires normally. A different eligible + node may have won in the meantime; verify its current serving evidence and + count that as relocated, releasing the unused reservation. Never overwrite + a new owner to enforce the planner's preference. +6. Confirm source release and destination activation separately. A successful + transport send, reservation, started drain, or unowned root is insufficient + to mark a Cell as serving elsewhere. + +Release evidence names the final Cell incarnation, authority epoch, committed +sequence, and exact root. Destination evidence must bind its current owner +session and new epoch to the same incarnation and a sequence at least as new +as the release position. Verify actor-backed readiness at that position using +the runtime's existing admission and receipt-position checks, alongside current +authority. An old advertisement, cached success response, or root hash by +itself is insufficient; the destination may already have advanced to a newer +verified root. The reference application also performs receipt-bound readback. + +Use the current `CellNode::release_idle_cell_at(cell, source, generation, +incarnation, epoch)` for settled source release. It delegates to the actor's +existing close and release path, which returns the final `PublishedPosition`. +Persist that result under the accepted action identity. Reading authority +after an ordinary `release_idle_cell` reply cannot replace it: a successor +may already have published a different root. A dropped reply or a crash before +result publication still requires the unknown/recovery rules below. + +The actor's `Error::CellReleaseRefused` distinguishes a definite refusal before +this exact request began canonical deactivation from an uncertain close result. +The host publishes Rejected with the underlying error. Existing release-refusal +transitions cancel and join unused receiver credit before retiring the attempt. +Emergency local eviction can invalidate a sampled source independently; its +Idle authority state cannot substitute for this attempt's Released evidence. +An accepted action with no retained original result remains Unknown, even when +the source actor is now absent. Observation or transport errors alone grant no +permission to cancel a possibly accepted source effect. + +Keep the reservation charged until it is consumed by activation, explicitly +cancelled and joined, or proven absent. Reservation expiry permits local +cancellation; it does not make an unobserved remote task disappear. If the +destination fails after release, retain the exact unowned root, reconcile any +unknown attempt, and use ordinary recovery/activation on another eligible +node. The donor never resumes from its old SQLite handle. + +### Receiver reservation ownership + +Implement one reservation object per attempt in the host executor. Bind it to +the fleet scope, Cell incarnation, receiver boot session, exact cost, and +expiry. Keep the actual runtime and LTX resource tokens in that object; the +journal stores their bounded identity and cost, not a claim that tokens +survive a process restart. + +| Step | Required resource ownership | +| --- | --- | +| Prepare | Charge Cell/worker slots, native and cache memory, file descriptors, restore scratch, and disk headroom through the existing ledgers. Publish Reserved only after all required tokens are held. Roll back a partial refusal through normal token cleanup. | +| Activate | Move the held tokens into the existing reserved worker activation and restore path. Check measured requirements against the reservation before ownership acquisition. Obtain any additional credit first or refuse; never drop credit to reacquire it later. | +| Serve | Transfer lasting resident charges to the activated Cell. Release temporary restore charges only after their work and scratch owners have closed. | +| Cancel or fail | Join accepted restore/activation work before returning its tokens. If ownership was acquired, use canonical actor close and authority handling; cancellation cannot silently abandon a writer. | +| Receiver reboot | Old in-memory tokens do not return. Reconcile the old session's accepted work and prove its process ended before clearing unknown cost or preparing another attempt. | + +Start with `CellRuntime::activate_restored_reserved` in `cell/actor/acquire.rs` +and the worker's `reserve_activation` implementation. Extend their +token ownership boundaries where necessary; do not add an independent host +counter that can disagree with normal cold activation. Test prepared movement +concurrently with ordinary activation and reader hydration on the same ledger. + +The runtime foundation now supplies `CellRuntime::prepare_receiver` and +`activate_prepared_receiver`. It uses `DiskReservation::into_budget` and +`DiskBudget::finish_preparation` through the canonical operation host; the +scoped disk budget inherits parent node admission. Preparation holds the full +disk envelope and completion returns only unused bytes. Native memory, +descriptors, Cell slots, and affine SQL-job tokens use the ordinary ledger; +lasting Cell charges remain owned by its worker. + +The runtime owns the bounded local receipt and accepted activation task. +Dropping an RPC waiter cannot cancel takeover, and shutdown joins it before +the actor/worker barrier. Expiry permits cancellation only while credit remains +unused. A local `Activated` hint is not durable or current serving evidence; +the executor must still inspect authority and the actor. Source acceptance and +scope binding, receiver actions, and acquisition-basis integration have focused +local evidence. Complete recovery and process/provider evidence remain +W4 work. See the [runtime receiver contract](../crates/cellule-runtime/docs/deployment.md#prepare-capacity-before-releasing-a-cell) +and [LTX resource contract](../crates/cellule-ltx/docs/safety.md#prepared-disk-credit). + +Requests rejected before execution may use the current routing retry rules. +An ambiguous mutation retains its original request ID and uses Resolve; +movement must not introduce an unconditional command retry. + +### Receiver cleanup after source release + +Reservation cleanup and ownership relocation are separate facts. Extend W1 +action authorization and transitions, and W4 execution, so an unused preferred +reservation can be cancelled while an attempt remains Released or Activating. +Persist the cleanup request before dispatch and retain that attempt's permit. +The current reducer implements `CleaningReceiver` after a proven release. +Cleanup returns to Released while preserving the exact position and permit; +it never becomes Cancelled. Preferred-session serving requires independent +resource settlement. Source release still unresolved remains blocked until its +canonical release/recovery proof is established. + +| Observed state | Permitted progress | Terminal claim | +| --- | --- | --- | +| Before release acceptance, unused credit | Commit cancellation, free credit, and confirm no source effect was accepted. | Cancelled only after both facts are proved. | +| Release may have started, outcome unknown | Inspect the original accepted effect; cancel only credit proven unused and record cleanup independently. | No Cancelled or permit retirement from cleanup alone. | +| Released, unused or expired credit | Cancel/join that credit, retain release evidence, and seek canonical activation on an eligible node. | Keep Released/Activating until current serving evidence is proved. | +| Ordinary acquisition won on the preferred session | Verify its actor, release position, and authority; independently free the unused prepared credit. | Same-session ownership does not prove the prepared credit was consumed. | +| Another node won | Verify its current eligible serving evidence and cancel the preferred receiver's unused credit. | Retire only after serving and cleanup proofs are committed. | +| Prepared activation accepted | Join its tracked task and inspect canonical authority/actor state. | A cancelled waiter or expiry does not free active work or close a writer. | + +Extend bounded evidence to distinguish consumption by prepared activation from +cleanup of unused credit. Update reducer predicates and codec cases together; +do not set `receiver_cleaned` solely because the owner session matches the +preferred destination. If recovery or ownership was accepted, cleanup follows +the canonical actor lifecycle and cannot stop a writer merely to free an +advisory reservation. + +Public W4 tests must race ordinary acquisition against preparation on the same +ledger, expire credit immediately after confirmed release, and drop a cleanup +reply. Each test must reach current serving plus confirmed resource settlement, +or retain an inspectable charged blocker without claiming cancellation. + +## Durable controller state and bounded execution + +Use an application-owned strongly consistent CAS backend. Its logical records +are specified here; concrete object paths, database schema, credentials, and +deployment installation belong to the adapter. + +| Record | Contents | +| --- | --- | +| Fleet head | Format version, scope, CAS revision, controller claimant/epoch/lease expiry, active operation ID, bounded outstanding movement permits, and committed references to intent, enrollment, and progress pages. | +| Node intent | Stable physical NodeId, desired mode, monotonically increasing intent revision, operation ID, and targeted observed session. The stable intent survives reboot. | +| Operation | ID, request idempotency key, request digest, mode, target node, creation/deadline, phase, last error/blocker summary, and progress cursor. | +| Cell attempt | Exact identities, source generation/authority observation, snapshot digest, reservation, action state, confirmed release/activation evidence, and last movement time. Store in bounded pages. | +| Enrollment entry | Request identity, participating physical nodes and boot sessions, role/epoch, checked intent revisions, and pending/completed/retired evidence. Retain failed-session obligations until canonical closure. | +| Accepted action and result | Stable attempt/action identity, exact execution inputs, committed acceptance, and immutable result or unresolved marker. Retain checked release/acquisition/recovery evidence until its attempt and retention obligations are settled. | + +`compare_exchange` must linearize controller epoch, active operation, and fleet +permit allocation together. For an object-store adapter, write immutable +versioned progress/operation pages first, then CAS one fleet head referencing +their digests. For a transactional database adapter, commit the equivalent +state atomically. A progress page is not published merely because its PUT +succeeded. Derive aggregate counts from committed transitions and rebuild +them after restart; never acknowledge a maintenance request before its intent +is reachable from committed journal state. Orphan journal pages are harmless +and may be collected only by an adapter-specific retention policy. + +Use one concurrent planned node drain per fleet initially. Repeated identical +requests return the same operation; reuse of an idempotency key with different +content is a conflict. A second maintenance request is reported as busy. +Automatic relief can share the remaining movement budget, with maintenance +and sustained pressure prioritized over ordinary balancing. + +The controller lease is acquired and renewed by CAS. Every journal transition +checks its epoch. Action receivers verify current journal authorization before +starting a side effect and bind it to one durable attempt. Advancing the epoch +prevents the old controller from allocating new work. An already accepted old +attempt may finish; the successor adopts its outstanding permit and observes +its result before allocating replacement work. A controller lease alone does +not cancel a delayed RPC or an accepted actor operation. + +When no accepted action can be found, `ResolveUnaccepted` must prove absence +for the exact attempt, effect, physical node and boot inside the same head CAS +transaction. Advancing the head revision fences delayed old authorizations +before retry. If acceptance wins first, resolution fails and the controller +adopts that original work. A separate lookup cannot authorize retry. Keep both +permits charged; expired preparation or release proceeds to independent +receiver cleanup without fabricating source-release evidence. + +Durably record accepted action identity before invoking the runtime. On a lost +reply, inspect the actual actor/session/Cell authority and reconstruct the +result; do not assume the source still owns the Cell or repeat a release +against a newer generation. An unknown result remains charged to the budget. +Never reclaim a permit just because its controller lease or caller deadline +expired. Require terminal action evidence, confirmed cancellation/cleanup, or +orchestrator evidence that the relevant process ended. A partition may thus +stall optional movement while local admission and normal recovery continue. + +Keep committed node intent for previously drained physical nodes when the +fleet head advances to a later operation. A head containing only the current +operation is insufficient to enforce cordons after reboot. For an object-store +adapter, the head must reference the immutable intent registry as well as +progress; for a database adapter, update and retain those rows transactionally. +Likewise, publish completed-move history in the same CAS that retires its +permit. Restart must preserve cooldown even if the process dies immediately +after retirement. + +| Uncertain action | Required reconciliation before advancing | +| --- | --- | +| Prepare reply lost | Inspect the exact receiver session and attempt. Reuse its reservation or confirm cleanup; never reserve again under a new attempt while the old cost is unknown. | +| Release reply lost, original source still alive | Inspect the original generation and accepted-action record together with fresh Cell authority. Retry only the identical authorized action when its original execution is known not to have started. | +| Source dies during release | Retain the permit. Use the existing fenced recovery path and its exact publication/follower proof; do not infer a clean release position from a newer owner's root. Record a distinct recovered outcome if clean-release evidence cannot be reconstructed. | +| Activation reply lost | Verify current authority and actor readiness at the required position. A cached acknowledgement is insufficient; reconcile the preferred reservation if another owner won. | +| Receiver session disappears | Prove that session's process ended or join confirmed cancellation. Settle its charged resources independently of finding a new owner. | + +### Atomic action acceptance and result publication + +Extend the proposed journal adapter with the following logical operations. +Use its existing boxed-future style. The exact Rust names are deliverables of +W1 and W4; the transaction requirements are part of this design. + +```text +accept_action(action, physical_node, boot_session, now_ms) + -> New(accepted_record) | Existing(accepted_record, result) | Conflict +load_action(scope, action_key, physical_node, boot_session) + -> accepted_record and immutable result or unresolved state +publish_action_result(accepted_record, checked_result) + -> committed result or the identical already committed result +authorize_inspection(request_with_nonce_head_registry_endpoint_deadline, now_ms) + -> current read authorization or conflict; no effect or cached result +record_acquisition_basis(original_acceptance, checked_idle_control, captured_at) + -> durably committed original basis or conflict +load_acquisition_basis(original_acceptance) + -> retained basis or explicit absence +``` + +First acceptance must atomically check the committed head revision, live +controller epoch, action phase, scope, exact endpoint session, deadline, +physical-node intent, and charged attempt, then publish the accepted record. +A separate head read followed by an unconditional acceptance write does not +satisfy this contract. With immutable object pages, acceptance becomes visible +only through the successful head CAS; with a database it is one transaction. +An ambiguous backend response requires a read by the same action identity. + +The current `FleetAction::key` deliberately omits controller authorization +fields and some execution inputs. Treat it as an index. For a duplicate, +compare the original immutable movement specification, including physical +nodes, incarnation, generation, epoch, cost, snapshot digest, and deadline. +For maintenance compare operation identity, request digest, target, and intent +revision. An adopted controller may refresh authorization fields without +changing those inputs. A different execution payload under the same key is a +conflict. Deadline extension follows an explicit committed operation revision; +it cannot silently rewrite an already accepted action. + +The host binds its executor at startup to one fleet/application scope, +physical NodeId, and current leased session. The application supplies and +authenticates that binding. None of the local movement methods establishes +remote authorization. First acceptance checks the current head; an existing +accepted effect may complete after controller replacement and retains its +original permit and identity. + +Once accepted, the host owns execution and result publication through +completion even if the transport waiter is dropped. Use bounded retained +receipts and tracked finite tasks, with charges on the existing resource +ledger. Integrate their join into the existing host drain. Do not register +successful finite actions as long-lived facility tasks whose completion would +make the node unhealthy. A facility draining under `shutdown_lock` must not +wait for an action that needs the same lock. + +Publish an immutable result only from checked canonical evidence. Preserve +the original source error separately from its bounded blocker classification. +If an effect may have happened, a returned `Fenced`, timeout, or transport error +does not prove rejection. Keep it unresolved unless inspection establishes a +definite result. A repeated identical result is idempotent; incompatible +terminal results are a conflict requiring investigation. A retained activation +result still needs fresh authority and actor checks before the reconciler +counts the Cell as currently serving elsewhere. + +Use the separate `FleetInspectionRequest`/`FleetInspectionObservation` path +for that fresh check. It binds the exact head action, registry version, endpoint, +nonce and capture deadline, and retains actual capture start/end. A new pass +uses a new nonce. `validate_for` compares the entire request, rejects future +or expired evidence, and bounds age from capture start. The host's read-only +inspection cannot resume recovery with only retained input. Retained action +results stay historical. `apply_fleet_action` refuses raw Inspect dispatch; +the request-bound method is the canonical observation path. New inspection +record kinds 17 and 18 require the +same reader deployment gate as other new fleet records. + +After an exact clean release, the fresh request may target an actual successor +on another node or a new boot of the preferred physical node. Validate the +current journal and registry, authenticate that endpoint, and prove native +actor admission plus current authority at or beyond the release. Source node +and session aliases and mismatched preferred session/node pairs remain invalid. +This read cannot authorize acquisition, recovery or receiver cleanup. Effect +acceptance and replay remain bound to their original source or receiver. + +When its direct serving check is unresolved, the reconciler uses a bounded +authenticated Cell row +under the rechecked durable roster to select an established current boot. That +row is a routing hint; the request-bound native inspection supplies serving +proof. Retirement rechecks the actual successor. Resource cleanup always +targets the original receiver and requires its independent retained proof. +Partial role coverage can select a known writer but cannot prove absence or +node finalization. Existing record bytes and action keys are unchanged; old +inspection readers refuse the newly permitted endpoint shape. Qualify and +deploy upgraded readers before enabling this route across mixed versions. + +For prepared acquisition, retain the exact canonical Idle control, Cell and +incarnation, epoch, pinned root/sequence, observation time, and original accepted +action. Require no owner or recovery claim, compatibility with the receiver, +and a position consistent with the proven release. Confirm durable basis +publication before invoking acquisition with the same authority observation. +The basis is historical input; it proves neither a successful ownership CAS +nor an active actor, and it is not failed-owner recovery evidence. + +An identical basis write returns the originally retained observation time; +republishing must not refresh it. A changed control conflicts and cannot +overwrite that evidence. A lost basis-write reply prevents takeover until a +lookup confirms the same record. Later successor publication cannot erase it. +W4 must test a paused basis write, an ambiguous committed write, a failed +acquisition CAS, and a successor advancing its root before result lookup. +Each case checks durable evidence and actual authority independently. + +W1 and W4 must implement durable action acceptance and result records alongside +the canonical publication/release boundary. Use the distinct +`Recovered` outcome for an attempt whose source failed and whose clean release +cannot be proved. It carries the original Cell/source identity, the verified +recovery basis and position, and current actor-backed successor evidence. +Retain the checked acquisition/recovery basis before successor admission so +later publication cannot erase the evidence needed to reconcile an attempt. +Use existing pinned roots and recovery manifests; these records confer no +ownership and change no Cell authority or LTX format. + +The reducer must keep clean release and recovery separate in status and +history. A recovered attempt may retire only after the original source is +fenced, canonical recovery is proved, an eligible successor serves the same +incarnation at the required position, and unused receiver work is joined. +Missing historical evidence remains an explicit unknown-outcome blocker. +W9 must interrupt release before and after its authority CAS, then advance the +successor's root before controller restart. A newer root or sequence alone +cannot satisfy either evidence path. + +Each reconcile pass first reconciles outstanding attempts, then admits new +work. Persist progress before advancing dependent side effects. CAS conflicts +cause a reread; storage or authorization failures stop new planned moves. +Apply bounded exponential retry with jitter to transient failures and preserve +the original source error. Wake on results, lease events, and operator changes; +periodic polling is the fallback. + +The fleet budget limits admitted planned movement, not all physical fleet I/O. +Emergency local pressure eviction, ordinary requests, and crash recovery keep +their existing independent admission paths. Count those separately in +telemetry; do not claim that a fleet permit cap bounds partitioned residual I/O +or unrelated recovery work. + +On an Active donor under Shedding or Critical pressure, pass the actor's actual +`last_used_ms` through `CellTransferDemand` and prefer recent settled Cells. +Emergency local eviction remains oldest-first. This reduces competing selection +without holding actors or pausing local protection. Drains and ordinary balance +retain Cell identity ordering. A snapshot is still advisory: exact release must +refuse a racing close, preserve its error and settle unused receiver admission. + +| Bound | Initial choice and status | +| --- | --- | +| Periodic reconciliation | Proposed 15 seconds, with progress-triggered wakeups. One pass ends at its supplied deadline. | +| Freshness, residence and cooldown | Retain the planner's 30-second observation window and 60-second ordinary residence/cooldown. Validate clocks; use monotonic local deadlines. | +| Transfers | Retain the planner's maximum two per batch; additionally admit at most two unresolved planned attempts fleet-wide initially. | +| Restore demand | Retain the 8 GiB per-batch ceiling and cap the sum reserved by unresolved planned attempts at 8 GiB initially. | +| Receiver oversubscription | Charge every planned reservation in projected demand and actual local admission. Advertised headroom must include reservations. | +| Fleet input | Retain the planner's 10,000-node/demand input bounds; page larger inventories and report unsupported complete-snapshot size. | +| Node/progress pages | Proposed 128 entries; do bounded work per pass and retain cursors. | +| Node-local movement | Reuse the existing actor `MovementBudget`; add no host-specific competing rate limiter. | + +These are implementation starting bounds, not established production capacity +or latency claims. Put them in one validated profile rather than scattered +environment switches. A Cell exceeding the byte ceiling reports a blocker; +changing that ceiling requires measured receiver capacity and a new profile. + +## Reconciliation pass recipe + +Implement one pass in this order. The caller supplies time and a monotonic +deadline; the pure reducer receives observations and reads no clock. + +1. Load the committed head and acquire or renew its controller lease by CAS. + On conflict, reread. Without a live epoch, return without dispatching. +2. Inspect every unresolved attempt before admitting a replacement. Commit + newly established evidence, cancellation, and cleanup through the reducer. + Atomically publish terminal history when retiring a permit. +3. Apply committed node intent. Ensure maintenance cordon is installed on the + current physical node/session; session replacement updates the operation + without clearing desired mode. Refresh pending role obligations. +4. Collect bounded observations. For ordinary balancing, require a complete + roster and the post-batch freshness barrier. For pressure or maintenance, + require fresh source and receiver evidence without inventing complete + fleet counts. +5. Select work in priority order: finish accepted actions, evacuate the + maintenance node, relieve sustained pressure, then correct ownership skew. + Call the existing planner and project outstanding reservations before + ranking receivers. Admit only what the shared count/byte permits allow. +6. CAS the exact attempt and its next action phase before dispatch. Prepare + the receiver before committing source release intent. Authenticate and + revalidate the committed action at the receiving endpoint. +7. Observe the result and CAS it. If the reply is ambiguous or the deadline + ends, retain the attempt and its cost for the next pass. Every dependent + action follows a committed predecessor; no remote success advances only + an in-memory controller state. +8. Finalize maintenance only after relocation, inventory, and role barriers + pass. Execute the existing drain lane, verify Stopped and withdrawal, then + commit Completed. Return bounded progress and the next wake deadline. + +A pass may stop after any committed step. Reconstructing the facade from the +same journal must continue from that point. The application can disable new +planned moves while continuing steps needed to settle accepted attempts and +inspect maintenance; stopping optional scheduling does not erase intent. + +For each inspection, construct a new nonce bound to the complete action, +exact registry version, physical node and boot, and exclusive capture deadline. +Validate the response against that entire request, its original capture +interval, and a finite maximum age measured from capture start. A durable +action result supplies historical effect evidence; current serving requires +the fresh authority and actor check. Recompare the head and registry in the +transaction that commits a dependent decision. A conflicting revision requires +a new observation pass, and a timed-out capture leaves an unknown result. + +## Maintenance lifecycle and busy Cells + +```mermaid +stateDiagram-v2 + [*] --> Requested + Requested --> Cordoned: intent persisted and admission closed + Cordoned --> Evacuating: fleet observes current intent + Evacuating --> Closing: relocated writers and settled role obligations + Closing --> Completed: host stopped and session withdrawn + Evacuating --> Evacuating: blocker, retry, or extended deadline + Closing --> Closing: inspect and resume incomplete shutdown +``` + +`Blocked` and `DeadlineExceeded` are conditions attached to the current phase, +not successful terminal phases. A deadline stops scheduling further movement; +already accepted work is observed to completion. Before terminal shutdown, keep +the source lease, required facilities, and remaining owners alive. Once normal +shutdown has begun, report its actual partial state and error; do not promise +that all facilities can be restarted after a closing failure. + +Persist the cordon by stable NodeId before acknowledging the maintenance +request. A restarting process checks that intent before opening acquisition or +readiness, so reboot cannot silently rejoin during maintenance. Rejoining +requires an authorized newer Active intent and a new validated serving session. +The initial workflow drains through Stopped; it does not offer cancellation +that reopens an already quiescing actor. + +Scale-down's idle path alone cannot evacuate an actively used node. Add a +maintenance-specific actor transition that stops new foreground admission for +one selected Cell and lets already accepted work finish through the existing +coordination schedule. Inventory and stop new Activity/Effect claims for that +Cell while retaining the completion paths needed to settle existing claims. +Do not stop all node work producers before their outstanding work can settle. + +This requires distinguishing new foreground/claim admission from completion +of an already issued lease. Keep a bounded completion path that validates the +existing exact lease token and normal command identity; it cannot issue new +claims or bypass the normal transaction/durability gate. Once those obligations +are settled, close that path too before final publication and release. A late +completion after release uses current routing and existing lease/deduplication +checks at the successor. + +The maintenance readiness check is distinct from the existing conservative +idle-transfer predicate. It describes what can travel in the exact root after +source execution has stopped. Implement and qualify these semantics explicitly; +never relax `TransferWorkInventory::is_settled` globally to make drains pass. + +| Work at the maintenance barrier | Required treatment | +| --- | --- | +| Accepted SQL/primitive command or query | Complete or resolve according to existing admission/outcome semantics; retain the durability gate. | +| Queued, unclaimed durable Queue/Workflow/Effect work and future timers | May move only once new source claims are stopped and the exact published root contains the work. The successor must resume it through normal scheduler and deduplication semantics. | +| Live Queue delivery, Activity, or external Effect lease | Allow authorized completion, or wait for existing expiry/reclaim rules. Never erase a lease to force progress. Unknown external execution remains a blocker. | +| Running Workflow waiting for an external event | Preserve its durable state and prove an event/completion routed to the new owner behaves correctly. This is a new maintenance qualification requirement. | +| Due Cron tick or Effect send | Stop new dispatch at the barrier; settle any accepted dispatch and preserve deduplicated successor delivery. | +| Blob upload/stream, backup pin, migration, unknown schema/inventory | Keep the existing retention and ownership obligation until its owner proves closure or safe transfer; otherwise report a blocker. | +| Unpublished acknowledged follower tail | Wait for exact object coverage for planned release. A failed owner follows existing recovery instead. | + +Before each release, refresh both source generation and destination admission. +Keep public routing to remaining source owners while the node is only +cordoned. Reject acquisition locally even if an old advertisement is cached. +Remove public readiness when node request admission closes; keep the peer and +recovery paths required by remaining obligations until their lifecycle ends. + +## Evacuating readers and follower responsibilities + +A node with zero owned Cells may still retain the only recoverable recent tail +for a different owner. Add a separate role-obligation inventory before +maintenance can claim the node is safe to stop. + +1. Stop new reader and follower enrollment. Existing follower appends remain + admitted while their enrolled epoch needs them; a stale advertisement must + not admit an entirely new lane after cordon. +2. Reconcile reader policy away from the node and verify the replacement count + required by application policy. Release local read views through the host's + existing reader lifecycle. The manager now uses `CellReadReplica::close_and_join` + to detach every peer clone's view and join accepted native/query/refresh work + before removing ownership. Cancelled waiters retain that obligation for the + next drain. Managed producers now expose `ReadReplicaManager::evacuate` for + one exact Established reader under the current Evacuating operation. It + traverses the complete durable roster, checks selected Active managed boots, + probes actual readiness before and after closure, and confirms the original + retirement through the existing producer. No adequate spare preserves the + local reader. Periodic repair on a Draining node cannot bypass this check by + pruning changed placement. The returned per-reader interval is not complete + role settlement; persist/revalidate its replacements and finish failed-owner + and fleet-wide barriers. `FleetFailedReaderRetirement` now closes an exact + failed receiver request through the same journal under permanent canonical + fencing and independently joined original process evidence. Capture of the + process request does not require other roles to settle first; boot retirement + still does. A failed source with a live receiver needs ordinary joined reader + closure. Provider qualification and persisted replacement policy remain + required. See the [reader lifecycle contract](../crates/cellule-host/docs/read-replicas.md#evacuate-a-managed-reader). +3. Enumerate local follower lanes and all authoritative node-log epochs that + reference the physical node, including expired/recovering owner sessions. + Missing or incomplete inventory blocks finalization. Live advertisements + alone are insufficient evidence. +4. Ask each live owner to object-cover and retire the old epoch, then recruit a + replacement ensemble excluding the maintenance node through the existing + node-log rotation/retirement path. If supported by the declared profile, + existing object-proof fallback can continue writes during recruitment. +5. For a dead owner, finish the existing seal/gather/recovery-overlay protocol + before releasing required tail copies. A missing owner is not proof that + its follower data is unneeded. The runtime now supplies explicit recovered + member retirement, fresh canonical receiver authorization and a complete + native proof before the Retired tombstone CAS. Preserve original enrollment + identities. `FleetRecoveredFollowerRetirement` now publishes/revalidates the + complete original member set through the durable journal. `FleetFailedBootRetirement` then binds the original Established boot, + completed related requests, canonical permanent fence and retained process + evidence supplied by `FleetFailedBootProcesses`. The application authenticates + actual termination/nonexecution and joins accepted external jobs; neither + expiry nor recovery creates that proof. Finish actual process/provider + qualification, controller integration, operation action ownership and replacement-policy checks before + treating the physical role as settled. +6. Confirm that no admitted new enrollment or unresolved tail obligation can + appear after the final inventory barrier, then complete host shutdown. + Retain retired files and their fences under the current grace/collection + rules; maintenance is not permission to delete retained data. + +An application requiring a particular reader/follower redundancy level keeps +maintenance blocked until replacements meet that policy. No-spare-capacity and +object-store outages have an explicit blocked result. Maintain the current +single-writer and command durability contracts throughout. + +Operation completion requires: all affected owned Cells confirmed serving +elsewhere at an adequate position, zero local live/transitioning owners, +settled reader/follower obligations, successful facility/runtime drain, +`NodeState::Stopped`, and confirmed withdrawal of the target session. Report +`released_cells`, `activated_cells`, `remaining_cells`, unresolved attempts, +blocker counts, and outstanding role obligations separately. + +## Operator controls and runbooks + +Implement these application commands against the same committed journal used +by the reconciler. These are proposed logical operations, not framework HTTP +routes or an existing CLI. The application supplies authenticated identity and +authorization. Operator acknowledgements follow durable CAS publication. + +| Command | Required input | Durable effect and response | +| --- | --- | --- | +| Request maintenance | Scope, stable NodeId, observed session, idempotency key, deadline, expected intent revision. | Persist Draining intent and an operation ID. Return its committed revision and phase; identical retries return the same operation. | +| Inspect operation | Operation ID and bounded progress cursor. | Return phase, deadline, released/activated/remaining counts, unresolved attempts, role obligations, blocker classes, and last progress time. Include the current source session. | +| Extend maintenance deadline | Operation ID, expected revision, later deadline. | CAS the deadline for the same unfinished operation. Never create a replacement attempt merely to extend time. | +| Stop new planned moves | Scope and expected policy revision. | Disable allocation of new movement attempts; continue inspecting and settling accepted attempts. Keep maintenance intent. | +| Resume planned moves | Scope, expected policy revision, qualified profile digest. | Enable only the modes that have passed their rollout gates. | +| Return node to service | Completed maintenance operation, exact intent revision, stable NodeId, and new validated session. | Commit a newer Active intent, then allow normal startup probes and admission. An old session or unfinished operation is rejected. | + +The operation ID is the support handle. Keep identifiers in paginated status +and correlated logs; metrics use bounded phase and blocker labels. Publish +`safe_to_take_offline` only when the complete finalization proof is committed. +A phase name, deadline, zero writer count, or controller acknowledgement alone +cannot set that flag. + +### Planned maintenance + +1. Check receiver capacity and required reader/follower redundancy using fresh + observations. Submit one idempotent maintenance request for the physical + node and save its operation ID. +2. Verify committed cordon and local refusal of new roles. Existing owners + remain routable while evacuation progresses. +3. Inspect separate relocation and role-obligation progress. Resolve typed + blockers using the cases below; extend the deadline on the same operation + when appropriate. +4. Take the process or machine offline only after Completed, Stopped, session + withdrawal, and `safe_to_take_offline` are confirmed. +5. On return, use a new session and the authorized newer Active intent. Run + normal startup probes before readiness; publish fresh capacity before the + node becomes a proactive receiver. + +### Overload and failed progress + +| Observed condition | Operator action | Required recovery evidence | +| --- | --- | --- | +| Sustained pressure | Verify the signed tier and limiting resource; inspect eligible donors and receiver headroom. Add capacity or reduce offered load if there is no receiver. | Pressure returns through classifier recovery; admission opens only if node mode is Active. | +| One hot or oversized Cell | Inspect its actual size and validated cost. Partition or limit application demand, or provision a receiver/profile with qualified capacity. | The Cell fits real admission bounds; ordinary cooldown prevents repeated relocation. | +| No destination capacity | Keep source serving and maintenance cordon intact. Add eligible capacity or change authorized redundancy policy. | A fresh receiver reservation succeeds before source release. | +| Busy or externally leased work | Inspect work class, exact lease, and publication progress. Let accepted work complete or follow existing expiry/reclaim rules. | Fresh maintenance inventory proves quiescence; no manual lease deletion. | +| Unknown action outcome | Inspect the exact attempt, source generation, receiver reservation, and current authority. Restart the reconciler against the retained journal if needed. | Terminal or confirmed cleanup evidence is committed before permit reuse. | +| Controller or journal unavailable | Restore CAS service and supervised controller execution. Preserve node intent and accepted work. | A new live epoch adopts outstanding attempts; no new work starts without committed authorization. | +| Foreign follower obligation | Inspect every referenced epoch, including dead owners. Request ordinary owner rotation or complete fenced recovery. | Required tails are object-covered or recovered and safely retired under current rules. | +| Deadline exceeded | Keep the operation inspectable. Resolve the blocker and extend its deadline; inspect actual state if Closing already began. | The same operation resumes and reaches all finalization barriers. | +| Closing facility failure | Preserve the original error and inspect retained tasks, leases, SQLite handles, and withdrawal state. Resume the canonical shutdown path. | All retained owners are joined and Stopped plus withdrawal are confirmed. | + +Exercise each runbook through the W8 example or W9 fault runner. A runbook +passes only when its documented status distinguishes blocked, released, +serving elsewhere, and safe to stop without relying on internal log guesses. + +## Compatibility and rollout + +The advertisement JSON structures use `deny_unknown_fields`. Consequently, +adding fields to schema 3 can break an old decoder before it reaches the +existing unknown-placement-version fallback. Do not assume a version increment +alone preserves identity routing in an old binary. + +Use a reader-first rollout: + +1. Add explicit version-aware decoding and signature verification for schema + 2 and 3. Preserve schema 2 serialization/signing bytes exactly. New optional + fields must be omitted from schema 2 production output. Reject tampered, + partial, and internally inconsistent known-version records. +2. Deploy those readers everywhere that decodes node records, including + controllers, gateways, recovery workers, and administrative tools. Continue + emitting schema 2 and leave automatic movement disabled. +3. Confirm that the complete deployment population has the new decoder, then + enable schema 3 writers through application rollout policy. Require fresh + schema 3 observations for the new proactive-movement contract. +4. Enable observation-only planning, then a bounded maintenance canary, then + pressure relief, and finally ordinary balancing after their evidence gates. + +To roll back movement, disable new planned attempts and reconcile already +accepted work. Keep node cordons and operation history until resolved. To roll +back the binary decoder, first return writers to schema 2 and prove every +persisted schema 3 record reachable by old scanners has been safely replaced +or retired under existing authority rules. Waiting for advertisement TTL alone +does not prove that strict old readers will never encounter an expired record. +Keep the reader-capable release as the rollback floor until that check passes. + +Deploy journal/action readers before emitting new operation states as well. +The current additions preserve previous layouts and discriminants: + +| Addition | Wire discriminator | +| --- | --- | +| CleaningReceiver phase | 10 | +| Recovering and Recovered phases | 11 and 12 | +| Recover remote action | 7; Retire 6 remains journal-local. | +| Recovered outcome | 11 | +| RecoveryBasis and RecoveryEvidence record kinds | 10 and 11 | +| RegistryVersion, IntentPage, EnrollmentRecord, EnrollmentPage, and EnrollmentSpec record kinds | 12 through 16 | + +Only the new Recovered phase carries its recovery evidence extension. Old strict +readers reject these unknown states/kinds; mixed-version qualification must +exercise that failure and the reader deployment gate. + +Keep Cell IDs, descriptors, control roots, object layouts, LTX formats, and +command receipts unchanged. New journal and action envelopes carry an explicit +format version and canonical bounded encoding. Product management messages use +the application's authenticated adapter; extending the existing runtime peer +protocol, if necessary, requires updating its producers, consumers, contract +file, validator, and compatibility fixtures together. + +## Source map and first implementation slice + +Paths below are relative to the workspace root. Existing files are reading +and extension points; proposed files are deliverables. Read producers, +consumers, tests, and the nearest `AGENTS.md` before changing each contract. + +| Area | Existing entry points | Proposed implementation location | +| --- | --- | --- | +| Pure decisions and journal records | `crates/cellule-runtime/src/fleet/placement/mod.rs`; current `fleet/operations/` foundations | Focused records, transitions, evidence, and codec files under runtime `fleet/operations/`. | +| Signed observations and admission | Runtime `node/capacity.rs`, `node/advertisement/`, `node/directory/advertisement.rs`, `fleet/admission/` | Extend these paths and their colocated tests; one shared admission state. | +| Ownership and demand inventory | Runtime `cell/actor/inventory/`, `cell/worker/inventory.rs`, `cell/actor/tasks/residency.rs` | Finish paginated host observation and its generation checks. | +| Admission and activation credit | Runtime `cell/actor/receiver.rs`, `cell/actor/acquire.rs`, `cell/worker/mod.rs`; LTX host resource admission | Host action binding and receipt ownership around the runtime's prepared receiver; keep actual tokens on the canonical runtime path. | +| Exact release and busy work | Runtime `cell/actor/runtime.rs`, `cell/actor/lifecycle/scheduling.rs`, `cell/actor/tasks/movement.rs`, `coordination/`, `primitives/maintenance.rs`; host `node/scale_down.rs` | Consume `release_idle_cell_at` evidence, then extend canonical effects and actor messages for busy maintenance; preserve idle-transfer semantics. | +| Driver and application adapters | Host `lib.rs`, `builder.rs`, `node/mod.rs`, `fleet/actions.rs`, `fleet/journal.rs`, `fleet/cells.rs`, and `fleet/movement/` | Add focused driver and observation/transport adapter modules under `fleet/`; extend the existing action and journal paths. | +| Reader/follower evacuation | Host `durability/`, `read_replicas/`; runtime `node/directory/log.rs`, `follower/`, `recovery/manifest/` | Extend requested rotation and canonical role closure; consume bounded inventories. | +| Finalization | Host `node/lifecycle.rs`, `node/scale_down.rs`, `tasks.rs` | Share the existing drain lane; require fleet evidence before finalization. | +| Public scenarios and evidence | Runtime `tests/fleet.rs`, host `tests/node.rs`, runtime `qualification/` | Focused integration modules, host `minion/main.rs`, and the versioned qualification profile. | + +The first implementation slice is one settled SQL Cell moving between two +independently leased nodes inside a three-node reference fleet: + +1. Complete W1 contracts, including unknown and recovered outcomes, retained + physical-node intent, and atomic permit/history publication. +2. Finish W2–W3 observation and local admission needed by that slice. Obtain + two real unchanged demand samples and fresh receiver evidence. +3. Implement W4 prepare, exact release, reserved activation, and result lookup + through the real host. Include duplicate and lost-response tests immediately. +4. Implement W5 against strict CAS reference adapters. Commit each action + phase before dispatch and reconstruct the driver from the retained journal. +5. Add the first W8 overload scenario as this slice's executable demonstration; + expand it with maintenance and fault scenarios as W6–W7 land. W8's final + exit still requires every declared scenario. + +Keep this slice's public test and example in each subsequent package's focused +checks. Add busy primitives, foreign tails, process faults, and deployment +qualification in dependency order; the settled slice does not establish full +maintenance support. + +### First slice commit sequence + +Each row is a reviewable implementation increment. The next row consumes its +exported contracts and public evidence; avoid adding unused parallel paths. +Use the focused gates in [Execution checkpoints](#execution-checkpoints) and +retain the selected test counts. + +| Order | Change and handoff | Required public evidence | +| --- | --- | --- | +| 1 | Complete W1 accepted-action, recovered-evidence, retained-intent, and history contracts. Add bounded codec and reducer cases before connecting I/O. | Duplicates preserve inputs; unknown work retains cost; stale epochs cannot allocate; clean release cannot be fabricated from a successor root. | +| 2 | Implement a strict reference journal and enrollment registry behind the adapter contracts. Publish pages through head CAS and check acceptance atomically. | Two clients racing acceptance or permit allocation produce one committed effect; lost CAS replies can be reread; retained intent and cooldown survive adapter reconstruction. | +| 3 | Complete the existing W4 host executor's recovery and unknown-result inspection. Reuse its startup binding, owned finite-action completion, prepared receivers, exact source release, and retained acquisition basis. | Prepare refusal leaves the source serving; duplicate release cannot target a newer generation; dropped waiters retain execution; concurrent shutdown joins work and frees resources. Original task failures remain inspectable across repeated drain attempts. | +| 4 | Finish the W2–W3 complete observation adapter, including roster revision and enrollment barriers. Connect real demand and pressure samples. | Missing members and changed revisions block balancing; changed generation invalidates demand; cordon rejects new enrollment while retaining existing obligations. | +| 5 | Add W5 `reconcile_once` using the existing planner and reducer. Reconcile charged attempts before choosing a donor; journal every dependent action. | Three independently leased nodes converge; competing controllers stay within the shared count/byte caps; reconstructed driver resolves lost replies without a second writer. | +| 6 | Export the first W8 overload example and its invocation. Keep the same driver, adapters, action executor, and receipt checks as the integration test. | An overload scenario moves eligible Cells, reads acknowledged state through receipts, prints separate release/activation/blocker counts, and exits with no owned tasks or credit. | + +Then execute W6 busy maintenance, W7 role evacuation and finalization, the +remaining W8 scenarios, W9 qualification, and W10 operational rollout. Supply +measured qualification thresholds before running W9. External prerequisites +are a declared strict CAS backend, authenticated application endpoints, +stable physical-node identity, a complete enrollment registry, and the +documented process/provider fault environment. Record any missing prerequisite +as a package blocker; it cannot be replaced by a simulated success result. + +## Implementation work packages + +Each package has a concrete exit condition. Split large packages into reviewable +commits along the listed modules; do not enable a partly implemented behavior. +Record completed evidence beside the package when execution begins. + +The first useful increment is W1–W5 with settled-Cell movement and its matching +acceptance cases. W6–W7 add complete planned maintenance semantics. W8–W10 make +the feature reproducible for embedders and qualified for deployment. Begin +focused tests with each package; W9 consolidates process/provider evidence and +does not postpone correctness tests until the end. + +### W1 Freeze operation contracts and failure semantics + +Dependencies: none. + +Extend `crates/cellule-runtime/src/fleet/operations/mod.rs` and its focused +sibling files for records, transitions, codecs, and tests. Reuse the existing +registration in `fleet/mod.rs`. Complete the proposed typed observations, +outcomes, and retained registries alongside the current action, blocker, +phase, and permit records. Keep the pure module free of I/O and wall-clock reads. +Define journal/action schema versions and exact maximum record/page sizes. +Implement the distinct recovered evidence path and immutable accepted-action +results specified above; preserve both across controller replacement. +Include revision-checked deadline extension and stop/resume policy transitions +for the operator controls above. Stopping new attempts preserves all accepted +attempts, reservations, and physical-node intent. +Implement the [receiver cleanup transitions](#receiver-cleanup-after-source-release), +including independent credit settlement after release and explicit evidence +that prepared activation consumed the credit. + +Exit: deterministic transition tests cover duplicate events, out-of-order +responses, lost release replies, expired controller ownership, source-session +replacement, and permit retention for unknown work. No transition grants Cell +authority or turns `Released` into `Activated` without evidence. + +### W2 Add signed operational observations and admission reasons + +Dependencies: W1. + +Change `src/node/capacity.rs`, `src/node/advertisement/{mod,codec}.rs`, +`src/node/directory/advertisement.rs`, `src/fleet/placement/mod.rs`, and actor +admission/state as required. Add schema 3 mode, pressure, and sample fields; +implement the explicit schema 2 reader/writer behavior above. Bind the local +pressure classifier's stable output and lifecycle admission reasons to the +published observation. Update reader/follower candidate filtering. + +Exit: signed-field tampering fails; pressure and cordon cannot overwrite each +other; every placement path rejects the same ineligible receiver; schema 2 +fixtures retain identical signature input bytes. Rollout remains reader-only. + +### W3 Expose bounded actor demand and maintenance diagnostics + +Dependencies: W1. + +Change actor `runtime.rs`, `state.rs`, `task.rs`, lifecycle observations, and +worker accounting. Add bounded node pages with generation-bound costs, +residence evidence, all owned/transitioning Cells, and blocker details. Add +role-obligation inventory using existing node-log and follower-store records. +Keep advisory observation separate from action authorization. + +Exit: busy Cells appear with blockers; last-use time is not residence time; +unknown disk demand cannot become zero-cost movement; invalid cursors restart +without silently skipping ownership; cancelled observations leak no permits. + +### W4 Implement admitted receiver reservations and exact actions + +Dependencies: W1, W2, W3. + +Extend `crates/cellule-host/src/fleet/mod.rs` and its current local action, +journal, provider, and movement modules. Complete `CellNode::fleet_snapshot` +and maintenance support in the existing `apply_fleet_action`; preserve the +current host exports. Reuse resource ledgers and `release_idle_cell_at`. Serialize +node drain/finalization against the existing `shutdown_lock`; short local +actions must not hold that lane while waiting on remote network I/O. +Preserve original task and join failures across repeated drain calls; consuming +a completed task handle cannot turn a previous failure into successful shutdown. +The bounded node task group now retains each original join/result, shares it +across concurrent or cancelled drain waiters, and keeps genuine task failures +as sources on retry. Ordinary deadline-aborted work retains its cancellation +join; the existing node-log facility watcher retains its canonical work and +can finish after a deadline. These local guarantees do not supply the complete +role barrier or fleet finalization handoff required by W7. +Join every accepted finite job even when another job fails. Settle its resources +and retained publication obligations independently, and keep the original +failure diagnostic after settled native resources can be released. A failed +join cannot skip its sibling jobs or establish host Stopped. + +Use typed outcomes with original source errors. Give duplicate reservations +and activations a single attempt identity and adopt existing results after +uncertainty. Actual receiver activation consumes the reservation without +double charging. Failed/cancelled preparation joins work before freeing cost. +Capture immutable checked release and acquisition/recovery results at the +canonical boundaries, before later authority changes can hide their basis. +Thread these results into the W1 evidence types and action-result lookup. +Complete trusted Cell lookup and basis publication before acquisition; test +their ambiguous responses. Exercise unused-credit cleanup in Released and +Activating phases, including a same-session ordinary-acquisition winner. + +Exit: public host tests demonstrate receiver refusal before source release, +receiver loss after release, duplicate dispatch, lost result recovery, stale +actor generations, and concurrent shutdown without leaks or a second writer. + +### W5 Implement the reusable bounded reconciliation driver + +Dependencies: W1 through W4. + +Complete driver and adapter modules under `cellule-host/src/fleet/`. Reuse the +present interfaces and `FleetReconciler::reconcile_once`. Drive the pure +operation reducer and existing `fleet_balance`/`plan_transfers`; make these +actual production call sites. Journal before dispatch, reconcile outstanding +attempts first, honor the full-snapshot barrier, and apply one shared planned +movement budget across donors. Retain the actor's existing local budget. + +Exit: an in-process reference adapter moves Cells between three independently +leased runtimes; a reconstructed reconciler resumes the persisted operation; +two competing reconcilers cannot double allocate an attempt or exceed the +shared permit cap. Missing journal/membership evidence stops new optional work. + +### W6 Add controlled maintenance of busy Cells + +Checkpoint: exact-source foreground quiescence, native lease completion, +separate maintenance inventory, and canonical `release_maintenance_cell_at` now +exist. An explicit journal-bound maintenance release phase/action connects the +host executor to that path without reinterpreting ordinary release records. +Public SQL/Queue/Effect/Activity, restored Workflow waits/timers, accepted Cron +dispatch with due-work handoff and deduplicated successor delivery, host release +and pure action contracts have focused coverage in the +[execution evidence](fleet-operations-progress.md). The driver now accepts +busy maintenance demand with a separate configured peak envelope under exact +Evacuating intent and fresh source/receiver evidence; ordinary pressure/count +moves still require settled samples. Complete the remaining host fault cases, +every primitive's acceptance matrix, Blob owners and process/provider +qualification before claiming this work package complete. + +Dependencies: W1, W3, W4. + +Extend the existing coordination kernel with the explicit maintenance +quiescence transition and actor scheduling/worker inventory support. Add +Cell-scoped claim suppression to installed Activity/Effect facilities while +retaining authorized completion and expiry paths. Implement the maintenance +readiness matrix for SQL, Queue, Workflow, Effects, Cron, and Blob obligations. +Keep ordinary idle transfer checks conservative. + +Exit: sustained incoming traffic cannot starve an accepted maintenance +quiescence request; accepted commands retain outcome/durability semantics; +late completions and retries use current authority; unclaimed durable work +resumes after exact-root activation. Unproven primitive classes report explicit +blockers and prevent declaring general maintenance support complete. + +### W7 Complete role evacuation and resumable node finalization + +Dependencies: W2, W3, W5, W6. + +Extend host durability/read-replica orchestration and runtime follower +inventory/retirement adapters to evacuate foreign obligations. Add an explicit +requested rotation trigger to the existing durability supervisor so maintenance +does not wait for its normal frame/age threshold. Use the current retirement, +coverage, enrollment, and recovery paths. + +Refactor local scale-down and fleet finalization to share the same actor +release and node drain internals. Fleet finalization additionally requires +relocation and role-obligation evidence. Do not call the current autonomous +`drain_for_scale_down` loop as the executor of a destination-reserved fleet +batch: it can release additional candidates without that batch's permits. +Preserve its documented local behavior or document an intentional matched +release change; do not silently equate its aggregate count with fleet success. + +The host now retains the entire canonical drain attempt in one fixed task slot, +including its reverse facility order and the existing runtime/withdrawal phases. +Shutdown and the terminal scale-down step transfer their shared lane guard to +that task. Lost callers cannot drop an accepted callback or release the lane +while closing work still runs. Local Running/Returned/Joined observations keep +original results and first/latest failure diagnostics; phase deadlines resume +through the same resource owners. This supplies the native closing owner needed +for terminal action handoff. Managed boots can bind their original authenticated +directory version and Established registry row before readiness. The native +closing task checks canonical withdrawal and committed boot retirement before +Stopped, retaining its original evidence across a deadline or ambiguous reply. +This is wired into the reference example. Complete role settlement, the original +action's join before handoff, and committed operation completion remain required. + +Exit: maintenance with zero local writers but uncovered foreign follower +tails remains blocked; live-owner rotation and dead-owner recovery both clear +it safely; incomplete deadlines remain inspectable; successful operations +confirm host stop and withdrawal. Active maintenance intent survives reboot. + +### W8 Supply the runnable embedding example and production adapter recipe + +Dependencies: W5, W7. + +Complete `crates/cellule-host/minion/main.rs` with support +files under that example's directory. Reuse its local durable journal adapter. +Exercise three nodes and shared strict +CAS-backed test storage with deterministic identities. Supply overload, +maintenance, controller-restart, and receiver-loss scenarios. The example must +invoke the exported driver and real CellNode paths; no duplicate safety logic. + +Document the production adapter recipe in the framework and host guides: +register enrollment before readiness; publish signed observations; persist +NodeId intent and controller epochs; authenticate and deduplicate actions; +serve paginated operation status; and schedule `reconcile_once` in a supervised +application task. Use an example journal namespace outside canonical Cell +storage paths and prove its conditional-write semantics on the chosen backend. + +Exit: the example prints separate released/activated/blocked counts, verifies +receipt-bound readback, restarts the controller from retained journal state, +and ends with no owned tasks or reservations. The application checklist contains +every necessary backend/route/rollout obligation without assuming Crab paths. + +### W9 Add measured fleet fault and compatibility qualification + +Dependencies: W2 through W8. + +Extend runtime/host integration suites and the existing qualification harness. +Register focused new modules in `crates/cellule-runtime/tests/fleet.rs` and +`crates/cellule-host/tests/node.rs`; place private tests beside implementation. +Extend coordination simulations/models for the new maintenance transition. +Reuse the existing `cell_movement_probe` where its contract fits. + +Add a versioned fleet-operations profile and measured scenario runner. Bind +source, binary/image, profile, topology, resource limits, offered load, +movement history, raw faults, and outcomes to the normal evidence pipeline. +Implement every case in the acceptance matrix below; verify selected tests +actually ran rather than accepting a zero-test filter result. + +Exit: public-path fault tests pass, strict old/new codec behavior is exercised +in an actual mixed-version run, and measured movement/availability/resource +gates pass in an isolated provider-backed environment. Existing qualification +profiles and required evidence are not weakened to accommodate failures. + +### W10 Publish operator documentation and enable in stages + +Dependencies: W8, W9. + +Update `docs/framework.md`, `docs/roadmap.md`, host lifecycle/read-replica +guides, runtime deployment/canonical scaling references, and runnable example +instructions. Replace historical claims with links to current implementation +and dated evidence. Update the root prelude inventory only if new symbols are +intentionally added to the runtime root exports. + +Document overload, no-capacity, stuck drain, controller loss, follower blocker, +maintenance deadline, and return-to-service runbooks. Enable observation-only +planning, maintenance canaries, pressure relief, then balancing. Keep the +application's stop-new-moves control available throughout rollout. + +Exit: an operator can submit an idempotent maintenance request, inspect its +phase and blockers, resume after controller restart, and confirm that the node +is safe to take offline using the documented status contract. + +## Execution checkpoints + +Use three delivery milestones. Keep new automatic actions disabled until +their entire milestone has passed; schema 3 production also requires the +reader deployment gate. Maintenance remains incomplete until busy-work and +foreign follower cases pass. + +| Milestone | Packages | Reviewable result | Required evidence | +| --- | --- | --- | --- | +| Settled Cell movement | W1–W5 | Public driver and host actions move an idle Cell between three nodes, with admitted cost and durable restart. | Exact receipts survive movement; concurrent controllers stay within two unresolved attempts and 8 GiB; stale generations and lost replies cannot release another owner. | +| Complete node maintenance | W6–W8 | Busy-Cell quiescence, reader/follower evacuation, resumable finalization, and four runnable example scenarios. | Continuous load cannot starve maintenance; dead-owner follower tails remain protected; reboot retains cordon; only Stopped plus withdrawal completes the operation. | +| Deployment qualification | W9–W10 | Measured campaign, mixed-version rollout/rollback, integration recipe, and operator runbooks. | All acceptance cases and predeclared resource/latency gates pass against recorded binaries and provider environment. | + +For each package, make the smallest reviewable sequence: contract and focused +tests, implementation through the existing canonical path, then public-path +tests and documentation. Mark its exit only after the corresponding tests +run. Preserve unrelated working-tree changes; do not reset partial foundations +to the baseline merely to simplify a patch. + +The following commands are implementation gates, not tests run by this planning +pass. Set the checkout-specific `CARGO_TARGET_DIR` as shown in +[Verification commands](#verification-commands). All must exit zero; filtered +commands must select a nonzero number of relevant tests. + +| Packages | Focused commands | Additional required inspection | +| --- | --- | --- | +| W1 | `cargo test -p cellule-runtime --lib fleet::operations --locked` | Every bounded codec has malformed, oversized, unknown-version, and round-trip cases; permit retirement and history publication are atomic. | +| W2 | `cargo test -p cellule-runtime --lib node::tests --locked`; `cargo test -p cellule-runtime --lib fleet:: --locked`; `cargo test -p cellule-runtime --lib follower:: --locked` | Independent old-reader fixtures preserve schema 2 signing bytes; actual actor samples supply both tier and capacity. | +| W3 | `cargo test -p cellule-runtime --lib cell::actor:: --locked`; `cargo test -p cellule-runtime --lib cell::worker:: --locked` | Busy and transitioning entries remain visible across pagination; costs include restore demand and cannot silently become zero. | +| W4–W5 | `cargo test -p cellule-host --test node --locked`; `cargo test -p cellule-runtime --features test-support --test fleet --locked` | Tests call exported actions/reconciler, inspect ledgers, and reconstruct a controller with retained journal state. | +| W6–W7 | The W3 and W4–W5 commands; `cargo test -p cellule-runtime --features test-support --test primitives --locked` | Cover each readiness-matrix row through public behavior, plus follower-only maintenance and simultaneous shutdown. Run coordination models through their existing documented route. | +| W8 | All four example commands in [Verification commands](#verification-commands) | Each exits zero, checks receipt-bound readback, and joins all owned tasks and reservations. | +| W9–W10 | Broad verification commands below in CI or an isolated snapshot; existing qualification harness with the new committed profile | Store raw process/provider evidence and mixed-binary records. Document the actual commands and environment used; a unit suite is insufficient. | + +After every layout or public-contract change, also run +`python3 scripts/check-boundaries.py`, +`python3 scripts/check-module-layout.py`, and +`node crates/cellule-runtime/docs/validate.mjs`. After documentation changes, +run both document validators. The final milestone additionally requires +format, feature/target compilation, lint, and API documentation gates. + +Record each package's completion with source revision and diff digest, commands, +selected test count, exit status, evidence location, and remaining environment +limitations. A package with an unrun required gate remains pending. + +Pause the affected package and report concrete evidence if the chosen provider +cannot linearize the head/intent/permit contract, costs cannot be bounded, +source failure cannot yield exact recovery proof, or complete follower +inventory cannot be obtained. Keep the typed blocker and continue independent +packages. Do not substitute a weaker authority, fake measurements, or a +successful shutdown count for the missing proof. + +## Acceptance matrix + +| Scenario | Required assertions | Primary evidence | +| --- | --- | --- | +| Node addition and skew | Weighted counts converge with complete fresh observations; no immediate move-back. | Planner properties and three-process integration. | +| Sustained pressure and recovery | No proactive placement onto pressured nodes; local protection works without controller; pressure recovery cannot remove a cordon. | Actor/host tests and constrained process run. | +| Oscillating pressure | Dwell, residence, cooldown, and byte/count budgets prevent repeated movement. | Deterministic sample sequences and measured soak. | +| Missing, stale, forged, duplicate, or mixed samples | Count balancing stops; no invalid sample authorizes receive or release. | Codec/planner/property tests. | +| Concurrent donors and cold activations | Projected reservations and actual admission prevent capacity oversubscription. Same-session ordinary acquisition does not masquerade as prepared-credit consumption. | Host integration under concurrent load. | +| Receiver credit expires or cleanup reply is lost after release | Cancel only proven-unused work; preserve release evidence and the charged attempt until serving and credit settlement are confirmed. Never report Cancelled after a possible source effect. | Public host race and lost-reply tests with ledger assertions. | +| Acquisition-basis write stalls, conflicts, or loses its reply | No takeover before confirmed durable basis; retain original control/time through successor publication. Failed CAS proves no ownership only when its canonical result establishes that fact. | Journal fault injection around canonical acquisition and subsequent root advancement. | +| Receiver failure before release | Source remains authoritative; failed reservation is reconciled. | Public movement action test. | +| Receiver failure after release | Exact root remains recoverable; another eligible node restores acknowledged state. | Process fault after confirmed release. | +| Lost release or activation reply | No false rollback, newer-generation release, duplicate move, or premature permit reuse. | Controller restart plus dropped responses. | +| Source loss around release CAS | Clean release and canonical recovery have separate proofs and status; successor publication cannot erase required evidence. Unknown outcomes retain their permits. | Kill before/after release CAS, advance successor root, then reconstruct controller. | +| Competing controllers and expired leases | New controller adopts unresolved attempts; delayed old actions cannot allocate new work. | CAS adapter and partitioned process tests. | +| Busy SQL Cell during maintenance | New admission closes, accepted work settles, writer count remains one. | Continuous-load receipt/readback test. | +| Queue, Workflow, Effect, Cron, Blob obligations | Exact state and deduplication survive relocation; unknown/live external obligations block as specified. | Primitive public behavior scenarios. | +| Foreign follower-only responsibility | Node cannot finish maintenance while another owner's acknowledged tail depends on it. | One- and two-follower fault scenarios, including dead owner. | +| Enrollment races with cordon | Pre-cordon accepted work is inventoried or joined; post-cordon enrollment is refused locally and by registry CAS. No finalization from a filtered or incomplete roster. | Public concurrent enrollment/drain test with a lost reply and a changed registry revision. | +| Deadline or provider outage | Explicit incomplete status; no forged success or forced durability bypass. | Faulted journal, object store, and shutdown tests. | +| Node reboots during maintenance | Persistent physical-node cordon is respected before acquisition opens. | Fresh-process restart with same NodeId and new session. | +| No destination capacity or one oversized Cell | Clear capacity/budget blocker; no repeated release attempts or fake completion. | Planner and end-to-end status test. | +| Concurrent drain, cancellation, and shutdown | One drain lane, correct deadline semantics, no leaked tasks/SQLite/reservations. | Host lifecycle and runtime release-progress suites. | +| Schema reader-first deployment and rollback | Old decoder failure is understood; bridge readers maintain required routing; unsafe producer/decoder rollback is refused. | Versioned fixtures and mixed binaries against persisted records. | + +Every movement test records authority owner/session/epoch transitions and +command receipts. Compare restored application state with those receipts; +counting successful tasks or log messages alone cannot prove correctness. + +## Operational evidence and limits + +Emit bounded metrics for pressure tier, placement eligibility reason, +unresolved attempts, reserved restore bytes, confirmed releases, confirmed +activations, movement duration, drain phase/age, blocker class, remaining +owners, and follower obligations. Keep operation/Cell/attempt IDs in bounded +status pages and correlated logs/traces. Preserve runtime publication and +follower-proof telemetry so movement can be correlated with foreground p99. + +Before a measured campaign, commit the new qualification profile's workload, +topology, node resource ceilings, allowed logical request error rate, absolute +and baseline-relative p99 limits, convergence deadline, and drain deadline. +These are required inputs; absent limits fail qualification. Do not derive +passing thresholds from the run being assessed. A steady baseline and the +movement run use the same binary, offered load, provider, and resource limits. + +The first campaign must include heterogeneous Cell costs, a hot single Cell, +node addition/removal, overlapping maintenance and pressure, controller and +receiver loss, and follower-only obligations. Report offered and completed +work, definite rejections, unknown mutation outcomes, restore bytes, and time +to regain reserve separately. A permanently overloaded fleet may remain +blocked: redistribution cannot create capacity. + +## Verification commands + +Run commands from the workspace root. For Rust checks on a workstation with +the mounted Workspace volume, set a target directory unique to this checkout: + +```sh +export CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/cellule-f9383af7-fleet-operations" +``` + +The documentation-only plan can be checked now: + +```sh +python3 scripts/check-doc-links.py +python3 scripts/check-doc-rust-fences.py +``` + +During implementation, run focused public suites after their work packages: + +```sh +cargo test -p cellule-runtime --features test-support --test fleet --locked +cargo test -p cellule-host --test node --locked +python3 scripts/check-boundaries.py +python3 scripts/check-module-layout.py +node crates/cellule-runtime/docs/validate.mjs +``` + +The following example commands are deliverables of W8. The current tree supports +`overload` and `controller-restart` with the evidence limits recorded in +[execution evidence](fleet-operations-progress.md); `maintenance` and +`receiver-loss` remain unimplemented: + +```sh +cargo run -p cellule-host --example fleet_operations --locked -- overload +cargo run -p cellule-host --example fleet_operations --locked -- maintenance +cargo run -p cellule-host --example fleet_operations --locked -- controller-restart +cargo run -p cellule-host --example fleet_operations --locked -- receiver-loss +``` + +Use CI or an isolated verification snapshot for the broad gates and process +tests. Provider and fault campaigns also require their documented environment: + +```sh +cargo fmt --all --check +cargo check --workspace --all-targets --all-features --locked +cargo test --workspace --all-features --locked +cargo test -p cellule-ltx --no-default-features --locked +cargo clippy --workspace --all-targets --all-features --locked -- -D warnings +RUSTDOCFLAGS='-D warnings' cargo doc --workspace --all-features --no-deps --locked +python3 scripts/check-boundaries.py +python3 scripts/check-module-layout.py +python3 scripts/check-doc-rust-fences.py +python3 scripts/check-doc-links.py +node crates/cellule-runtime/docs/validate.mjs +``` + +## Completion checklist + +- [ ] Runtime planning has a reusable production execution path. +- [ ] Signed pressure and node mode drive consistent remote and local admission. +- [ ] Capacity reservations and unresolved-attempt accounting survive failures. +- [ ] Maintenance handles busy Cells with qualified primitive semantics. +- [ ] Reader and foreign follower obligations gate safe node shutdown. +- [ ] Durable operation intent, controller fencing, status, and restart work. +- [ ] The reference example and application integration recipe are runnable. +- [ ] Fault, compatibility, lifecycle, and measured qualification gates pass. +- [ ] Documentation distinguishes current capability from measured evidence. +- [ ] Staged deployment and rollback have an operationally verified procedure. diff --git a/docs/fleet-operations-progress.md b/docs/fleet-operations-progress.md new file mode 100644 index 00000000..152f651b --- /dev/null +++ b/docs/fleet-operations-progress.md @@ -0,0 +1,5078 @@ +# Fleet operations implementation evidence + +The [implementation plan](fleet-operations-plan.md) remains the full scope. +This page records focused checkpoints; it does not establish complete fleet +balancing, maintenance, or deployment qualification. + +## October 3 2026 committed original writer reload checkpoint + +`FleetOriginalWriterInventory::load` reconstructs the original writer manifest +through the existing journal's committed operation/process pointer and every +ordered immutable page. It validates the bounded header before allocation and +the complete page set before returning. Capture replay uses this same path. +No partial inventory escapes when a final page is missing or corrupt. Adapter +errors retain their source; one absolute deadline bounds the inline reader. +The journal's existing finite owner retains accepted storage work. + +`None` means no committed inventory at that read. A validated returned manifest +with zero owners represents explicitly retained empty input. Neither establishes +that an accepted capture cannot publish later. Original scope, boot, Controls, +catalog witnesses and collection times remain unchanged. This opaque value is +historical input, not current serving, suffix availability or role settlement. +The [host integration recipe](../crates/cellule-host/docs/original-writers.md) +documents both capture and reconstruction. + +Five new public minion cases cover absence versus complete empty capture, stale +full snapshots, missing/corrupt final pages, original SQL errors and expired +deadlines/invalid keys. Existing reconstruction and canceled publication cases +now use the loader, preserving all 66 original epochs across two pages. The +joined process remains a lifetime stand-in, not an OS-crashed CellNode or +external-job qualification. + +| Verification on the isolated source snapshot | Result | +| --- | --- | +| Full workspace, all features, locked | 1,627 passed; no failures; 36 documented ignored cases. Nested filtered child summaries excluded. | +| Complete canonical minion target | 223 passed; no failures or ignored cases. | +| Distinct passes | 1,850; focused repeats and overlapping local LTX excluded. | +| Focused original writer cases | 13 selected and passed, including all five new cases. | +| Local LTX without default features | 54 passed. | +| All-target/all-feature check, Clippy, API docs | Passed; warnings denied for lint and docs. | +| Format, boundaries, layout, links, Rust fences, SQL/peer, diff | Passed; all 125 Rust snippets parse. | + +The snapshot contains 755 Rust/Cargo paths; manifest SHA256: +`38a5f2367e124f66b5273bc0c6125232ccb801a362bc7473865777bf12b335ef`. +Source archive, command/status records and logs are retained under +`/tmp/cellule-original-writer-reload-evidence`. Snapshot and mounted target use +suffix `cellule-original-writer-reload-704fa11`. This checkpoint adds no persisted +format, authority, retry owner, scheduler or production maintenance effect. + +Highest remaining implementation: aggregate every authenticated original writer +and sealed suffix with current native serving, original physical boot/process, +reader/follower replacement policy and accepted-work barriers. Then implement +SettleRoles/Finalize and join original actions before terminal drain handoff. +Cross-session failure adoption, Cron/Blob owners, complete maintenance and +receiver-loss minion scenarios, W9 campaigns and W10 exercised operations remain +required. The full W1–W10 plan remains active. + +## October 3 2026 lineage publication I/O checkpoint + +The native publisher strictly creates fresh lineage without an absence GET. +A create conflict reads and validates the existing record, then performs at most +one additive ETag merge. Delayed merges cannot erase earlier inputs. Failed and +lost create/update replies retain the original source error unless exact retained +inputs are independently confirmed. The existing publisher owns retries. + +An opaque native `RootPreparation` allows this metadata write to overlap root +and dependency uploads. All native graph checks precede that observation. Its +construction is private; it supplies no uploaded-root, authority, restore, +serving or acknowledgement rights. The same caller-owned preparation future +joins uploads and metadata, using the shared native origin I/O permits. Only +completion of both yields `PreparedRoot`. Cancellation drops inline work and +releases its permit. A bounded private one-entry confirmation avoids repeating +that exact metadata write before authority CAS. External and rebased compaction +proposals retain their exact inputs through the canonical path before CAS. +Verified private compaction composition supplies the final original predecessor +in the native factory before metadata runs. That context is cleared from read +views; the complete proposal and its retained derivation name the same input. + +The runtime recovers its original typed metadata errors while preserving native +LTX retry classes and hints. No detached task, second retry owner, authority, +queue, scheduler, persisted codec or dependency is added. Native prepared views +do not retain the callback. Minion remains canonical at +`crates/cellule-host/minion`, with Cargo target `fleet_operations`. + +Eleven new cases cover fresh publication I/O, competing/stale additive merges, +failed/lost replies, ordering with paused native uploads and metadata, failed +uploads, original source/retry preservation, exact native predecessor identity +and single-permit cancellation cleanup. Existing foreground compaction and +migration cases verify the original published prefix through the rebased path. + +### Verification and remaining qualification + +Final verification passed 1,627 workspace tests and all 218 minion cases: +**1,845 distinct passes**, with 36 documented workspace cases ignored. Local LTX +without default features passed 54 overlapping cases. Workspace all-target, +all-feature check, Clippy and API documentation passed with warnings denied. +Format, boundaries, layout, Rust fences, links, SQL/peer and diff gates pass. +Both unchanged routing profiles were collected against the same 754 frozen +Rust/Cargo paths. Manifest SHA256: +`0958df40474c8f2caaee7753962c5a8f230bfaf5e98b7b20cc359f876b9fdb08`. +Raw commands, source archives, errors and provider evidence remain under +`/tmp/cellule-lineage-routing-evidence`. The isolated checkout and mounted target +use suffix `cellule-reader-prelease-8698183`. + +Earlier published `a2edbe5` is mergeable and passes follower/object capacity, +workspace/MSRV, Compose smoke, contract, website, fuzz and enabled fast models. +Both routing profiles and their aggregate fail: all eight command lanes measure +79.5–88.1% of baseline against the unchanged 90% throughput gate, with +1,024–1,025 extra reads per 1,024 commands. Downloaded original artifacts are +retained. A failing native regression recorded 16 absence GETs for 16 roots; +fresh strict creation removes those GETs. One diagnostic local pair then showed +the remaining serial lineage PUT adding about 3 ms to the authority phase. +That incomplete comparison is retained and does not qualify a fix. + +The first overlapping candidate, published as `4f646b5`, completed a local +leased diagnostic pair. Its compaction metadata named the private compacted +predecessor before the complete proposal was rebased. The second strict create +then hit Store's conflict retry schedule. The candidate's 53 local-c1 spikes +added roughly 25.8 seconds to authority work; mean authority time was 28.24 ms +against 3.05 ms baseline. The incomplete comparison is retained, and the next +pair was interrupted with its child processes joined after this cause was found. +A new real 41-publication compaction regression failed with one lineage read +instead of zero. Moving verified compaction composition into the native factory +before metadata retention passes with zero reads and exactly 41 lineage writes. +Public LTX coverage also compares metadata with the final rebased proposal under +one shared I/O permit. Final full verification was rerun after this correction. + +The final candidate overlaps that PUT and retains the final predecessor under +existing admission. The complete local four-pair leased comparison passes every +unchanged gate. Object-only fails six lanes: forwarded command c1, local command +c16, forwarded query c1, uncached forwarded routing c16, and local/forwarded +expired bursts c16. Command throughput ratios are 77.2% and 86.8% in the two +failing command lanes; all query read counts match baseline. These Mac/ARM runs +are diagnostic local evidence, not Linux deployment qualification. All sixteen +benchmark processes confirmed 6,144 commands and exact final sequence recovery. +The owned provider was inspected, logged and removed after every child exited. + +The temporary wrapper passed both modes to its final single-mode summary and +was rejected. The unchanged comparator was then applied to the saved complete +rows, separately for each mode and together, preserving every failure without +rerunning measurements. The published `4f646b5` Linux run also failed all eight +command lanes; its original artifacts are retained. At published `704fa11`, +follower/object capacity, workspace/MSRV, Compose smoke, contract, website, fuzz +and enabled fast models pass. Its Linux object-only routing artifact also passes +all unchanged gates, with command throughput ratios of 91.7–97.2%; original +artifacts are retained separately from the failed Mac diagnostic. Linux leased +routing remains running. Routing qualification remains open. Workload, pair +count, thresholds and expected +correctness evidence are unchanged. The initial diagnostic wrapper mistakenly +compared one pair and was rejected by the unchanged four-repeat comparator; the +final wrapper compares only complete profiles. A pre-admission verification +snapshot passed but was superseded by the shared I/O-permit correction. Initial +test compilation also caught a shadowed test helper. None qualifies final source. + +The [complete remaining streams](#ci-blocker-and-remaining-work-streams) still +apply. Next after routing qualification: complete authenticated original writer, +suffix, physical boot/process, reader/follower policy and accepted-work +aggregation; then SettleRoles/Finalize and original action joining before drain. +Cross-session failure adoption, Cron/Blob owners, maintenance/receiver-loss +minion scenarios, W9 deployment campaigns and W10 exercised operations remain +unfinished. Full W1–W10 remains active. + +## October 3 2026 original sealed suffix checkpoint + +`VerifiedRecoveryPrefix` binds one exact original `PinnedRecoveryCell` to the +retained closed owner and canonical acquisition metadata, then verifies native +materialization lineage and the complete current successor origin graph. The +selected Serving owner, epoch, incarnation and root are rechecked after that walk. +The original manifest epoch and every sealed boundary remain unchanged. An +interrupted ownership claim can materialize at a later epoch; missing earlier +acquisition metadata cannot replace or alter the original suffix. + +The runtime wrapper reuses existing shared transient-memory admission and LTX +I/O facilities. The caller owns a finite deadline; no detached work, authority, +publication or scheduler is added. Missing original owner/acquisition metadata, +wrong original scope, corrupt records and unavailable current origin data refuse. +Direct authority callers own equivalent admission and authenticated mappings. + +Host recovered serving compares the entire journal input/result with canonical +acquisition metadata. A pinned overlay requires a read-only lookup through the +existing recovery provider, the original digest-verified manifest row and this +exact suffix proof. An 8-MiB transient token covers acquisition/manifest metadata +before I/O and stays charged through verification. The host then repeats the +ordinary native actor, selected authority and inventory checks. Historical +result replay remains historical and supplies no refreshed serving observation. + +Two new authority refusal tests preserve original read errors and reject matching +endpoints with different recovery inputs. A genuine public interrupted-claim case +leaves no acquisition record at its first claimed epoch, materializes through the +ordinary later takeover, advances the root and verifies the original suffix. +Normal and interrupted cases mutate all 16 original scope/boundary fields, remove +and corrupt canonical acquisition metadata, remove current origin bytes and check +memory refusal and token cleanup. Two host cases require canonical acquisition +metadata despite a live actor. Two further host cases build real captured SQLite +frames, fsync them to a follower, canonically seal and materialize the tail, advance +the successor and require the original manifest on fresh serving inspection. +Missing/corrupt evidence refuses; restoring exact original bytes permits fresh +adoption. Complete node shutdown asserts zero retained resources. + +| Verification | Result | +| --- | --- | +| Full workspace, all features and locked dependencies | 1,616 passed; no failures; 36 documented cases ignored. Nested crash-fixture child summaries excluded. | +| Complete canonical minion target, all features | 218 passed; no failures or ignored cases. | +| Distinct final passes | 1,834; focused repeats and overlapping local LTX excluded. | +| Local LTX without default features | 54 passed; no failures or ignored cases. | +| Workspace check, all targets/features | Passed. | +| Workspace Clippy and API documentation | Passed with warnings denied. | +| Format, boundaries, layout, Rust fences, links, SQL/peer and diff | Passed; 124 Rust snippets parse. | +| Frozen Rust/Cargo source, including nested qualification locks | 750 paths; SHA256 `fa5b6cdb82e924c3053c7dbb64490f02514761ea0b59fba8973f3bd71e38d67c`. | +| Isolated checkout / mounted Workspace target suffix | `/tmp/cellule-reader-prelease-8698183` / `cellule-reader-prelease-8698183`. | +| Published parent | `1898f05db4d6f29aabb177b38510f75bdf41b4f1`. | + +Raw commands, frozen source, final logs, downloaded routing evidence and initial +diagnostics are in `/tmp/cellule-recovered-prefix-evidence`. The first negative +fixture selected Recovering rather than the required Serving state; the initial +interrupted-claim advertisement exceeded the existing lifetime limit. Both +fixtures were corrected without weakening production contracts. Full suites and +gates ran on the final frozen Rust/Cargo source. Initial runs do not qualify it. + +### CI blocker and remaining work streams + +Published parent `1898f05db4d6f29aabb177b38510f75bdf41b4f1` is mergeable. +Workspace, MSRV, follower/object capacity, contract, website, fuzz, fast models +and Compose pass. Both routing performance jobs and their aggregate fail. +Their command lanes fail the unchanged 90% throughput gate. The downloaded +leased artifact records 0.807–0.888 candidate/baseline throughput and approximately +one additional origin read for each of 1,024 commands. The new lineage preflight +read is a concrete investigation lead; causality and a qualified fix remain open. +Passing correctness tests cannot close this performance gate. The earlier +unchanged reader-balance failure also remains causally unexplained. Profiles, +workloads, thresholds and expected evidence remain unchanged. + +| Priority | Stream | Required delivery | +| --- | --- | --- | +| 0 | Routing CI regression | Reproduce the measured command regression, distinguish lineage metadata I/O from other costs, fix the cause while retaining exact durable prefix proof, and pass both unchanged routing profiles. | +| 1 | Complete observation and writer proof | Aggregate every original writer and sealed suffix with authenticated complete backend, physical boot/process and operation scope. Consume current native successor serving, reader/follower replacement policy and accepted-work observations through the existing reconciler. | +| 2 | Maintenance completion | Implement SettleRoles/Finalize, join original accepted actions before terminal drain handoff and confirm checked stop/withdrawal/boot retirement. Finish Cron/Blob external owners and primitive fault matrices. | +| 3 | Movement convergence | Finish receiver-session replacement, recovery/adoption, refusal/unknown and source/receiver-failure reconciliation; complete sustained pressure/count and concurrency evidence. | +| 4 | Canonical minion scenarios | Deliver complete runnable maintenance and receiver-loss scenarios at `crates/cellule-host/minion`; Cargo target remains `fleet_operations`. | +| 5 | W9 qualification | Complete process/provider fault, mixed-binary, load/soak and capacity campaigns in their documented environments. | +| 6 | W10 operations | Exercise operator runbooks and collect staged rollout evidence. | + +This checkpoint supplies one exact original sealed-suffix observation, including +interrupted claims. It does not aggregate all original writers/suffixes, prove +complete role replacement, pin storage or finish maintenance. SettleRoles and +Finalize remain refused. The full W1–W10 goal remains active. + +## October 3 2026 exact root prefix and current origin checkpoint + +The canonical publisher and recovered-overlay materialization retain native +`PreparedRoot` links before their root authority CAS. `CellRootLineage` uses the +additional version 1 Cell/incarnation/digest path with an 8-KiB checksummed record. +Up to 64 distinct verified inputs accumulate through ETag CAS; delayed writers +cannot erase links. This supports byte-identical roots derived from different +inputs, including representation-only compaction. Failed authority publication +can leave verified proposal metadata; it grants no ownership or acknowledgement. +Existing control/LTX formats, paths and signed peer messages are unchanged. + +`VerifiedRootPrefix` reaches the exact original digest, scope, TXID, checksum and +sequence through native verified preparation links. Higher counters cannot prove +that prefix. The canonical LTX origin inventory then authenticates every current +successor dependency, including body/index bytes and complete directory coverage. +Missing legacy/manual-publication metadata, unrelated prefixes, corrupt metadata +and unavailable origin data refuse. Old root documents may disappear after +compaction while retained links still prove derivation and the current graph +remains fully available. Metadata retention is not a root pin. + +The host's serving checks consume the exact released or materialized recovery +root, then repeat the native actor query, selected-root/owner authority check and +native inventory check after origin verification. Fresh inspection uses these +checks; historical action result replay is unchanged. The runtime wrapper charges +transient metadata through its existing shared retained-byte ledger before I/O. +Its conservative envelope covers fixed graph/cache/fetch/decode work, bounded +lineage maps and at most 10,000 origin objects. Accepted fleet work stays under +its existing finite action owner; no scheduler or task lane is added. Direct +callers own equivalent admission and deadlines, and application Store adapters +supply bounded stream chunks. + +Seven new lineage unit cases exercise the maximum codec, every truncation and +byte corruption, malformed scope/position/order, additive replay/reconstruction, +competing inputs, original write failure, lost replies, native publication +ordering and corrupt/foreign metadata. A new peer case preserves the existing +Unavailable/NotStarted error contract. Four new public runtime cases exercise +real acknowledged writes through movement/compaction, missing old documents, +missing/corrupt current origin bodies, legacy/unrelated/scope/limit refusal and +shared memory admission/release on cancellation, success, error and shutdown. +Two new public host cases retain a live successor actor while deleting its +lineage or current root, require fresh inspection refusal, restore the exact +original bytes and verify fresh adoption. Existing native quiet/foreground +compaction, migration and genuinely recovered pinned-tail cases also check exact +prefixes; the LTX inventory case checks exact and insufficient count limits. + +| Verification | Result | +| --- | --- | +| Full workspace, all features and locked dependencies | 1,609 passed; no failures; 36 documented cases ignored. Nested crash-fixture child summaries excluded. | +| Complete canonical minion target, all features | 218 passed; no failures or ignored cases. | +| Distinct final passes | 1,827; focused repeats and overlapping local LTX excluded. | +| Local LTX without default features | 54 passed; no failures or ignored cases. | +| Workspace check, all targets/features | Passed. | +| Workspace Clippy and API documentation | Passed with warnings denied. | +| Format, boundaries, layout, Rust fences, links, SQL/peer and diff | Passed; 123 Rust snippets parse. | +| Frozen Rust/Cargo source, including nested qualification locks | 747 paths; SHA256 `d6a8e49f73a539a7f59ad2e116c265ed03b48e921e4ab1282a4aa2cca4fcf3b0`. | +| Isolated checkout / mounted Workspace target suffix | `/tmp/cellule-reader-prelease-8698183` / `cellule-reader-prelease-8698183`. | +| Published parent | `6a9443f097c69143f984c0e109b0454f3f3a4c84` | + +Raw commands, frozen source archives, final logs and initial diagnostics are +retained in `/tmp/cellule-root-prefix-evidence`. Initial compilation caught a +missing export/import and test API errors. The memory-admission test needed a +`Duration` import. The first Clippy run rejected the new two-root error's large +payload; its exact details are now boxed. Full suites and gates were rerun on +the final frozen source. Initial runs do not qualify it. + +Published parent `6a9443f` is mergeable and passes every enabled CI check, +including follower/object capacity, both routing modes and their aggregate, +Compose, workspace, MSRV, contract, website, fuzz and fast models. The earlier +unchanged reader-balance failure remains causally unexplained; passing repeats +establish no fix. Profiles, workloads, expected evidence and thresholds remain +unchanged. Fresh CI must qualify this new checkpoint. + +### Remaining work streams + +| Priority | Stream | Required delivery | +| --- | --- | --- | +| 1 | Complete observation and writer proof | Bind every original writer and sealed recovery suffix to its authenticated canonical backend, exact successor prefix/current native serving and the full revisioned fleet barrier. Consume complete writer, reader/follower replacement-policy and accepted-work observations in the existing reconciler. | +| 2 | Role evacuation and maintenance completion | Implement SettleRoles/Finalize with all original role proofs, join original accepted actions before terminal drain handoff, and confirm checked stop/withdrawal/boot retirement. Finish Cron/Blob external-owner integration and primitive fault matrices. | +| 3 | Movement resilience and convergence | Finish recovery/adoption across receiver sessions, refusal/unknown and source/receiver-failure reconciliation, ongoing intent supervision, sustained pressure/count convergence and concurrency evidence. | +| 4 | Canonical minion scenarios | Deliver complete runnable maintenance and receiver-loss scenarios through `crates/cellule-host/minion`; Cargo target remains `fleet_operations`. | +| 5 | W9 qualification | Run complete process/provider fault, mixed-binary compatibility, load/soak and capacity campaigns with their documented environments. | +| 6 | W10 operations | Exercise operator runbooks and collect staged rollout evidence. | + +This checkpoint proves one per-movement prefix and availability observation. It +does not aggregate every original physical-boot writer or recovery suffix, prove +reader/follower replacement policy, pin storage, or complete fleet maintenance. +SettleRoles and Finalize remain refused. The full W1–W10 goal remains active. + +## October 3 2026 canonical acquisition metadata checkpoint + +The ordinary Idle acquisition, published takeover and rootless takeover now +retain the exact successful ownership-CAS input and claimed/materialized Control +before actor admission. The immutable `CellAcquisitionRecord` preserves pinned +recovery overlays and their canonical materialized positions across later +publication, release and compaction. Strict creation adopts only identical +committed data after a lost reply; competing different data refuses. The bounded +origin reader validates the typed Cell/incarnation/epoch path, version, canonical +Controls and their ordinary Takeover/PublishRecovery transitions. + +This is acquisition metadata, not complete successor-prefix proof. It can +survive failed activation; it establishes no current serving, root-retention pin, +exact dependency availability or maintenance settlement. Initial bootstrap, +direct activation of already-claimed Controls, older binaries and cancellation +before retention can leave no record. Absence must block any proof requiring +that record. Published acquisition retains ordinary rollback; rootless failure +leaves its claimed Recovering Control available to the ordinary recovery path. +Existing control/LTX formats, paths and peer messages remain unchanged; the new +version 1 acquisition path is an additional persistence contract. + +Eight unit cases cover canonical round trips, all truncations, malformed scope, +root/owner/epoch/code drift, immutable replay and conflicts, competing writers, +lost replies, source-read failure, cancellation before publication and corrupt or +oversized metadata. Two new public Idle cases exercise actual metadata failure, +rollback, actor non-admission, exact-root SQL after a lost reply and receiver +resource release using a dedicated disk budget. Existing genuine recovered-tail, +rootless takeover and prepared-receiver cases assert exact metadata; accepted +receiver work retains it after a dropped waiter and lost CAS/metadata replies. +The storage guide documents ordering, bounds, reconstruction and proof limits. + +| Verification | Result | +| --- | --- | +| Full workspace, all features and locked dependencies | 1,595 passed; no failures; 36 documented cases ignored. Nested crash-fixture child summaries excluded. | +| Complete minion target, all features | 218 passed; no failures or ignored cases. | +| Distinct final passes | 1,813; focused repeats and the overlapping local LTX run excluded. | +| Local LTX, no default features | 54 passed; no failures or ignored cases. | +| Workspace check, all targets/features | Passed. | +| Workspace Clippy and API documentation | Passed with warnings denied. | +| Format, boundaries, layout, Rust fences, links, SQL/peer and diff | Passed; 122 Rust snippets parse. | +| Frozen Rust/Cargo source, including nested qualification locks | 740 paths; SHA256 `9aa13b336df94e30e4211720f30da368897dcbbf23507b20ce4a5937974d8475`. | +| Isolated checkout / target suffix | `/tmp/cellule-reader-prelease-8698183` / `cellule-reader-prelease-8698183` beneath the mounted Workspace target. | +| Parent revision | `e532964006d56a510f2663628356e9283f4d991a` | + +Raw commands, initial diagnostics and logs are retained in +`/tmp/cellule-acquisition-history-evidence`. +The initial compile failed on two test-only API names; a later corrupt-object +fixture incorrectly used immutable Store publication to overwrite its original +object. The first public disk assertion used the process-wide default budget and +observed unrelated tests' charges. It now uses a dedicated receiver budget while +preserving all zero-resource assertions. Those initial runs are not passing +evidence. + +Published parent `e532964006d56a510f2663628356e9283f4d991a` is mergeable; all +enabled CI checks pass, including follower/object capacity, both routing modes +and their aggregate, Compose, workspace, MSRV, contract, website, fuzz and fast +models. The earlier unchanged reader-balance failure remains causally unexplained; +passing repeats establish no fix. Minion remains the canonical executable source +at `crates/cellule-host/minion`, with Cargo target `fleet_operations`. + +### Highest remaining priorities + +Implement complete successor lineage, every original acknowledged prefix and +exact dependency availability, then fresh native serving. Higher sequence values +and historical acquisition metadata cannot substitute for that proof after +compaction. Consume complete writer, reader/follower replacement-policy and +accepted-work observations in the existing controller before `SettleRoles` or +`Finalize`. Join original action work before terminal drain handoff; finish +cross-session receiver recovery/adoption, Cron/Blob external owners, maintenance +and receiver-loss minion scenarios, W9 process/provider/mixed-binary/load gates and +W10 exercised runbooks/rollout. The full W1–W10 goal remains unfinished. + +## October 3 2026 complete original writer inventory checkpoint + +`FleetOriginalWriterCapture` joins the original failed boot through its existing +process provider, traverses every authenticated application/tenant catalog and +canonical owner history, and retains each full original Control and target. +Rootless, recovering and fully object-covered originals remain in the set even +when a later canonical takeover has removed the original owner. Complete catalog +entries with no initial Control and ownership epochs outside the original boot +remain explicit in the per-scope witness. Missing legacy/restored owner history +is a typed blocker. + +`FleetOriginalCatalogs` is an application authentication boundary. Its stable +witness attests the original complete configuration and actual canonical backend +mappings; source construction validates shape only. Collection rereads that set, +revalidates all original catalog heads/ETags, joins the original process again and +rechecks the full boot/intent/registry/head/claimant barrier. At most 128 scopes, +10,000 total catalog entries and 10,000 inspected ownership epochs are permitted; +excess refuses without truncation. Original observations pack in canonical +64-row pages with the ordinary one-MiB envelope and a 64-KiB manifest. + +`FleetOriginalWriterJournal` uses minion's existing accepted SQLite work owner. +First publication checks the full fresh barrier and commits all pages, the one +immutable operation/process pointer and one registry advance in the same +transaction. Exact replay returns original bytes and capture times, without +provider recollection or another registry advance. Different original inputs +conflict. Independent clients serialize through the same SQLite file; lost replies +and canceled waiters preserve the accepted durable commit. + +The [integration recipe](../crates/cellule-host/docs/original-writers.md) records +ordering, bounds, source authentication and reconstruction. New codec kinds 24–26 +use the existing version/domain; kinds 1–23 and all existing IDs, control formats, +object paths and signed messages are unchanged. Minion remains canonical, with +Cargo target `fleet_operations`. + +Seven runtime cases cover full-Control round trips, all truncations and envelopes, +explicit authenticated empty sets, duplicate/reordered/missing/foreign pages, +scope/count/time/budget failures, root/code/schema binding and repeated original +epoch ordering across page boundaries. Eight minion cases retain 66 actual original +owner Controls across two canonical application/tenant catalogs and 68 entries, +with real sole-authority takeovers. They cover atomic replay/reconstruction, +provider causes, missing history, stale barriers, catalog/process changes, +independent competing clients, dropped waiters and lost commit replies. +Their child process is a lifetime stand-in, supplying no OS-crashed CellNode, +accepted external-job, successor-prefix or dependency-availability qualification. + +| Verification | Result | +| --- | --- | +| Full workspace, all features and locked dependencies | 1,585 passed; no failures; 36 documented provider/process/performance/manual or documentation cases ignored. | +| Complete minion target, all features | 218 passed; no failures or ignored cases. | +| Distinct final passes | 1,803; focused repetitions and the nested LTX child-process result excluded. | +| Workspace Clippy, all targets/features | Passed with warnings denied. | +| Workspace API documentation | Passed with warnings denied. | +| Format, boundaries, layout, Rust fences, links, SQL/peer and diff | Passed. | +| Frozen Rust/Cargo source, including nested qualification locks | 737 paths; SHA256 `ef4ac129a38eac68f5b1244a01d1cf46b3617b9d1c72b37dfc55f248d8ba7eb8`. | +| Isolated checkout / target suffix | `/tmp/cellule-reader-prelease-8698183` / `cellule-reader-prelease-8698183` beneath the mounted Workspace target. | +| Raw commands, logs, initial diagnostics and manifests | `/tmp/cellule-original-writer-evidence` | +| Parent revision | `b0552ee3b60b8cd453844876d84a64eb535e63f2` | + +The original report counted one nested LTX crash-fixture child result twice; +the corrected workspace and distinct counts above exclude that child summary. + +Initial checks found a missing Digest import and typed IDs used as ordered keys; +byte-array keys preserve the existing ID contract. Test compilation then exposed a +missing test-only snapshot import. An initial stale-barrier test requested an +already-stopped scheduling state; it now changes the state and asserts the actual +registry advance before requiring refusal. All initial diagnostics remain +retained and provide no passing evidence. + +### Parent CI and reader balance investigation + +Parent `b0552ee` is mergeable. Both complete write-capacity modes, both routing +modes and their aggregate, Compose smoke, MSRV, contract, website, fuzz and fast +model checks passed. Rust workspace run `37112788891` failed in the unchanged +balanced three-process smoke at `process_replica.rs:193`: the twelve successful +replica reads were not evenly split within its unchanged one-read tolerance. +The failure log has no actual per-reader counts before that assertion. + +An isolated container used the exact CI RustFS image digest and disposable +fixture credentials, a fresh bucket and unique object prefixes. The exact +process test passed once, followed by five same-binary repeats. These local +passes do not explain the CI failure and establish no fix. Ten further same-binary repeats during workspace verification also passed. These +contention reproduction logs and the original binary SHA256 remain with the evidence. +Qualification profiles, workload counts and acceptance thresholds are unchanged. +Fresh CI is required after publication; parent results do not certify new source. + +### Highest remaining priorities + +Verify every retained original acknowledged prefix and exact root dependencies, +then verify current successor authority and actor serving. Historical metadata +provides no root-retention pin or successor proof; sequential catalog heads are +not a global transaction. Integrate complete writer, reader/follower policy and +native accepted-work observations through the existing observer/controller. +`SettleRoles` and `Finalize` still refuse without their actual evidence. Join +original action work before terminal drain handoff, finish failed-receiver and +cross-session recovery/adoption, Cron/Blob external owners, runnable maintenance +and receiver-loss scenarios, and W9–W10 process/provider/mixed-binary/load +qualification, exercised runbooks and rollout. The full W1–W10 goal remains open. + +## October 3 2026 complete catalog traversal checkpoint + +`CellCatalog::scan_all(limit)` now captures all 256 original heads before reading +pages and streams through the ordinary verified shard reader. Opaque +`CatalogScanReceipt` requires observed end-of-stream, a successful cumulative row +bound and final checks of every populated or absent head, including revision, +locators and ETag. A read/verification failure prevents completion and preserves +its original error. Later `revalidate()` uses the same original scoped adapter +without replacing the captured set. The receipt exposes tenant/application, +entry count and each original revision/page-digest list. This path starts no task, +storage listing, authority mutation or second scheduler. + +Eight new public cases cover empty heads, multi-page/multi-shard iteration, +independent adapters, tenant isolation, partial consumption, bounds before I/O, +cumulative failure, old immutable pages after provisioning, populated/absent +head changes, deletion, same-body ETag changes, missing pages, corrupt bytes and +later receipt invalidation. Catalog codecs, authority/history and routing +qualification bytes are unchanged. Minion remains canonical. + +| Qualification | Result | +| --- | --- | +| Focused public catalog tests | 19 passed; no failures or ignored cases, including eight new complete-scan cases. | +| Full runtime public integration suite, all features and locked | 209 passed; no failures; four documented provider cases ignored. The 19 catalog passes are included. | +| Runtime Clippy, all targets/features | Passed with warnings denied. | +| Runtime API documentation | Passed with warnings denied. | +| Format, boundaries, layout, Rust fences, links, SQL/peer, diff | Passed. | +| Frozen Rust/Cargo source | 729 paths; SHA256 `b585bf9e802c1f843ab46a8c3419a9ef08811683a0694b8307da83703c1dcebb`. | +| Isolated checkout | `/tmp/cellule-reader-prelease-8698183` | +| Evidence | `/tmp/cellule-catalog-scan-evidence` | +| Parent revision | `8659f49fde2f46a2a5cf1121fa8f114ba0d92fae` | + +Raw commands, logs, manifests and initial invocation diagnostics are retained. +An initial preparation invocation used the snapshot as its source, failed with +SameFileError and launched an old-source invocation. That invocation was deliberately +terminated with exit 143 and supplies no test evidence. The corrected frozen build +completed; process inspection during its slow mounted-target I/O is also retained. +These 209 distinct passes are not full workspace/minion, process/provider, +mixed-binary or load qualification for this new source. + +Parent PR head `8659f49` is mergeable and every required CI check passed, including +both capacity modes, both routing modes, Compose smoke, Rust/MSRV, contract, +website, fuzz and fast model checks. The leased routing artifact's actual virtual +merge candidate is `522b7b2ffd835036319d81251e9aac5e3eef8811`, against baseline +`18a1244a98c50a04c52da21325f95331055e9884`. Independent replay verified frozen binary +and harness SHA256, all raw samples/counts/recovery proof and the exact 120-row +comparison. There are no gate failures; the prior failing uncached lane's median +throughput ratio is 1.108014. + +Identical-binary calibration run `37110515841` passed both complete modes using +head `8659f49` as baseline and candidate. Independent replay verified identical +frozen binary hashes, all raw evidence and both exact comparisons. Leased uncached +throughput ratio is 0.962518; object-only is 0.974094. Leased uncached p99 generated +its retained 10% review alert (1.117907), below the unchanged 2.0 blocking limit. +The earlier parent's 89.669% throughput failure remains reproducible from its raw +measurements and causally unexplained. A passing repeat or four-pair calibration +cannot explain that earlier failure. Profiles and thresholds were not changed. +Parent CI does not qualify this new source; fresh CI is required after publication. + +Highest priority next is authenticated complete application/tenant enumeration +and durable operation-bound aggregation of every original affected Cell, with +original process/accepted-work joining, followed by fresh successor prefix/serving +verification. Sequential catalog heads are not a global transaction, durable +object pin or complete physical-node proof. Role/controller integration, failed +receivers, original action joining before drain handoff, maintenance commands and +W9–W10 qualification/runbooks/rollout remain unfinished. SettleRoles/Finalize remain +blocked on their actual required evidence; the full W1–W10 goal stays active. + +## October 3 2026 original owner history before authority departure checkpoint + +`CellAuthority::transition` now durably retains the full original owner control +before the ordinary release, takeover or tombstone CAS can remove or replace it. +This covers unpublished, recovering and fully object-covered controls, including +their original incarnation, epoch, code/schema, root and recovery overlay. It +starts no second authority, scheduler or task bank. Same-owner publication and +renewal add no history write. Storage work remains in the original caller's +accepted lifetime; unresolved history publication prevents owner departure. + +The new typed layout path is +`cells/v1/apps//cells//owner-history/v1//.json`. +Its body uses the existing canonical control codec and 8 KiB bound. Competing +departure proposals advance history with ETag CAS. Delayed older proposals cannot +replace later observations. Equal or later original observations can reconcile +lost history replies; otherwise the original write error returns to the existing +coordinator for full predicate revalidation. A retained proposal proves no +departure CAS. The control codec, transition table, original authority implementation +apart from this retention call, and all 26 existing layout methods are byte-identical +to the parent. Existing persisted paths and signed message domains are unchanged. + +`owner_observation` reads one exact retained epoch. Opaque `CellOwnerHistory` +collects every closed epoch in the current incarnation plus its current owner, +with caller row bounds and an exact authority recheck. Ordinary release closes +the same ownership epoch; tombstone adds a fence epoch without inventing an owner. +Missing legacy, restored or older-binary history returns the typed +`OwnerHistoryIncomplete` with Cell/incarnation/epoch. Foreign, malformed, +noncanonical, oversized, conflicting or changing observations fail closed. +Metadata grants no ownership, process joining, complete catalog scope, immutable +root pin, successor serving or maintenance completion. Immutable collection does +not delete the history metadata and does not pin its historical root graphs. + +Thirteen new authority cases cover original rootless/recovering succession, +object-covered release/acquisition/tombstone, idle tombstones, same-owner work, +source errors, lost history and authority replies, delayed and cancelled proposals, +missing history, bounded/canonical/scope conflicts and authority progress during +collection. A new peer case rejects an empty/successful interpretation of missing +history. Existing public recovered-tail and unpublished-takeover cases now reopen +full original history through independent authority adapters after actual successor +serving. The typed path test covers the exact new version 1 layout. + +| Final qualification | Result | +| --- | --- | +| Full workspace tests, all features and locked dependencies | 1,571 passed; no failures; 36 documented provider/process/performance/manual or documentation cases ignored. | +| Complete minion test target, all features | 210 passed; no failures or ignored cases. | +| Distinct final passes | 1,781; earlier focused/selected repetitions excluded. | +| Workspace Clippy, all targets/features | Passed with warnings denied. | +| Workspace API documentation | Passed with warnings denied. | +| Workspace all-target/all-feature compilation | Passed before the final test-only borrowed-slice cleanup; final Clippy compiled those targets again. | +| Format, boundaries, layout, Rust fences, links, SQL/peer, diff | Passed. | +| Frozen Rust/Cargo source | 728 paths; SHA256 `ee6a914c66822ee253f39b58493381070cc7af9fe167d5b4a8008c3228288e39`. | +| Isolated checkout | `/tmp/cellule-reader-prelease-8698183` | +| Evidence and source archive | `/tmp/cellule-owner-history-evidence` | +| Parent revision | `de010246a8f32c368dfe1628b3ec940471b623dc` | + +Raw commands, logs, build environment, source manifests and archives are retained. +Initial private-helper, exhaustive-error mapping and test-import compile diagnostics +are recorded. The selected pre-Clippy suites passed 1,053 runtime/host/executable +tests and two path cases; Clippy then rejected four unnecessary test clones. The +final assertions use borrowed slices with identical expectations. Production +validators, qualification profiles and thresholds were not weakened. Final full +workspace qualification uses the isolated snapshot. Ignored cases supply no evidence +for provider, full process, mixed-binary or fleet-load completion. + +Parent CI passed follower/object capacity, workspace/MSRV, contract, website, +fuzz, Compose smoke and fast model checks. Its leased routing comparison failed +`forwarded_query_uncached_route/c16`: median throughput was 89.669% of baseline, +below the declared 90% minimum; read/hop counts and correctness matched. The exact +virtual merge candidate was `714b6c97ba121baaaf5c8ad3cb45e7abaa1bff2c`, compared with +`18a1244a98c50a04c52da21325f95331055e9884`. Raw logs and the frozen comparison artifact +are retained under this evidence directory. Its cause remains unproven; a later +passing run cannot explain this failure. Parent CI does not qualify this new source. + +Highest priorities next are diagnosing that measured routing regression, then +authenticated complete catalog traversal and durable operation-bound aggregation +of original owner history with original boot/process joining. Fresh successor +prefix/serving verification remains required. Per-Cell history does not establish +the original physical-node set. Receiver failure adoption, complete role/controller +integration, original action joining before drain handoff, remaining primitive +faults, complete maintenance commands and W9–W10 qualification/runbooks/rollout +remain unfinished. SettleRoles/Finalize stay blocked on their required evidence; +the full W1–W10 plan remains active. Minion remains canonical. + +## October 3 2026 original process confirmation before recovery checkpoint + +`NodeDirectory::fenced_session` reads the immutable original physical boot fence +while its log still needs recovery. `closed_session` continues to require a +canonically Retired log. Neither observation proves process termination or Cell +relocation. This distinction permits original process and accepted-work joining +before retaining affected Cells and starting dependent recovery effects. + +`FleetFailedBootProcessRequest::capture_fenced` binds the original Established +boot and permanent fence in a version 2 request identity. Recovery phase, +manifest, claim adoption and observation times cannot change that identity. +`confirm` uses the existing application process provider twice, checking the +complete current roster, original boot, permanent fence and live claimant around +the reads. It preserves provider errors and rejects changed evidence, stale +barriers and nonmonotonic or excessive intervals. The provider owns authentication, +durable joining of the original process and all accepted native/external work, +and session nonreuse; a PID observation alone is insufficient. + +`FleetFailedBootRetirement::capture_retained` reuses that original request after +related roles close and the native log becomes Retired. Its fresh terminal +capture and publication rechecks do not restamp the original request interval. +The existing terminal version 1 request and retirement event encoders remain +byte-identical. Version 1 and version 2 requests are distinct: restart must retain +the original basis and witness, and cannot silently convert a published event. +The public request `canonical()` getter now returns an optional original terminal +observation; `fence()` is always available. Local callers and documentation use +the updated API. This is a source API change, not a persisted format rewrite. + +Seven new executable cases cover actual child joining before canonical recovery, +stable identity through sealing and retirement, independent adapter reconstruction, +lost committed replies, a still-running original child, original provider errors, +changed witnesses, suspended-provider registry races, log progress, foreign request +evidence and regressing clocks. One new runtime case distinguishes early fencing +from strict closure through claim adoption and recovery progress. The child is a +process-lifetime stand-in in a cold follower ensemble: these cases do not qualify +an OS-crashed CellNode, an original Cell tail or external job joining. + +| Final selected qualification | Result | +| --- | --- | +| Runtime library | 545 passed; three existing provider cases ignored. | +| Runtime lifecycle | 144 passed; four existing provider cases ignored. | +| Host library / public node / minion | 34 / 106 / 210 passed. | +| Distinct selected tests | 1,039 passed; no failures; focused repeats excluded. | +| Runtime/host Clippy, all targets/features | Passed with warnings denied. | +| Runtime/host API documentation | Passed with warnings denied. | +| Format, boundaries, layout, Rust fences, links, SQL/peer, diff | Passed. | +| Frozen Rust/Cargo source | 726 paths; SHA256 `dbd6094dbcec4a15eed39a667c7214c7dac52d751b71dd49e1875041464c5f31`. | +| Isolated checkout | `/tmp/cellule-reader-prelease-8698183` | +| Evidence and source archive | `/tmp/cellule-fenced-process-evidence` | +| Parent revision | `8e4976e4893edde7658cccb208305f033712e340` | + +Commands, build environment, raw logs, complete source manifests and the source +archive are retained. The first focused test run had 14 passes and one incorrect +fixture error-class assertion; the complete-roster barrier correctly returned its +original conflict error. The final assertion checks that exact error. Initial +syntax and compatibility-audit script diagnostics are also retained. Production +validators and qualification requirements were not weakened. Seven ignored cases +require their documented isolated provider environment. These selected suites do +not establish full workspace, provider, mixed-binary or measured load qualification. + +Highest priority next is operation-bound retention of every original affected Cell +before recovery/relocation effects, including object-covered, unpublished and +transitional writers, followed by fresh current successor prefix and serving +verification. A scan after takeover cannot reconstruct the original ownership set. +Process confirmation supplies no complete Cell inventory. SettleRoles/Finalize +remain blocked on that evidence and complete authenticated observation. Receiver +failure adoption, reader/follower controller integration, original action joining +before drain handoff, remaining primitive faults, complete maintenance commands +and W9–W10 qualification/runbooks/rollout remain unfinished. The full W1–W10 plan +stays active. `crates/cellule-host/minion` remains the user-approved canonical +executable location; the Cargo target stays `fleet_operations`. + +## October 2 2026 complete recovered-suffix metadata checkpoint + +`RecoveryManifestStore::load_manifest` exposes every original recovered scope in +one digest-verified immutable manifest, including other applications. Its bounded +read shares the overlay loader's canonical decoder, digest and original leader/epoch +checks. Publication and both readers share one conversion to control-ready rows. +The decoder now rejects duplicate or reordered full scopes; existing producers +already sort and reject duplicate scopes. Version 1 bytes, object paths and signing +domains are unchanged. Original storage errors remain available to the caller. + +The opaque `RecoveryManifestInventory` retains original application, Cell, +incarnation, epoch, predecessor and final prefix references after successors clear +their overlay pointers. Reading it starts no recovery, authority mutation, native +admission or task owner. It verifies metadata, not bundle availability or current +successor state. Object-covered Cells without a recovered suffix are absent; an +empty suffix has no manifest. The full original writer set must still be retained +before dependent effects, and every successor must be verified separately. This +checkpoint cannot settle roles or finalize a failed node. + +Seven new unit cases cover independent adapter reconstruction, three genuine +recovered scopes across two applications, ordinary bundle reopening, unavailable +bundles, original storage errors, invalid scope/digest/path, duplicate/reordered +rows, noncanonical or malformed metadata and bounded reads. The public takeover +case reads the canonical sealed identity before materialization and reopens the +same original metadata after actual successor serving. Existing lost-observation, +object-covered and unpublished-owner cases remain intact. The follower guide now +shows the actual version 1 persisted schema and the public read recipe. + +| Final selected qualification | Result | +| --- | --- | +| All-feature manifest library cases | 14 passed; no failures or ignored cases. | +| All-feature public recovery cases | Seven passed; no failures; one existing RustFS case ignored without its documented isolated environment. | +| Runtime/host Clippy, all targets/features | Passed with warnings denied. | +| Runtime/host API documentation | Passed with warnings denied. | +| Format, boundaries, layout, Rust fences, links, SQL/peer, diff | Passed. | +| Frozen Rust/Cargo source | 722 paths; SHA256 `f4a4aae3d86025bdbb5321c92bf5dbfc82593513c668efb8cba371952a818a85`. | +| Isolated checkout | `/tmp/cellule-reader-prelease-8698183` | +| Evidence and source archive | `/tmp/cellule-recovery-inventory-evidence` | +| Parent revision | `2a155238311fa1bfc61ffb07f136dbb9a49b6af1` | + +These 21 selected passes do not represent a full workspace rerun or provider, +process, external-job, mixed-binary or load qualification. Commands, raw logs, +source manifests and build environment are retained. The initial static command +attempted Git's diff check in the non-Git snapshot; that invocation failed and the +proper active-worktree diff check passed. No Rust failure or weakened expectation +was involved. Parent CI has passed both follower/object capacity, workspace, MSRV, +contract, website, fuzz and the fast model checks; Compose was still running when +inspected. Those parent results do not qualify this new source. + +Highest priority remains operation-bound retention of every original affected Cell +before recovery/relocation effects, including fully object-covered and transitional +writers, followed by fresh current successor/prefix/serving and failed-process +proof. Complete authenticated production observation, receiver-session failure +adoption, reader/follower controller actions and original action join before drain +handoff remain required. SettleRoles/Finalize, remaining primitive faults, complete +maintenance/receiver-loss commands and W9–W10 qualification/runbooks/rollout remain +unfinished. `crates/cellule-host/minion` is the user-approved canonical executable +location; the Cargo target stays `fleet_operations`. The complete plan stays active. + +## October 2 2026 durable live-owner follower replacement checkpoint + +`FollowerEvacuationRecord` now retains the complete original Retired ensemble, +the original Established donor digest, object-covered retirement watermark, +signed source boot, full capture head/registry, maintenance operation and interval. +Every replacement binds its exact Established request and signed boot identity. +The bounded codecs support the canonical one/two-member ensembles; existing +record kinds and signing domains remain unchanged. Shape validation supplies no +native retirement or authentication rights. + +`FleetFollowerEvacuationJournal` stores revisioned application redundancy policy, +immutable captures and the latest operation/request pointer in the existing +registry transaction domain. Policy absence blocks publication. First publication +checks the full barrier, live controller, current operation/intent, policy and +complete original/replacement rows before advancing the registry. Exact replay +returns history without restamping, advancing the registry or restoring an older +pointer. The SQLite adapter retains accepted work in its original finite backend +owner through cancelled waiters and lost replies. + +`FleetFollowerEvacuationVerifier` reloads current policy and complete roster, +checks the current canonical ensemble and pinned signed boots, and collects +complete source/receiver inventories through `FleetSnapshotTransport`. Requests +use the existing finite native snapshot owner, including its 32-epoch enrollment +page limit. All categories are rechecked after authority discovery. A native +source fence cannot hide behind a still-live directory record. Installed epochs +before their first append need the actual source binding and delivered producer; +inactive directory enrollment alone supplies no installation proof. + +Refresh during Evacuating or Closing retains the original retirement, including +after later rotations evict its original local receipt. It observes the current +ensemble and policy without starting another rotation, retirement or recovery. +Publication and fresh confirmation preserve original errors independently; +capture clocks are monotonic and bounded by thirty seconds and caller/operation +deadlines. Copied collector buffers remain the application's accounting duty. +Failed original owners require canonical recovery and affected-Cell successor +evidence; this live-owner record cannot settle every role or finalize a node. + +Twelve public cases use four actual managed CellNodes, signed canonical boots, +native follower stores, an acknowledged SQLite Cell and independent journal +clients. They cover full ensemble publication, policy CAS races, cancellation, +lost replies/source errors, policy changes during publication, superseded pointer +replay, missing history/stale barriers, regressing clocks, withdrawn receivers, +original leader shutdown, local fencing, Closing/deadline refresh and a third +rotation after original receipt eviction. Shutdown joins original owners and +checks empty native resource ledgers. Five codec cases cover bounded envelopes, +complete row identities, policy/scope/time faults and operation adoption. + +Initial compile, directory-only, snapshot-limit and codec-fixture failures remain +under `initial/`. The codec fixture now uses the required maintenance transitions +and drain evidence; production validators and qualification requirements are +unchanged. One unchanged queued-query lifecycle case hit its outer seven-second +waiter timeout during final qualification. Its standalone and full lifecycle +reruns passed with the same assertions; the raw failure remains under +`initial/minion-queued-timeout/`. The executable was relocated to +`crates/cellule-host/minion` during verification. Final qualification uses that +canonical location; its Cargo example target remains `fleet_operations`. + +| Final source and evidence | Recorded value | +| --- | --- | +| Parent revision | `72c5d3a704b69cc1cbc8d3736652dffbceb7dd7d` | +| Frozen source | 721 Rust/Cargo paths, including nested qualification locks. | +| Sorted JSON manifest SHA256 | `5a5c48f26dfece662f4abaccadc3bdbbca1dc77ab995c13f05774181a9a26864` | +| Isolated checkout | `/tmp/cellule-reader-prelease-8698183` | +| Evidence and source archive | `/tmp/cellule-follower-policy-evidence` | + +Final all-feature qualification passed 537 runtime library, 144 runtime lifecycle, +34 host library, 106 public node and 203 executable tests: 1,024 distinct passes, +no failures. Seven existing provider cases remain ignored without their documented +isolated RustFS environment and supply no evidence. Focused repeats are excluded. +Runtime/host Clippy across all targets and API documentation passed with warnings +denied. Complete active/isolated Rust/Cargo path sets and bytes match qualification. +Commands, statuses and build environment remain in `verification.json` and +`build-environment.json`; format, boundaries, layout, documentation links/Rust +fences, SQL/peer and diff gates passed. These selected in-process tests supply no +OS-crash, external-job, provider-deployment, mixed-binary or measured fleet-load +qualification. + +Highest priorities next are failed-owner replacement/successor evidence, complete +production observation and reader/follower controller integration. Receiver-session +recovery, failed Pending producers and live-receiver joining, affected writer +relocation and SettleRoles/Finalize remain open. Terminal drain handoff must join +the original fleet action before entering the existing drain owner. Remaining +primitive faults, maintenance/receiver-loss executable scenarios and W9–W10 +process/provider/mixed-version/load/runbook/rollout qualification remain required. +The complete fleet operations plan remains unfinished. + +## October 2 2026 durable reader replacement evidence checkpoint + +Reader evacuation now produces immutable operation-bound manifests and ordered +replacement pages. The full 10,000-reader policy bound fits 79 pages, each with +at most 128 entries and the existing 64-KiB envelope. The manifest retains the +full original head digest and bootstrapped registry version, exact original +Established row digest and Retired history, serving authority, policy revision +or absence, required prefix and capture interval. Pages bind the complete capture +basis, ordinal, signed boot identities, exact Established replacement requests +and observed native prefixes. Complete validation rejects missing/reordered/foreign +pages, duplicate physical nodes/sessions/request keys and regressed prefixes. +Existing persisted record kinds and domains remain unchanged. + +`FleetReaderEvacuationJournal` commits all pages, the manifest and its latest +per-operation/original-request pointer in the existing journal transaction domain. +First publication compares the full snapshot, live controller, current operation, +original retirement and active replacement rows/intents, then advances the shared +registry. Exact replay returns immutable history without advancing the registry +or restoring a superseded pointer. The reference SQLite implementation uses its +existing finite backend owner and immediate transaction; accepted writes survive +caller cancellation. No new task bank, authority or native closing path is added. + +`FleetReaderEvacuationVerifier` loads every page and the latest pointer after +adapter reconstruction. It checks complete roster barriers, operation/intent, +current source authority and read policy, exact selected signed boots/enrollment +rows and ready native prefixes twice. Its separate fresh interval does not +restamp history. Policy, owner, boot or operation adoption can produce a refreshed +candidate from the same original native retirement through the ordinary current +recruitment path. Evacuating and Closing phases both permit this fresh metadata; +refresh starts no native opening or closure. Publication preserves durable history +and original shared errors independently of a failed final confirmation. +Native capture and fresh verification are bounded by monotonic thirty-second +intervals and the caller/operation deadlines. Historical records grant no complete +role settlement or terminal node finalization. + +Eight public example cases use actual managed CellNode readers, SQLite and the +existing signed directory/native peer path. They cover immutable page publication, +independent journal reconstruction, lost replies and cancelled waiters, policy +changes during publication, zero-reader refresh, superseded-pointer replay, +unavailable replacements, corrupted pages/stale barriers, suspended probes across +source authority change and refresh during Closing. Fixture shutdown joins the +original owners and checks zero native resource ledgers. Seven runtime codec cases +cover the full policy bound, malformed envelopes and scope/lifetime/history rules. +These establish selected in-process behavior; they supply no OS-crash, external-job, +provider-deployment, mixed-binary or measured fleet-load qualification. + +The initial reply-pause fixture was not wired into the post-commit response and +three cases failed to reach their intended race. That raw failure remains under +`initial/`. Qualification below uses the corrected final source. An empty directory +left from an earlier isolated module relocation caused the first layout check to +fail; removing that directory changes no Rust source or compiled path set. + +| Final source and evidence | Recorded value | +| --- | --- | +| Parent revision | `92313e8293ecc76d525cf3659f837af0f014217a` | +| Frozen source | 708 Rust/Cargo paths, including nested qualification locks. | +| Sorted JSON manifest SHA256 | `eaa4010b17ff44ce37f62c0236320ae445e17cef8c297d233f2831077ad74066` | +| Isolated checkout | `/tmp/cellule-reader-prelease-8698183` | +| Evidence and source archive | `/tmp/cellule-reader-policy-evidence` | + +Final all-feature qualification passed 532 runtime library, 144 runtime lifecycle, +34 host library, 106 public node and 191 example tests: 1,007 distinct passes, no +failures. Seven existing provider cases remain ignored without their documented +isolated RustFS environment and supply no evidence. Focused repeats are excluded. +Runtime/host Clippy across all targets and API documentation passed with warnings +denied. Complete active/isolated Rust/Cargo path sets and every byte match final +qualification. Commands/statuses and build environment are retained in +`verification.json` and `build-environment.json`. Format, dependency boundaries, +module layout, document links/Rust fences, SQL/peer and diff gates passed. The +host recipe and example guide describe the public APIs and their proof scope. + +Highest priorities next are durable follower replacement-policy evidence and +revalidation, complete production observation and reader/follower controller +integration. Failed receiver/source/Pending producer reconciliation, affected +writer relocation and SettleRoles/Finalize still require complete barriers and +the original action join before terminal drain handoff. Remaining primitive faults, +maintenance/receiver-loss executable scenarios and W9–W10 process/provider, +mixed-version, load, runbook and rollout qualification remain open. The complete +fleet operations plan remains unfinished. + +## October 2 2026 failed receiver reader closure checkpoint + +`FleetFailedBootProcessRequest::capture` now exposes the original Established +boot and permanent canonical session fence at a complete bootstrapped roster +while role requests remain unresolved. It starts no native/process effect and +permits only provider confirmation. Boot retirement still requires all related +requests settled and complete physical follower-reference checks. Snapshot, +mutable boot status and collection times are metadata; the original request +digest remains unchanged across role publication, recapture and reconstruction. +Existing boot publication shares the same canonical/process confirmation path. + +`FleetFailedReaderRetirement` binds one original Pending, Established or replayed +Retired request to its exact receiver physical node/session and original boot. +Source failure cannot retire a view on a live receiver. The existing application +process provider must authenticate and durably retain original termination or +nonexecution, join all accepted native/external work and producers, and exclude +session reuse. Expiry, native absence, recovery or another lifetime cannot prove +it. The capsule creates no drain task, recovery path or second action bank. + +Publication confirms the entire original journal barrier again after the provider +read, then publishes through the existing enrollment journal. Its deterministic +event binds original full spec/acceptance and immutable process identity/witness; +legitimate establishment completion cannot change that event. Returned history, +committed rows and original shared source failures remain independently inspectable +when a reply or final read fails. Fresh full roster, original boot/reader, permanent +canonical receiver fence and unchanged process evidence confirm within the original +monotonic thirty-second interval. Recapture adopts lost replies without refreshing +original timestamps. Journal/backend owners join accepted work after cancellation. + +Ten public example cases use two actual canonical SQLite readers with application- +owned Pending acceptance before native activation, an actual receiver CellNode, +joined shutdown, retained reader clones and the durable SQLite journal. One native +opening retains an unresolved establishment result. Original lifetime evidence is +stored only after shutdown, local joining, rejected cloned-view queries and zero +native resource ledgers. The source retains readable acknowledged data. Cases cover +running-original refusal, failed-source/live-receiver refusal, exact original process, +payload and acceptance comparison, Pending/Established history, boot closure ordering, +replay, independent adapter reconstruction, lost publication replies, cancelled waiters, +suspended provider reads, stale barriers, regressing/expired capture clocks and failed +or changed final process evidence. Original process identity survives role publication. + +The first eight-case run found a fixture expecting Fenced after the entire runtime +had closed; canonical query admission correctly returns RuntimeClosed first. The +fixture now checks that exact native outcome together with actual lifetime joining. +That raw failure and source manifest remain under `initial/`; qualification below +uses the corrected final source. These are in-process native lifetime cases with no +external-job workload, not OS-crash or provider-deployment qualification. + +| Final source and evidence | Recorded value | +| --- | --- | +| Parent revision | `a9cee96cda6d944dad6e5711a0ac8352baa1f6e9` | +| Frozen source | 696 Rust/Cargo paths, including nested qualification locks. | +| Sorted JSON manifest SHA256 | `dda6c960a80859e941de7d407f28d74de56c84cd73711af0e7bb6d81cf33e568` | +| Isolated checkout | `/tmp/cellule-reader-prelease-8698183` | +| Evidence and source archive | `/tmp/cellule-failed-reader-evidence` | + +Final all-feature qualification passed 525 runtime library, 144 runtime lifecycle, +34 host library, 106 public node and 183 example tests: 992 distinct passes, no +failures. Seven existing provider cases remain ignored without their documented +isolated RustFS environment and supply no evidence. Focused repeats are excluded. +Runtime/host Clippy across all targets and API documentation passed with warnings +denied. Complete active/isolated Rust/Cargo path sets and every byte match final +qualification. Commands/statuses and retained build environment are in +`verification.json` and `build-environment.json`. Format, dependency boundaries, +module layout, document links/Rust fences and SQL/peer gates passed; the public +host recipe and example command document the exact scope. + +Highest priorities next are persisted reader/follower replacement-policy evidence +and revalidation, complete production role observation and controller integration. +A failed source with a live receiver still needs ordinary joined reader closure; +failed owner/Pending producer reconciliation remains required. Affected-writer +relocation, SettleRoles/Finalize and the original action join before terminal drain +handoff remain open, followed by remaining primitive faults, complete maintenance/ +receiver-loss executable scenarios and W9–W10 process/provider/mixed-version/load/ +runbook qualification. This closes individual failed receiver reader enrollments; +the complete physical maintenance operation and full plan remain unfinished. + +## October 2 2026 original failed boot closure checkpoint + +`NodeDirectory::closed_session` now exposes a fresh opaque permanent fence for +an exact physical node/session, with no leader log or its exact Retired log. +Missing/live/expired advertisements and Open/Recovering/Sealed logs refuse, +including inactive enrolled logs. The original fencing/expiry and terminal +ensemble/manifest survive claim renewal. No record, wire or persistence codec +changes; this canonical read starts no retirement or process effect. + +`FleetFailedBootRetirement` binds that fence to the complete bootstrapped roster +and the original Established request/history. Missing/duplicate boot requests, +unresolved original reader/follower/Pending rows and contradictory live foreign +references prevent capture. A new boot may retain other registered foreign epochs +on the same physical node; old closure cannot retire or substitute that role. +`FleetFailedBootProcesses` is a read-only application boundary for authenticated +durable original process/nonexecution evidence. Providers must join the process +and its accepted external jobs/producers, exclude reuse of the same session and +retain the same original witness across reconstruction. Expiry, takeover, +missing inventory, recovery success and a replacement process cannot supply it. +The evidence constructor validates shape, never provider authentication. + +Publication rechecks the entire original journal barrier after the provider +read, then uses the existing enrollment journal. A committed row and its +original shared source failure remain inspectable when a publication reply or +final check fails. Fresh complete roster, exact returned original history, +canonical authority, physical references and the same durable process witness +must confirm within the original monotonic thirty-second interval. Fresh +recapture adopts original times after cancellation/lost replies or controller +restart. The application/adapter retains and joins accepted backend work; +this capsule creates no task, kill/drain operation or second action bank. +Active intent rebinding now documents either checked planned withdrawal or +complete failed-boot/process/role closure as the old-session prerequisite. + +One runtime case covers missing/live/expired, exact physical identity (distinct +from SessionId), claimant expiry and immutable original closure through claim +renewal. Eight Unix example cases combine actual cold canonical recovery and +two-member retirement, the SQLite journal and a real child lifetime. They cover +running-original refusal, kill/wait before durable evidence, independent adapter +reconstruction, replay without timestamp refresh, lost retirement reply with +original source/Arc retention, cancellation followed by backend joining, +unretired logs, unresolved roles, duplicate requests, foreign process witnesses, +regressed clocks, stale snapshots, final process error/changed evidence, delayed +original-session responsibility and a new boot's foreign role on the same node. +The child is a lifetime stand-in, not a CellNode process or external-job workload; +those fixtures do not supply multi-process/provider deployment qualification. +The existing real acknowledged-tail recovery case remains in the lifecycle suite. + +The first runtime-focused case passed. Host compilation then exhausted the +mounted Workspace volume before running any host case. That source manifest and +raw build failure remain in `disk-full/`. Only this task's generated isolated +target artifacts were removed. Final qualification disables incremental build +storage and debug symbols, preserving all features, test assertions, workloads +and named qualification profiles. Its environment is retained separately. + +| Final source and evidence | Recorded value | +| --- | --- | +| Parent revision | `6fba32403c13f3a9ecf8444cf1763990bedc4524` | +| Frozen source | 688 Rust/Cargo paths, including nested qualification locks. | +| Sorted JSON manifest SHA256 | `a1cab1a03e91c77a97df6cf8a3407d2162125b205233af7a87b857874368dfda` | +| Isolated checkout | `/tmp/cellule-reader-prelease-8698183` | +| Evidence and source archive | `/tmp/cellule-failed-boot-evidence` | + +Final all-feature qualification passed 525 runtime library, 144 runtime lifecycle, +34 host library, 106 public node and 173 example tests: 982 passes, no failures. +Seven existing provider cases remain ignored without their documented isolated +RustFS environment and supply no evidence. Earlier/focused repeats are excluded. +Runtime/host Clippy across all targets and API documentation passed with warnings +denied. Format, boundaries, module layout, document links/Rust fences and SQL/peer +gates passed. Complete active/isolated Rust/Cargo path sets and every byte match +qualification. Commands/statuses and the build environment are retained in +`verification.json` and `build-environment.json`; guides include the public recipe +and focused command. + +Highest priorities next are failed-reader/process integration, persisted +replacement-policy revalidation and complete production role observation. +Affected-writer relocation, `SettleRoles`/`Finalize`, the original action's join +before terminal drain handoff, remaining primitive faults, four executable +scenarios and W9–W10 provider/compatibility/load/runbook qualification remain +required. This closes one original boot enrollment under explicit application +process evidence, not a physical maintenance operation. The full plan stays active. + +## October 2 2026 recovered follower enrollment publication checkpoint + +`FleetRecoveredFollowerRetirement` binds a fresh canonically Retired epoch to +its complete original member requests in the durable roster. The receiver-side +runtime authorization now exposes the original physical leader from the exact +tombstone; a session-derived NodeId cannot replace it. Capture requires the +bootstrapped full journal barrier, exact leader/epoch/ensemble/manifest, uniform +original source endpoint and one non-refused request per original member. +Missing, duplicate, foreign or differently settled requests refuse the capsule. + +Publication uses the existing enrollment journal and deterministic events bound +to immutable specs and first acceptance. It preserves establishment history and +original retirement timestamps across fresh recapture after a lost reply. All +member waiters settle with separate retained errors; the adapter continues to +own accepted backend work after cancellation/deadline. Successful writes remain +observable when a sibling or the final check fails. Confirmed closure requires +the complete post-publication roster, unchanged original member records, fresh +canonical authority and a non-regressing interval of at most thirty seconds. +The API starts no native RPC, recovery path, task or second finite action bank. + +Seven public example cases use actual signed boot enrollment, a cold two-member +native ensemble, canonical recovery/sealing/retirement and the SQLite journal. +They cover Pending/Established originals, independent-client reconstruction after +native collection, stable duplicate times/digests, lost publication reply and +original I/O source/Arc retention, cancellation followed by backend joining, +unretired authority, stale barriers, expired claimants, duplicate requests, +clock regression and a delayed new request at the final roster barrier. +The failed leader's boot remains an unresolved obligation after follower closure. +These cold-lane fixtures establish no Cell suffix pinning or failed-process proof; +the existing real lost-acknowledgement runtime recovery case remains in the +qualified lifecycle suite. + +Initial compilation found fixture PathBuf ownership and an incorrect helper +name. The first executable run also found a test that inspected facility display +text instead of its original I/O source. Those fixtures now use the public +journal API and check the actual source type/message. The lost-reply assertion +tracks the actual member selected by concurrent completion. Earlier manifests +and raw failure/focused logs remain separate from final qualification. + +| Final source and evidence | Recorded value | +| --- | --- | +| Parent revision | `db339147f7308ce11c3f599ccea0fbe4ddd6939a` | +| Frozen source | 681 Rust/Cargo paths, including nested qualification locks. | +| Sorted JSON manifest SHA256 | `696bb5d9801515b1585c7ce7ff30bdc853b921a112b8ffee6fe384b65772aad8` | +| Isolated checkout | `/tmp/cellule-reader-prelease-8698183` | +| Evidence and source archive | `/tmp/cellule-recovered-enrollment-evidence` | + +Final all-feature qualification passed 524 runtime library, 144 runtime lifecycle, +34 host library, 106 public node and 165 example tests: 973 passes, no failures. +Seven existing provider cases remain ignored without their documented isolated +RustFS environment and supply no evidence. Focused repeats are excluded from +these counts. Runtime/host Clippy across all targets and API documentation passed +with warnings denied. Format, boundaries, module layout, document links/Rust +fences and SQL/peer gates passed. The complete active/isolated Rust/Cargo path +sets and every byte match after qualification. All command arguments and exit +statuses are retained in `verification.json`; the host guide contains the public +publication recipe and the example guide supplies its focused command. + +Highest priorities for the next increment are failed-boot/process closure, +persisted replacement-policy revalidation and consumption through complete role +observation/maintenance actions. Affected-writer relocation, `SettleRoles`/ +`Finalize`, original-action joining before terminal drain handoff, remaining +primitive faults and W8–W10 remain required. This closure covers only the +original follower enrollment rows; the full implementation plan stays active. + +## October 2 2026 recovered follower retirement checkpoint + +Recovered failed-owner tails now close through an explicit capability of the +existing node-log transport. Each receiver authenticates a live requester and +rechecks the exact canonical Sealed/Retired epoch, original member and pinned +manifest. The native store accepts no caller-selected watermark. It uses the +existing lane lock, byte ledger, verified scanner and durable retirement marker; +active lanes require their exact original seal. Inactive enrollment only fences +empty lanes, refusing unexpected records even when they have a native seal. + +All original member calls join and retain their individual responses/errors. +Incomplete or contradictory receipts cannot produce the opaque confirmation +required for the canonical Retired CAS. The tombstone preserves the epoch, +ensemble and recovery manifest. Takeover and original recovery completion remain +valid. Exact canonical retirement can be adopted after controller restart or +local grace collection, without inventing missing member receipts. Grace-aged +collection still requires the unchanged native marker and external authority. + +Three unit cases cover live/Recovering and foreign/expired authorization, +unsealed and corrupt-seal refusal, different original prefixes, lost member +replies, contradictory receipts, durable fences, exact replay, native collection +and inactive unexpected records. Their authority fixtures do not establish +full recovery pinning. The real lost-acknowledgement lifecycle case does: it pins +the recovery overlay, retires the suffix, loses the committed retirement CAS +reply, adopts its canonical result, then restores the successor and verifies the +original command outcome without re-execution. Earlier qualification logs are +retained separately before the seal and inactive-lane refinements. + +| Final source and evidence | Recorded value | +| --- | --- | +| Parent revision | `af40830ebf495e348235fcdc016626844986b99a` | +| Frozen source | 676 Rust/Cargo paths, including nested qualification locks. | +| Sorted JSON manifest SHA256 | `7f70f0f5930eacc13553247c3a543f2fc1970abad8c1accaacf8b3ccbfe7fffb` | +| Isolated checkout | `/tmp/cellule-reader-prelease-8698183` | +| Evidence and source archive | `/tmp/cellule-recovered-retirement-evidence` | + +Final all-feature qualification passed 524 runtime library, 144 runtime lifecycle, +34 host library, 106 public node and 158 example tests: 966 passes, no failures. +The runtime suites ignored seven existing provider cases whose required isolated +RustFS environment was not supplied; they provide no evidence here. Focused +repeats are excluded from these counts. Runtime/host Clippy across all targets +and API documentation passed with warnings denied. Format, boundaries, module +layout, document links/Rust fences and SQL/peer gates passed. The complete +active/isolated Rust/Cargo path sets and every byte match after qualification. +All command arguments and exit statuses are retained in `verification.json`. + +This establishes runtime tail closure, not failed-process joining, replacement +policy or fleet finalization. Durable publication against every original member +request, failed-boot closure, complete production observation, affected-writer +relocation, `SettleRoles`/`Finalize`, terminal action handoff, remaining primitive +faults and W8–W10 remain required. The full implementation plan stays active. + +## October 2 2026 retained role observation checkpoint + +`FleetObservation::with_role_coverage` retains the original checked graph and +binds its digest to canonical planner inputs. Attachment rejects a second graph, +scope/registry mismatches and intervals outside the original observation. +After confirming the journal, the reconciler compares the graph's exact full +head, registry and roster digest before planning. A controller renewal invalidates +an earlier graph even when enrollment and intent revisions are unchanged. +The reference observer now retains its graph through this exported path. +Partial adapter coverage remains partial after attachment. + +Three public cases use actual managed follower boots, native collectors and the +durable journal. They verify retained original metadata and rejected replacement +or restamping, an earlier head with unchanged registry, and complete role graph +attachment to a deliberately partial adapter observation. The latter runs the +exported driver, retains IncompleteObservation and allocates no count movement. +The stale-head case checks that no effect or inspection was dispatched and no +attempt was allocated. All fixtures join shutdown and exact boot/member retirement. + +Initial runs found two fixture mistakes: the registry advance API requires its +expected revision, and managed fixtures deliberately keep scheduling disabled. +The driver cases now explicitly enable reference scheduling before collecting +their evidence. Production checks and expected barriers remain unchanged. +Earlier compile/failure logs are retained separately from final qualification. + +| Final source and evidence | Recorded value | +| --- | --- | +| Parent revision | `8a43d80aafbc581e39e5fd3e13b355c7de3099dc` | +| Frozen source | 673 Rust/Cargo paths, including nested qualification locks. | +| Sorted JSON manifest SHA256 | `28331425f9220b9e84ecf665e6bc8fb35580c6c2baa6a131e54f43b60a46f7a8` | +| Isolated checkout | `/tmp/cellule-reader-prelease-8698183` | +| Evidence and source archive | `/tmp/cellule-retained-role-evidence` | + +Final qualification passed 34 host library, 106 public node and 158 example +cases: 298 distinct tests, no failures or ignores. Host Clippy and host/runtime +API documentation passed with warnings denied. The complete active/isolated +Rust/Cargo path sets and every byte match after qualification. Focused repeats +are excluded from the final count. + +This retains one checked input barrier; it does not establish complete production +observation or maintenance completion. Replacement-policy publication and +revalidation, failed-process closure, dead-owner recovery, affected-writer +relocation, `SettleRoles`/`Finalize`, terminal action handoff, remaining primitive +faults and W8–W10 remain required. The full implementation plan stays active. + +## October 2 2026 cross-node role coverage checkpoint + +`FleetRoleCoverage` checks every required original native boot and every +retained physical node's complete follower reference scan. All initial captures +precede the global native round; every exact foreign recheck follows that entire +round. Missing or duplicate inputs, regressed/stale intervals and incomplete +rechecks are refused. Beginning a new native or foreign recheck invalidates its +earlier confirmation, even when the new attempt fails or is dropped. + +An Established follower can have no persisted lane until its first append. +Coverage matches that obligation to the delivered original managed producer, +installed source binding, original Established member records and matching +current Open authority on every ensemble member. The local enrollment check +remains strict without this combined witness. Failed/unobserved owners and +quarantined receiver stores remain blockers. The canonical input digest binds +the original roster, all native fingerprints, exact foreign rows and actual +collection/recheck intervals. The reference writer observer consumes this check +before its final full journal confirmation. + +Six public cases exercise ordered global rounds, missing/duplicate inputs, +order-independent digests, dropped native rechecks, failed foreign rechecks, +actual enrolled empty lanes after four-node rotation and changed full rosters. +They retain original rows/timestamps and join real managed fixture shutdown +with exact retirement counts. The earlier five-case focused run is retained +separately from final-source qualification. + +| Final source and evidence | Recorded value | +| --- | --- | +| Parent revision | `7b6df307d6730826fd6609656a7d92372a7840c1` | +| Frozen source | 672 Rust/Cargo paths, including nested qualification locks. | +| Sorted JSON manifest SHA256 | `c75db3afe1fa91164b5d6e363030c37d0f713288d8410365a128382d99a277d7` | +| Isolated checkout | `/tmp/cellule-reader-prelease-8698183` | +| Evidence and source archive | `/tmp/cellule-role-coverage-evidence` | + +Final qualification passed 34 host library, 106 public node and 155 example +cases: 295 distinct tests, no failures or ignores. Host Clippy and host/runtime +API documentation passed with warnings denied. The complete active/isolated +Rust/Cargo path sets and every byte match after qualification. Focused repeats +are excluded from the final count. + +This supplies combined role coverage, not settled roles or shutdown permission. +Authentication, unexpected advertisement discovery, current Cell authority, +replacement-policy publication/revalidation and failed-process closure remain +required. Pending work remains an obligation. Complete observer integration, +dead-owner recovery, affected-writer relocation, `SettleRoles`/`Finalize`, terminal +action handoff, remaining primitive faults and W8–W10 remain unfinished. + +## October 2 2026 live-owner follower evacuation checkpoint + +`CellNode::follower_evacuation` checks the original requested rotation against +the current Evacuating operation and full durable roster. It preserves the +original completion Arc and retirement timestamps, requires every old member's +Retired row, and verifies the complete newer ensemble outside the donor against +an explicit application member minimum. The returned evidence retains original +signed replacement boots, current Open authority, Established member rows, +exact intent revisions and the checked head/registry. Native owner readiness, +admission mode, binding and actual lease are rechecked alongside directory +authority and member boots. Temporary and retained metadata use the existing +node ledger; deadlines retain original source errors. + +The API starts no effects. The same supervisor owns accepted retirement and +recruitment across cancellation, deadlines and lost replies. Missing local +completion, insufficient replacement, changed/withdrawn boots, fenced owner or +stale operation produces no settlement evidence. No-spare recruitment remains +outstanding after object-covered old retirement; this does not invent a newer +binding or a completed maintenance operation. + +Seven public cases use actual three/four-node managed boots, native persisted +follower lanes, an acknowledged SQL command, the canonical object barrier and +the durable reference journal. They cover exact replay, no spare, lost member +reply with the original error Arc, invalid originals/member requirements, +expired waiters, replacement withdrawal, owner fencing and deadline extension. +Retired files/fences remain present. Every fixture joins shutdown, checks empty +native resource ledgers and requires exact boot/follower retirement counts. +The reference heartbeat now signs actual retained follower bytes as well as +available receiving credit. + +Initial qualification found two test API assumptions and a query attempted +after its actual drain closed query admission. Readback now precedes the +publication barrier and the case verifies its resulting covered watermark. +The first broad run passed 287 cases; Clippy found a redundant capacity default +after all fields became explicit. Earlier logs remain separate from final +source qualification under `/tmp/cellule-follower-evacuation-evidence`. +The explicit Open/Active authority check subsequently exposed a source boot +still carrying its closed startup sample. The managed fixture now publishes +that source's real post-start heartbeat before recording its live-owner proof; +the production admission check remains required. + +| Final source and evidence | Recorded value | +| --- | --- | +| Parent revision | `3721ee64d260d57eec2e59dda72632bb77c789ee` | +| Frozen source | 669 Rust/Cargo paths, including nested qualification locks. | +| Sorted JSON manifest SHA256 | `169abf0111599ceb01b9f0ff02e61e4a835e348566e48a3471e6d6814c4ad05b` | +| Isolated checkout | `/tmp/cellule-reader-prelease-8698183` | +| Evidence and source archive | `/tmp/cellule-follower-evacuation-evidence` | + +Final qualification passed 34 host library, 106 public node and 149 example +cases: 289 distinct tests, no failures or ignores. Host Clippy and host/runtime +API documentation passed with warnings denied. Format, boundaries, module +layout, document syntax/links and SQL/peer contract gates pass. The complete +active/isolated Rust/Cargo path sets and every byte match after qualification. +Focused repeats and earlier-source runs are excluded from the final count. + +Complete observer/failed-process barriers, policy evidence publication and +revalidation, dead-owner recovery, affected-writer relocation, +`SettleRoles`/`Finalize`, terminal action handoff, remaining primitive faults +and W8–W10 remain required. This checkpoint supplies one live-owner role proof, +not fleet-wide settlement or shutdown permission. + +## October 2 2026 foreign follower authority checkpoint + +`FleetFollowerReferences` traverses every canonical directory page for a physical +follower, including expired advertisements and fenced tombstones. It matches +references to the original full roster and rechecks exact authority rows after +native collection. Coverage and leader liveness are compared even though native +continuation fingerprints omit them. Changed or incomplete captures retain the +original rows and interval; no reference absence settles Pending work. + +The closed writer example now uses this collector and its final recheck. The +canonical heartbeat advertises real available follower-store bytes when that +component is installed. A new managed fixture uses three actual boot owners, +node-owned follower stores, canonical enrollment and an acknowledged SQL mutation. +Its object-coverage callback is paused to exercise the actual follower response +proof, then resumed and joined before observation. + +Four public cases cover original producer/foreign lane matching, lost retirement +replies with shared original errors, real object publication changing coverage +without changing topology, and deadlines/regressed clocks/expiry. A retired local +lane still has a foreign obligation until canonical owner closure confirms. +Two directory cases exercise real canonical continuation pages and refuse a +foreign member or regressed capture. They are directory contract fixtures, +separate from the managed three-node cases. + +Initial qualification caught two fixture assumptions: object proof can win before +log activation, and the supervisor retains its old binding while replacement +recruitment waits. The fixture now explicitly proves the follower path and the +rotation case checks confirmed old retirement plus the outstanding Recruiting +phase. No replacement is invented. Initial logs and a passing pre-layout run +are retained separately; final qualification follows the corrected module layout. + +| Final source and evidence | Recorded value | +| --- | --- | +| Parent revision | `57559a5ef3fe534ab3081f8e6f8eb9da4d3011cd` | +| Frozen source | 667 Rust/Cargo paths, including nested qualification locks. | +| Sorted JSON manifest SHA256 | `a16464289aca4f39375582da32328fa1f0e2b006e338cb5409bb8eb07ec01ee3` | +| Isolated checkout | `/tmp/cellule-reader-prelease-8698183` | +| Evidence and source archive | `/tmp/cellule-follower-references-evidence` | + +The final source passed 34 host library, 106 public node and 142 fleet example +tests: 282 distinct cases, zero failures or ignores. Host Clippy and host/runtime +API documentation passed with warnings denied. Format, boundary, module layout, +document syntax/links and SQL/peer contract gates pass. The complete path set and +every Rust/Cargo byte are compared against the isolated snapshot after the run. + +This supplies complete reference discovery and managed aggregate regression +coverage. Complete observer/policy/failed-process barriers, live-owner replacement +ensembles, dead-owner recovery, affected-writer relocation, `SettleRoles`/`Finalize`, +terminal action handoff, remaining primitive faults and W8–W10 still remain. + +## October 2 2026 merge qualification checkpoint + +Merged `main` at `18a1244a98c50a04c52da21325f95331055e9884` into the native +aggregate traversal checkpoint. The qualification driver retains `main`'s +canonical arrival collector and both new regression tests, with evidence writes +after the original arrival and drain clocks. Qualification thresholds remain +unchanged. + +The merged source passed 44 app integration tests, 144 runtime lifecycle tests, +32 host unit tests, 106 public node tests and 138 fleet example tests in the +isolated checkout: 464 passing cases and no failures. Twenty documented manual +or environment-dependent tests were ignored. App/host Clippy and host/runtime +API documentation passed with warnings denied, together with format, boundary, +module layout, document and SQL/peer contract gates. + +The exact 662 Rust/Cargo paths match the isolated snapshot. Their sorted JSON +manifest SHA256 is +`38d8c01b33be4254f8f5feac653081f8f9dffa53bed2a80d17550a4bc6bdb1cf`. +Merged-source logs and the manifest are retained in +`/tmp/cellule-native-inventory-merge-evidence`; the preceding checkpoint's +manifest and results are preserved there with `before-merge-` filenames. +Provider/process CI qualification remains separate from these local checks. + +## October 2 2026 native aggregate traversal checkpoint + +`FleetNodeInventoryScan` now traverses every native category under the exact +bootstrapped full roster, preserves native continuations, and rechecks complete +category fingerprints. Its result retains writer transitions, reader producer +jobs and original errors, persisted follower lanes, follower preparation state +and supervisor rotations. `validate_enrollments` matches exact original requests, +accepted timestamps and published evidence; missing bindings and failed boots +never become empty roles. Applications account these bounded copied buffers. + +The reference observer uses this collector and rechecks all seven categories +after its fleet-wide Cell authority scan. It retains independently revalidated +writers for pressure relief when topology invalidates a full traversal. Count +coverage remains disabled for that interval. Actor maps can overlap during +release; the collector preserves transitional keys rather than double-counting +them or treating them as absent. + +Seven new public-path cases cover native continuation, nonce reuse, incomplete +and changed rechecks, partial pressure inputs, a real canonical owner renewal, +managed reader capture before/after evacuation, and original errors after a +lost retirement reply. They use actual managed boots, actor/query lifecycles, +canonical authority and the durable journal. Shutdown checks empty resource +ledgers through the existing fixtures. + +Qualification found two regressions. First, a changing actor topology caused an +observer error during real movement; this now disables completeness and preserves +only independently checked pressure rows. Second, the equilibrium example +required immediately complete captures after its last movement. Instrumented +runs observed canonical renewals changing revision/progress without changing +owner, fence or root; the deterministic renewal case verifies that the exact +authority recheck correctly invalidates such an interval. The example now +requires two consecutive complete, fresh post-batch samples within its original +120-second convergence deadline. Extra allocation, active attempts and native +failures still fail immediately; the final complete count scan also remains +required. No residence, count, deadline, or receipt requirement was reduced. +The earlier failure logs and diagnostic traces are retained separately. + +| Final source and evidence | Recorded value | +| --- | --- | +| Parent revision | `b36f76321539e306bbce5d25973f6595dbac8df5` | +| Frozen source | 662 Rust, Cargo manifest and Cargo lock paths, including nested qualification locks. | +| Manifest SHA256 | `9610c21eb6045a461676c449c849cbb91330d05d5b560106c23abbbc2d3dc428` | +| Isolated checkout | `/tmp/cellule-reader-prelease-8698183` | +| Evidence and source archive | `/tmp/cellule-native-inventory-evidence` | + +The manifest SHA256 hashes sorted-path JSON with two-space indentation and a +final newline. The complete path set and every Rust/Cargo byte are compared +against the isolated snapshot and current checkout after qualification. + +| Final-source command | Observed result | +| --- | --- | +| `cargo test -p cellule-host --lib --test node --example fleet_operations --all-features --locked -- --test-threads=2` | 32 library, 106 public node and 138 example cases passed; zero failures/ignores. | +| `cargo clippy -p cellule-host --lib --test node --example fleet_operations --all-features --locked -- -D warnings` | Passed with warnings denied. | +| `RUSTDOCFLAGS='-D warnings' cargo doc -p cellule-host --all-features --no-deps --locked` | Passed with documentation warnings denied. | + +These are 276 distinct passing cases; focused repeats are excluded. Format, +crate boundaries, module layout, Markdown links/Rust syntax and SQL/peer +contract gates also pass. This provides canonical local traversal and reference +consumption. It does not finish W3/W7: complete foreign log authority discovery, +replacement-policy and failed-process evidence, role-enabled aggregate follower +fixtures, affected-writer relocation, `SettleRoles`/`Finalize`, terminal action +handoff, remaining busy-primitive/fault work, full W8 examples and W9–W10 remain +required. The full plan stays active. + +## October 2 2026 managed reader evacuation checkpoint + +`ReadReplicaManager::evacuate` checks one original Established reader against +the current Evacuating operation. It traverses the durable roster, requires +current managed boots and canonical policy selection outside the donor, and +uses authenticated native status before closure and after original retirement. +The final probe covers a refresh completed through a retained peer clone after +the first probe. Capacity refusal preserves the open donor; changed policy, +authority, boot or registry refuses completion evidence. + +Managed Draining reconciliation now preserves open views until explicit +evacuation or terminal native shutdown. Canonical closure and retirement keep +their original owners across cancelled waiters and lost replies. Roster pages, +retained records, candidate observations and returned evidence use the existing +runtime metadata ledger. Deadline clamping uses the actual operation clock; +timeout sources remain available. + +Ten new cases use three real leased/enrolled nodes, signed peer dispatch, native +SQLite readers and the durable reference journal. They cover no spare, nonzero +replacement, cancellation, deadline, lost retirement reply, policy change, +metadata refusal, replacement withdrawal before/after closure, and peer refresh +racing the first probe. Each case joins shutdown, checks empty native resource +ledgers, and confirms original boot withdrawal/retirement. Three roster unit +cases cover retained credit, capacity refusal and ambiguous/obsolete boots. + +A repeat of the existing movement suite exposed a valid BusyExecution refusal +after command acknowledgement. That fixture assumed acknowledgement implied +idle transfer readiness. It now waits for two stable, unblocked native samples +at the acknowledged prefix before dispatch. The production refusal contract and +qualification thresholds are unchanged. The first evacuation run also caught +reply-pause assertions that expected a receiver to survive waiter cancellation; +the corrected cases verify its disappearance, retained unpublished retirement +and exact replay. Both failure logs are retained. + +| Source and evidence | Recorded value | +| --- | --- | +| Parent revision | `7541774e8cc4b66d9b8dac2ef13abb6dbcce2aca` | +| Frozen source | 655 Rust, Cargo manifest and Cargo lock paths, including nested qualification locks. | +| Manifest SHA256 | `a67ee362cde52a8ab4eba3b7086b4f39ae572d505d1db7e51d0550ba62170c34` | +| Isolated checkout | `/tmp/cellule-reader-prelease-8698183` | +| Logs and source archive | `/tmp/cellule-reader-evacuation-evidence` | + +The manifest hashes sorted `SHA256 relative-path` lines with a final newline. +The isolated source and current working tree are byte-compared against the +complete manifest and path set. Documentation changes after freezing do not +change those Rust/Cargo bytes. + +| Final-source command | Observed result | +| --- | --- | +| `cargo test -p cellule-host --lib --test node --example fleet_operations --locked` | 32 library, 106 public node and 131 example tests passed; zero failures/ignores. The example includes overload, controller restart and real-time count equilibrium. | +| `cargo clippy -p cellule-host --lib --test node --example fleet_operations --all-features --locked -- -D warnings` | Passed with warnings denied. | +| `RUSTDOCFLAGS='-D warnings' cargo doc -p cellule-host -p cellule-runtime --all-features --no-deps --locked` | Passed with documentation warnings denied. | +| `cargo test -p cellule-runtime --lib node:: --all-features --locked` | 106 passed; zero failures/ignores and 418 filtered out. | + +These are 375 distinct passing cases; focused repeats are excluded. This +checkpoint proves a per-reader path, not complete fleet settlement. Affected +writer relocation, foreign follower/replacement and failed-process evidence, +complete observation barriers, terminal action handoff, `SettleRoles`/`Finalize`, +maintenance/receiver-loss examples and W9–W10 qualification remain required. + +## October 2 2026 native boot withdrawal checkpoint + +Managed boots can now bind their original authenticated directory version, +Established enrollment row and journal to the retained host drain before +readiness. The reference example installs that binding after atomic startup +confirmation. The original closing task joins facilities, runtime and lease +maintenance, withdraws the exact boot, checks its canonical terminal state, +and confirms durable retirement before exposing Stopped. Cancellation of the +sole caller leaves the original closing task owned. A deadline or ambiguous +committed retirement retains Draining and the same evidence for replay. + +Eight native example cases cover success/replay, caller cancellation, lost +retirement replies, a retirement deadline, an actual later signed heartbeat, +immutable boot binding, missing canonical storage and failed role closure. +They use the exported host API and the real directory/SQLite journal. No +runtime shutdown is repeated and no storage absence invents retirement. + +Review also found that canonical withdrawal could accept a stale-collected +tombstone that still carried node-log authority when the caller's original +token preceded recruitment. A new runtime regression reproduced that success +before the fix and passed afterward. Withdrawal now refuses that retained log. +`NodeDirectory::is_withdrawn` confirms a tombstone without log authority or a +recovery claimant; `is_retired` continues to mean a permanent fence. Native +closing and the reference application's partial-startup cleanup use the +stronger check. Persisted and signed record formats are unchanged. + +The previous count-equilibrium failure remains unresolved; the scenario now +retains its failed pass report. A passing replay is not a fix. Parent `238b816` +CI completed successfully for workspace/MSRV, decoder fuzz and follower proof, +but object-capacity repeat three failed: its skewed two-client window completed +59 of 60 planned requests and found no fully served capacity point. That raw +driver log and the provider artifact are retained with this checkpoint. These +results concern the parent binary and do not qualify this new source. + +Evidence is retained under `/tmp/cellule-boot-withdrawal-238b816-evidence`. +The final isolated run byte-checks all 648 Rust/Cargo/lock paths and complete +path sets before and after each command. Manifest SHA256: +`ced2be60817e744d551e84045adeebe9dca394579542d99b70f78e99c135d359`. + +| Final native scope | Result | +| --- | --- | +| Runtime library / public fleet, all features | 521 / 26 passed; three / one existing ignores. | +| Host library / public node, all features | 29 / 106 passed. | +| Fleet reference example, all features | 121 passed, including all eight new closing cases, overload, controller restart and count equilibrium. | +| Workspace all targets/features, locked | Check and Clippy passed with lint warnings denied. | +| Runtime/host API documentation and static gates | Passed with documentation warnings denied; format/diff, boundaries/layout, Rust fences, links and SQL/peer contracts passed. | + +These are 803 distinct passed cases and four existing ignores. Focused repeats +are excluded. The count pass does not erase its earlier failure or establish +its cause. No new process/provider, sustained-traffic or mixed-binary evidence +is claimed for this source. + +Complete role/replacement and affected-writer relocation evidence, the terminal +action's join and handoff, committed maintenance completion, the remaining +examples and W9–W10 qualification are still required. `SettleRoles` and +`Finalize` remain refused; boot closure alone cannot authorize them. + +## October 2 2026 pressure selection checkpoint + +The unchanged overload scenario reproduced a competing local eviction. The +trace binds the same Cell, actor generation and last-use timestamp: native +oldest-first eviction began at `1790968652472`, fleet allocation followed at +`1790968652474`, and receiver preparation refused the fenced source. The first +movement completed while the second was safely cancelled. This is an advisory +selection race, not permission to weaken source authority or count cancellation +as successful relocation. + +`CellTransferDemand` now carries the actor's actual `last_used_ms`. Under +Shedding or Critical pressure on an Active donor, the pure planner prefers +recent settled Cells. Native emergency eviction remains oldest-first and keeps +its existing budget and dwell rules. Normal balance and maintenance preserve +Cell identity ordering; invalid recency timestamps are excluded. This is a +contention reduction, not an actor reservation or a general convergence proof. +Exact source release and receiver cleanup still handle racing closures. + +The native fixture now boots Cells in ascending identity order, making the +lowest identities the oldest eviction candidates. This adversarial ordering +removes accidental success from randomized HashMap iteration. The corrected +fixture fails before the policy change with one movement. The initial fixture +compile error (CellId has no Ord implementation) and its correction to stable +identity bytes remain recorded separately. Pressure load, deadlines, residence, +receiver budgets, two-movement results and zero-resource assertions are intact. + +Two private planner regressions and a public permutation case cover the new +order, settlement, timestamp refusal, stable identity ties and unchanged drain +and ordinary balance behavior. All 20 paired native replays pass: 20 overload +and 20 controller-restart cases. Repeats add no distinct test coverage. These +results do not establish process/provider or mixed-version qualification. + +### Current verification + +All 646 Rust/Cargo/lock paths are byte-checked between the active checkout and +isolated snapshot before and after each command. Their final path sets also +match. Manifest SHA256: +`c7005daa497dca9ad9893176faf129b0beefbfbd9446e9e003d06f42d140bf70`. + +| Native scope | Result | +| --- | --- | +| Complete runtime library, all features | 520 passed; three existing ignores. | +| Complete public fleet suite, all features | 26 passed; one documented provider test ignored. | +| Complete host library and public node suite | 29 and 106 passed. | +| Complete fleet example, all features | 112 passed; count equilibrium check failed. Overload and controller restart passed. | +| Workspace all-target/all-feature locked check and Clippy | Passed; lint warnings denied. | +| Runtime/host API docs and static gates | Passed with documentation warnings denied; format, boundaries/layout, Rust fences, links and SQL/peer contracts passed. | + +These are 793 distinct passed cases, four existing ignores and one failure. +The balance failure states that equilibrium did not remain settled while +scheduling stayed enabled; its current error lacks the failed pass detail. +Its cause is unproven. The full failure, frozen executable and source remain +retained under `/tmp/cellule-pressure-selection-a2806db-evidence`, alongside +before/after, diagnostic and replay evidence. Required profiles and expected +results were not weakened. Parent `a2806db` CI has 112 example passes and the +original overload failure; native replay does not erase that CI evidence. + +### Remaining delivery streams + +| Stream | Unfinished work | +| --- | --- | +| W2–W5 | Complete role/current-authority observation, ongoing intent supervision, remaining failure/concurrency coverage and reliable count/pressure convergence. | +| W6 | Remaining primitive maintenance fault matrix, Blob external owners and sustained-traffic evidence. | +| W7 | Complete reader/follower replacement and failed-owner evidence, role evacuation, terminal action handoff and checked stop/withdrawal. `SettleRoles` and `Finalize` remain refused. | +| W8 | Runnable complete maintenance and receiver-loss scenarios. | +| W9 | Full process/provider fault campaign, soak and actual mixed-version qualification. | +| W10 | Exercised runbooks and staged rollout evidence. | + +The complete implementation plan remains active. This checkpoint does not +complete W1–W10 or establish fleet operation qualification. + +## October 2 2026 actual successor inspection checkpoint + +Fresh movement inspection can now target another physical node, or a new boot +of the preferred node, after the journal proves the exact clean source release. +The additional endpoint cannot authorize an effect or establish failed-source +recovery. Source aliases, mismatched preferred session/node pairs, insufficient +positions, changed requests and stale journal/registry barriers remain refused. +Original effect acceptance, replay and cleanup endpoints are unchanged. + +The reconciler first checks the preferred or retained serving endpoint directly. +If that check is unresolved, a bounded authenticated Cell observation under the +rechecked durable roster selects an established current boot. Its native actor +query and authority reads supply serving evidence. The original endpoint error +is retained when fallback succeeds. Retirement checks the actual successor; +unused receiver credit stays charged until its original cleanup proof joins it. +Partial role coverage supplies no absence or finalization proof. + +The public node regressions use independent resource budgets and actual +ordinary acquisition, authority, restored SQL and native inspection. They cover +another node and another session on the preferred physical node, reject Idle or +closed actors, and refuse activation effects on the alternate endpoint. The +second case qualifies session separation in the local API; it does not simulate +process death or qualify the durable reboot/enrollment protocol. +The durable example uses three leased nodes and the SQLite journal. It adopts +an ordinary winner, proves receiver cleanup before permit retirement, resolves +the original receipt and reads its original SQL value. No acquisition effect +is accepted on the alternate node and authority advances once. + +The corrected parent-code regressions fail at the valid successor request and +at the driver's missing adoption. Initial fixture failures and corrections, +original executables and sources are archived separately under +`/tmp/cellule-successor-inspection-1bd98d8-evidence`. An eager fleet scan also +failed the unchanged driver's observer-count assertion (five versus one); +fallback discovery preserves that assertion. A nested-if lint was corrected +without changing qualification profiles or assertions. + +Existing record bytes, indexes and action keys are unchanged. Older inspection +readers refuse the newly allowed endpoint shape. Deploy and qualify upgraded +readers before enabling it in a mixed-version fleet; this checkpoint adds no +mixed-binary or process/provider qualification. + +### Verification + +All 646 Rust/Cargo/lock paths match the active source and isolated snapshot +before and after every final command. +Manifest SHA256: +`5c92a99e5846aa5bbff3527a2224c0f24fced77735eae01976d282ff0eb67053`. + +| Final native scope | Result | +| --- | --- | +| Complete runtime library, all features | 518 passed; three existing ignores. | +| Complete host library, all features | 29 passed. | +| Complete public host node suite, all features | 106 passed, including both successor endpoint cases. | +| Complete fleet example, all features | 112 passed; required overload case failed. The new actual-successor scenario passed. | +| Workspace all targets/features, locked | Check and Clippy passed; lint warnings denied. Runtime/host API docs passed with warnings denied. | + +These are 765 distinct passed cases, three existing ignores and one failed +case. Focused repeats are excluded. The unchanged overload case proves one +successful movement and safely cancels the other, retiring both permits; it +fails the required two-release/two-activation result. Its original source, +executable and complete log are retained. No count, deadline, pressure profile +or assertion was relaxed. The separate controller-restart case passes this +native run, which cannot erase its parent CI failure or establish reliable +convergence across runs. + +Parent `1bd98d8` CI passes contract, MSRV, fuzz, fast models, website, object and +follower qualification and Compose smoke. Its workspace job passes 111 fleet +example cases and fails the controller-restart case with an unresolved Releasing +attempt after reply loss. Routing comparisons were pending at inspection. +The failure log is retained; its cause is not established by this change. +Reliable overload/controller-restart convergence, complete role/authority and +replacement-policy barriers, automatic intent supervision, Blob external +owners, fleet maintenance/failure scenarios and the complete deployment/fault +matrix remain open. `SettleRoles` and `Finalize` remain refused. Full W1–W10 +is incomplete. + +## October 2 2026 Cron maintenance checkpoint + +The public primitive regression in +[`blob_cron/maintenance`](../crates/cellule-runtime/tests/primitives/blob_cron/maintenance/mod.rs) +now exercises three independently budgeted runtimes, real SQLite, authority +CAS, immutable publication and signed Effect delivery to the compiled target +command. Its source Tick uses the public scheduler inside an accepted native +command transaction with pinned logical time and an explicit worker gate. +The barrier closes typed Tick and Effect claim admission while that transaction +is running. The successor's Ticks use the registered maintenance command. + +| Boundary | Checked result | +| --- | --- | +| Accepted source Tick | Its occurrence and durable outcome survive quiescence; original request resolution succeeds before release and after restoration. | +| Source release | Canonical maintenance release returns the authority's exact Idle root and closes the original actor. | +| Successor acquisition | Serving authority uses the successor session, increments the epoch, and retains the released root. | +| Due work during quiescence | The unfired schedule retains generation, occurrence and due time; a stale Tick generates nothing and a current Tick generates its first occurrence. | +| Retried Tick and delivery | Tick replay returns its original receipt. Two deliveries of each Effect return the same destination receipt and leave exactly one application row per occurrence. | +| Shutdown | All three runtime ledgers return memory, descriptors, disk, retained bytes, jobs, slots and unpublished log bytes to zero. | + +The fixture initially failed compilation on codec usage, then rejected its +application table's protected `cron_` prefix. A later zero-disk assertion +correctly observed reservations of other live runtimes through the convenience +Host's process-wide budget. Each simulated node now has an explicit independent +disk budget; the zero-resource assertions remain intact. Earlier sources and +failure evidence are retained separately from the final source under +`/tmp/cellule-cron-maintenance-9ed469c-evidence`. + +PR 37 is mergeable. Parent `9ed469c` CI's workspace job still fails the +required two-movement overload scenario: 111 example cases +pass, one movement succeeds and the other is safely cancelled. This checkpoint +adds Cron qualification without changing production scheduling, pressure, +admission or release rules. Reliable overload convergence, the remaining +primitive/fault matrix, Blob external owners, full role settlement and node +finalization, executable maintenance/failure scenarios, and process/provider +qualification remain open. Full W1–W10 is incomplete. + +### Isolated native verification + +All 643 Rust/Cargo/lock paths match between the active checkout and isolated +snapshot before and after every command. Manifest SHA256: +`32bc2b69c178f9e8295b3fa187e5d976995eb52654761e4e3a475d0934c2057f`. + +| Scope | Result | +| --- | --- | +| Complete public primitive suite, all features | 50 passed; none ignored or filtered. | +| Complete public protocol suite, all features | 30 passed; two existing provider diagnostics ignored. | +| Public runtime maintenance suite, all features | Two passed; 202 other runtime cases unselected. | +| Focused Cron replay, all features | Passed; excluded from the 82 distinct passed cases above. | +| Workspace all targets/features, locked | Check and Clippy passed with lint warnings denied. | + +The initial corrected fixture also passes its focused default-feature run. +These results establish the selected native public behavior; they add no +process, provider, mixed-version or fleet-wide maintenance qualification. + +## October 2 2026 exact source refusal checkpoint + +The unchanged overload scenario reproduced its unknown-release failure on +focused replay 12. Native local eviction selected partition 4 at +`1790959690253`; fleet Release was accepted at `1790959690255` and returned +Unknown with the original `CellNotActive` at `1790959690256`. The original +source, two tagged diagnostic probes, trace, identity derivation and frozen +executable are archived under `/tmp/cellule-pressure-race-5d350d1-evidence`. +The diagnostic executable SHA256 is +`2db7c5baa947dfda20d73fab2443226d4d597505f2c0555b43a797da51401dc4`. +The separate unchanged full diagnostic scope passed all 112 cases; that pass +cannot invalidate the reproduced race or qualify the final source. + +The canonical actor now identifies exact-position release refusal before this +request reaches deactivation using `Error::CellReleaseRefused`. It retains the +original validation, admission or inventory cause. Canonical close, authority +publication and response-loss errors remain uncertain. The host records a +Rejected outcome and the underlying original error, allowing existing +release-refusal transitions to join unused receiver credit. An independent +local eviction cannot count as this attempt's Released or Activated evidence. +Neither the automatic pressure classifier nor its movement budget changes. + +A deterministic public-host regression prepares actual receiver credit, joins +canonical source eviction and then submits the exact fleet release. Before the +fix it failed with Unknown instead of Rejected; the original executable SHA256 +is `2aa151c5a4e75382266fb2e110a2ee7898deb36795f0c7da77c49171f2ed8a6e`. +The qualified regression source and failure are archived separately from an +initial compile-only fixture error (comparing the opaque VersionedControl). +After the fix this regression passes, preserving source authority and joining +all receiver credit. A second public regression proves that prior journal +acceptance without its original result remains Unknown after independent local +eviction, with receiver credit charged. Both focused cases pass. + +### Final isolated native verification + +All 641 active/snapshot Rust/Cargo/lock paths match before and after each final +command. Manifest SHA256: +`4e84a44ffa02ed0b7f55e21cedecffa015664b3a9e2d42a46ac38de4de47395e`. + +| Scope | Result | +| --- | --- | +| Exact idle-release integration | 6 passed; 198 other integration cases unselected. | +| Runtime library | 515 passed; 3 existing ignores. | +| Host library | 29 passed. | +| Complete public host node suite | 104 passed, including both new cases and the immediate zero-disk-credit assertion. | +| Complete fleet example | 112 passed; none ignored or filtered; 71.41 seconds. | +| Application integration | 39 passed; same 16 documented manual ignores. | +| Workspace all targets/features, locked | Check and Clippy passed; lint warnings denied. | +| Runtime/host API documentation | Passed with warnings denied. | + +These are 805 distinct passed cases; focused repeats are excluded. A separate +pre-assertion source run also passed, and remains archived independently. +Format/diff, boundaries/layout, 110 Rust snippets, 1173 Markdown links and the +28 SQL/peer contract assertions pass. Final sources, logs, complete manifests +and executables are retained. No Linux, provider/process or mixed-binary +qualification was added in this checkpoint. + +Parent `5d350d1` CI still failed the original overload requirement with 111 +passes and one successful movement. Its object-capacity repeat 3 also failed +`skewed: no fully served capacity point`; the driver completed 58 of 60 planned +requests at that point. Original CI logs and the complete object-capacity +artifact are retained. Neither failure is erased by the positive native run; +capacity qualification and reliable overload convergence remain required. + +This checkpoint fixes refusal classification. It does not complete the required +two successful movements in the overload scenario: preparation can also refuse +an invalidated source, and the scenario disables new scheduling after the first +batch. Required movement/readback counts, deadlines, native pressure behavior +and qualification profiles remain unchanged. Full W1–W10, role settlement, +fleet finalization, executable maintenance/failure scenarios and deployment +qualification remain open. + +## October 2 2026 retained whole-host drain checkpoint + +The host now retains the complete canonical closing attempt in one fixed slot. +Shutdown transfers the existing drain-lane guard to that task before awaiting +its result. Cancelling the only caller cannot abandon an accepted facility +callback, release the lane early or prevent autonomous completion. The terminal +scale-down step uses this same owner and joins the original epilogue even when +it already observes Stopped. Native runtime shutdown is still invoked once. + +The existing reverse facility order, work cancellation, runtime close and lease +withdrawal phases remain canonical. A phase deadline leaves Draining; a later +attempt resumes the same underlying resource owners. Original task failures +remain fatal. Local drain observations retain original return/join results and +first/latest source errors; they do not establish role coverage or authorize +fleet completion. SettleRoles and Finalize remain refused. + +Three public regressions cover sole-waiter cancellation, scale-down queued past +its deadline on the original lane, and original failure history after retry. +The earlier pre-fix cancellation regression failed because the accepted +facility callback was abandoned. Its source manifest SHA256 is +`f1657e7eb530c26ac45d99401f20c69cd86f3a4e8af48b58475c70a13aaeee19`. +Original test source, log and executable are archived. + +The native-retirement fixture now pauses after the original supervisor joins, +before automatic host completion clears weak rotation receipts. All original +retirement, interruption, single-close, withdrawal and exact-readback assertions +remain. It additionally verifies joining the same host attempt and invalidation +of the weak request after Stopped. An earlier fixture version failed the public +suite, with 100 cases passing and one failing; its source and logs are preserved. + +Two synthetic controller tests retain their original 100 ms and 300 ms budgets, +starting the injected timeout at the observed first acceptance. A paused model +clock keeps unrelated SQLite setup and journal rereads from injecting an extra +fault. Original Elapsed, phase, permit and healthy-sibling assertions remain. +The frozen original Linux example replay reproduced the first-endpoint failure; +its full scope passed 109 cases and failed that case, count convergence and +controller restart. This replay is separate from final-source verification. + +### Verification scope and outstanding overload failure + +Complete source-set and byte checks cover all 641 active/snapshot/staged +Rust/Cargo/lock paths before and after each final command. Manifest SHA256: +`4377077c0762a8422f73642fdbdced6b6d105c0ce70d50656ff53c9dcedb33a6`. + +| Complete isolated native scope | Result | +| --- | --- | +| Host library | 29 passed. | +| Public host node suite | 102 passed; none ignored or filtered. | +| Fleet example | 111 passed; overload failed; 69.10 seconds. | +| Application integration | 39 passed; the same 16 manual ignores. | +| Host all-target/all-feature Clippy and API docs | Passed with warnings denied. | + +These scopes contain 281 passed cases and one failure; focused repeats are +excluded. Format/diff, boundaries/layout, 110 Rust snippets and 1173 Markdown +links pass. This is a closing-owner checkpoint, not successful fleet qualification. + +The pinned Rust 1.97.1 Debian 12/aarch64 runner uses two CPUs, four GiB, 256 PIDs, +and the existing descriptor hard limit of 524288. Only its inherited descriptor +soft limit is raised from 1024. Its application/host default-feature command +passes: application 10 library, three contracts, 39 integration with the same 16 +manual ignores, host 29 library, 102 public cases and five application doctests. +The full example passes 111 cases and fails overload, in 138.79 seconds; both +corrected controller models pass. Complete source checks pass at every boundary. +The runner exits 101 without OOM. A launcher source-copy attempt failed the +preflight before Cargo; its log is retained separately. + +The parent merge `4cc8ed3` also fails the full example in +[the Rust workflow](https://github.com/crabbuild/cellule/actions/runs/37030073904/job/110914390958): +111 passed and overload failed, in 79.70 seconds. It settled one movement while +retiring two attempts. The current native failure has the same counts; the Linux +failure leaves one original Releasing attempt unknown after 12 passes. No +required movement count, qualification profile or deadline has been weakened. +The local pressure/observed-generation race remains under investigation; these +results do not prove its cause or repair. Complete role observation, terminal +action handoff, cross-session recovery and all remaining W1–W10 work stay open. + +## October 2 2026 PR 37 merge conflict resolution + +The branch integrates main `0f4ca09` while preserving its retained maintenance +and shutdown ownership. Four conflicts involved actor admission, replica hint +fixtures, task supervision fixtures and the split durability module. + +- Transfer admission carries the upstream admitted owner fence and the existing + maintenance preflight record. +- Expiry-driven rotation uses the existing bounded supervisor claim. After + provider I/O, the claim is rechecked so accepted maintenance remains strict. + Fleet-bound recruitment delegates the same provider membership-read hook. +- Publication-hint tests retain their paused-clock guard, original bounds and + assertions, alongside upstream recruitment and directory diagnostics. +- The upstream rotation fixture uses the canonical retirement observation and + the same signed session as its host. The original deadline and close/recruit + count assertions remain unchanged. + +Verification uses an isolated snapshot with complete source-set and byte checks +before and after each command. Its 638 Rust/Cargo/lock paths have manifest SHA256 +`c1393c29486a7bc1bffcdf61e88db8cc3a5f6110abe4d9c4fa3d46d4fe62f784`. + +| Complete native command scope | Result | +| --- | --- | +| Workspace check, all targets and features, locked | Passed. | +| Runtime library, all features, locked | 515 passed; three existing ignores. | +| Runtime admitted-owner-fence integration filter | Three passed; 201 unrelated cases filtered. | +| Host library, all features, locked | 29 passed. | +| Public host node suite, all features, locked | 99 passed; none ignored or filtered. | +| Fleet example, all features, locked | 112 passed; none ignored or filtered; 68.65 seconds. | +| Application integration, all features, locked | 39 passed; the same 16 documented manual ignores. | + +These scopes contain 797 distinct passed cases. Host Clippy across all targets +and features and host API documentation pass with warnings denied. Format, +boundaries, module layout, 110 Rust snippets, 1173 Markdown links and the SQL/peer +validator's 28 assertions and 567 links pass. + +Initial compile failures and +an invalid-session fixture run are archived separately; they do not qualify +this source. Merged-source Linux, process/provider, mixed-binary and complete +W1–W10 qualification remain open. Prior checkpoint evidence below retains its +original source scope; earlier green results do not qualify this merge. + +## October 2 2026 retained node task joins checkpoint + +The parent `0132b60` task group removed handles from its bank before joining. +A failed first shutdown consumed the work failure. The next shutdown could +find no failure, withdraw lease maintenance and report Stopped. A new public +node regression reproduced that false success before the production change; +its first shutdown failed and its second returned Ok. The complete 637-path +before-source manifest SHA256 is +`6d53399d1ac48d31852cfdcda62bc77dc4d457bf9348f8c304e8fbc1f439ea77`. +The original failure, test source and executable are archived. An initial +compile-only attempt needed an explicit test-helper reference lifetime; it is +retained separately and supplies no behavioral evidence. + +The existing 256-task supervisor now retains every handle and terminal result +in its original bank. Concurrent callers share the exact join. Cancelling a +caller drops its waiter; no accepted task ownership disappears. After joining, +the first and later drains expose the same original error object through the +source chain. Sibling joins continue after a task failure. Node shutdown still +closes the runtime but stays Draining with lease maintenance retained when a +required task fails; a retry cannot erase that failure. + +Ordinary deadline abortion now targets the work task, retaining the supervisor +until it joins that work's cancellation and destructor. The cancelled work's +original JoinError remains on later drains. Group destruction still aborts its +remaining owned tasks; task admission and the 256-slot bound are unchanged. + +### Native facility watcher and resumable deadlines + +An intermediate implementation retained all errors but continued aborting the +node-log facility watcher on deadline. The full public host suite rejected it: +96 cases passed and two existing native-cleanup deadline cases failed on retry +with the watcher's stored cancellation error. Its source manifest SHA256 is +`fe4d15f94d557a466fc243797bcbe9d0de388f7954228ab5869e58cc09901911`. +Those logs and source are archived and do not qualify the final change. + +The watcher now uses a private retained-work registration in the same bounded +supervisor. Its deadline drops the waiter while the existing facility continues +to own the native supervisor; it does not abort that watcher. A later shutdown +joins the same native cleanup and original watcher result. Genuine native or +watcher failures still remain errors. No second scheduler, authority, public +configuration or alternative cleanup path was added. The existing deadline, +cleanup, withdrawal and zero-resource assertions remain unchanged. + +Five public regressions cover repeat shutdown without premature withdrawal, +cancelled waiters, concurrent joins, exact deadline cancellation and original +panic retention. They check original error identity and sibling completion. +The corrected source also passes both previously failing native-cleanup cases. + +### Verification scope + +Final native commands use an isolated source snapshot, all features and locked +dependencies. Complete source-set and byte checks run before and after each +command; all 637 Rust/Cargo/lock paths match final manifest SHA256 +`6e921d09fca19a8e91438e1bc078e315ff42847252691cb21dffc13a49184db4`. +The snapshot reuses its existing checkout-specific target directory beneath +`$HOME/Workspace/crabbuild-target`; previous qualified evidence and frozen +executables remain archived independently. + +| Complete final native scope | Observed result | +| --- | --- | +| Host library | 29 passed. | +| Public host node suite | 98 passed; none ignored or filtered. | +| Fleet operations example | 112 passed; none ignored or filtered; 67.98 seconds. | +| Application integration | 37 passed; the same 16 documented manual cases ignored. | + +These are 276 distinct native cases; the initial 16-case focused run is excluded +from that count. Host all-target/all-feature Clippy and API documentation pass +with warnings denied. Format/diff, workspace boundaries/layout, 110 documented +Rust snippets and 1173 local Markdown links pass. This source supplies no new +process, provider, mixed-binary or full fleet finalization proof. + +The isolated Linux ARM run exits zero with the same complete source checks +before and after each command: + +| Default-feature, locked Linux command | Observed result | +| --- | --- | +| `cargo test -p cellule-app -p cellule-host --locked` | Application library 10, contracts 3, integration 37 (the same 16 manual ignores); host library 29, public node 98; five application doctests. Host doctests contain zero cases. | +| `cargo test -p cellule-host --example fleet_operations --locked` | All 112 cases pass; none ignored or filtered; 81.35 seconds. | + +Environment: pinned Rust image, Debian 12/aarch64, Rust/Cargo 1.97.1, two CPUs, +four GiB and 256 PIDs. The inherited descriptor soft limit is raised from 1024 +to its existing 524288 hard limit, matching the earlier frozen-binary descriptor +diagnosis. No task, scenario, deadline or qualification bounds change. A launcher +attempt failed before any Cargo command because the copied script lacked its +shell interpreter line; its separate log/container record is retained. Adding +that line changed only the launcher, and the complete subsequent commands +supplied the results above. Logs, final sources, manifests and frozen native/ +Linux executables are archived independently. The task-owned container is +removed only after terminal status and evidence are captured. + +### Parent CI and remaining scope + +The parent qualification contract, coordination model, website, fuzz and +capacity workflows passed. Its Rust MSRV and workspace test steps passed, +but [the complete fleet example step](https://github.com/crabbuild/cellule/actions/runs/37016038610/job/110867015361) +failed: 111 cases passed and the overload scenario settled one movement rather +than its required two. The raw failure log SHA256 is +`f0f5be0fa5e3a133b5880143e973887eb9db547abc21ae69f60484651caec3ea`. +This checkpoint's complete local example pass does not establish that failure's +cause or claim it repaired. The historical x86 publication-hint cause also +remains unestablished. Neither failure changes the required profiles, resource +bounds, scenario counts or deadlines. + +This is a W4/W7 shutdown prerequisite. Complete authenticated role/current- +authority aggregation, replacement/failed-owner evidence, role evacuation, +terminal finalization handoff without self-join, cross-session recovery, +Cron/Blob owners, maintenance/failure examples, qualification and rollout +remain required. SettleRoles and Finalize remain refused. The full W1–W10 +implementation objective remains active. + +## October 2 2026 reader reconciliation lease-boundary repair checkpoint + +The parent `8698183` qualification contract failed during reader fixture startup, +before its native-opening scenario: `install_node_lease` returned +`Control("CellNode task group is unhealthy")`. Its workspace/MSRV campaign +passed, so that success did not supersede the separate contract failure. +The [failed contract job](https://github.com/crabbuild/cellule/actions/runs/37009635155/job/110846004625) +and original log are retained; the log SHA256 is +`1e6a9dfda287a5ecba829b02e0763ddfb4c8036ae363d86f7adb6cd74ea838df`. + +The new periodic temporary inventory used `try_reserve_node_bytes`, whose +live-lease check returns Fenced before lease installation. The immediate first +tick could therefore end the installed reader task before host startup. Thirty +unchanged native repetitions of the original closure case passed; those negative +reproductions did not disprove the ordering race. A controlled public loop +before lease installation then failed with the original Fenced source. +A second real producer case fenced the local lease after a lost acceptance +reply; the old reservation prevented periodic exclusion for two real ticks. +Both failures are archived, with their source manifests and executables. + +Both temporary inventory paths now use the existing metadata reservation API, +including the bound producer's retained-record index. This charges the same +bounded byte ledger without granting native admission or lease authority. +Native opening, activation jobs and public native inventory retain their live +lease checks. The activation lane, five-second interval, 64-attempt batch, +30-second attempt deadline and all resource bounds remain unchanged. + +The new public host case confirms healthy startup after the pre-lease loop and +joined shutdown. The new bound-producer case observes the original Refused +exclusion after fencing, preserving its spec and acceptance time; it also checks +fenced inventory, refused activation, no native opening and zero final resources. +Neither local result supplies failed-process closure or replacement policy. + +### Snapshot retry test synchronization + +The first Linux application/host run reached the previously failing reader +closure scenario successfully, then failed a separate snapshot case: it counted +four authorizations instead of two. The fixture spawned a retry and used +`yield_now` before releasing the original blocked job. A yield did not guarantee +that the retry had reached the retained job. If the original completed and was +joined first, the later retry could correctly begin another fresh capture. + +A diagnostic copy delayed only that retry by ten milliseconds and reproduced +the same four-versus-two failure. The fixture now explicitly polls the same retry +to Pending while the original job is blocked, before releasing it. The focused +case passes. Its exact two-call assertion, original request/page, authority +comparison, retained-byte checks and shutdown checks are unchanged. No snapshot +production code, deadline or qualification profile was changed. Diagnostic delay +code lives only in the evidence archive. + +### Deadline model fault boundaries + +The next Linux application/host command passed, including all 93 public host +cases and the reader-closure/snapshot-retry cases. Its complete example run +passed 110 cases and failed the existing accepted-endpoint timeout case: the +first Prepare acceptance row was absent. Its 150 ms real-time pass could expire +before acceptance under load. A diagnostic copy delayed only acceptance by +160 ms and reproduced that same missing-row assertion. Those failures remain +archived and rejected. + +The timeout is a model fault injection, not a measured SLO. That current-thread +model now pauses the test clock after setup, drives each of the two sequential +endpoints to confirmed acceptance, then expires its original deadline share. +The pass deadline remains 150 ms; the first endpoint uses one third and the +second uses half of the remaining budget, as the public driver does. One timer +resolution tick permits delivery of the Elapsed event. Both original deadline +sources, two retained attempts, 8 KiB restore permits, Preparing phase and +accepted-without-result assertion remain required. + +A separate public driver case pauses each endpoint before acceptance and proves +both Preparing permits remain charged, both acceptance rows remain absent, no +effect was dispatched, and no release occurred. It preserves the original +Elapsed sources. Host dev dependencies explicitly enable Tokio test-clock +utilities; product configuration, runtime deadline behavior, qualification +profiles and native/process SLO clocks are unchanged. A first helper assumed +concurrent endpoint waits and correctly failed its gate; it was corrected to +the driver's actual sequential partitions. A relative-copy diagnostic invocation +also failed to overlay the intended test source; it is archived and excluded. + +### Qualification scope + +Final source manifest: all 637 Rust/Cargo/lock paths, SHA256 +`8df5a9f2a4929493142ac67e0b860965580308fddcdf93141de1c962c5c4ef4e`. +The isolated driver checks complete active/isolated path sets and every byte +before and after each final command. Final results and original failures are +archived under `/tmp/cellule-reader-prelease-evidence`. + +| Final isolated native scope | Passed | Exceptions | +| --- | ---: | --- | +| Complete host library | 29 | None | +| Complete public host node suite | 93 | None | +| Complete fleet example | 112 | None | +| Complete application integration | 37 | 16 documented manual cases ignored | +| Runtime reader integration scope | 14 | 186 filtered; one documented RustFS case ignored | + +285 distinct native cases pass; focused diagnostic repetitions are excluded +from that count. Native commands use all features and locked dependencies with +Rust/Cargo 1.97.0. Host/runtime all-target/all-feature Clippy and API docs pass +with warnings denied. Format/diff, boundaries/layout, 110 Rust snippets, 1173 +Markdown links, 28 SQL/peer assertions and 567 validator links pass. + +The final Linux ARM run passes the exact application/host CI command with +default features and locked dependencies: application library 10, contracts 3, +integration 37 (the same 16 manual ignores), host library 29, public node 93, +and five application doctests. The complete default-feature example passes all +112 cases in 88.83 seconds. No case in the host/example suites is ignored or +filtered. Complete Linux source sets and bytes are checked before/after both +commands. The container terminates zero with no OOM. Its environment is Debian +12/aarch64, Rust/Cargo 1.97.1, the pinned Rust image, two CPUs, four GiB and 256 +PIDs. The inherited descriptor soft/hard limits are 1,024/524,288; only the soft +limit is raised to that existing hard limit, as documented by the earlier +frozen-binary descriptor diagnosis. No source/profile/Cell-count or resource +assertion is changed for this environment. + +The earlier source manifests +`4391e7ed6b8a5696e82134d320a06078db6d228a6da0cd968eca740437b06eff` and +`a042f610b8ccce4922781eb8bafe084cacadf51ba09f60a25f21d6cf80370fb7` +passed their native scopes but retained the respective snapshot/timer fixture +races. Their logs, source versions and frozen executables are preserved +separately and do not qualify the final manifest. Both failed Linux runs remain +rejected. The corrective run uses the same pinned Rust image, two CPUs, four +GiB, 256 PIDs and documented descriptor headroom. + +Full W1–W10 remains active. Complete role/current-authority observation, +replacement and failed-owner evidence, evacuation/finalization without +self-join, cross-session recovery, Cron/Blob owners, runnable maintenance and +receiver-loss scenarios, fault/mixed-binary qualification and rollout remain +required. SettleRoles and Finalize remain refused. The historical publication +hint failure has a separate, unestablished cause and is not claimed fixed here. + +## October 2 2026 periodic reader producer repair checkpoint + +The canonical reader manager's existing five-second loop previously scanned +only installed views. A lost acceptance before opening had no view and therefore +received no periodic repair. Selected views also retained an unconfirmed opening +result until another hint, and a cancelled removal could leave a fenced view +whose refresh could never succeed. + +The same loop now scans the sorted union of managed views and retained producer +requests. Its original 64-attempt batch and 30-second per-attempt deadline remain. +The two original indices are bounded independently; their captured sizes are +charged before copying, with the charge retained through the batch. Capacity +refusal preserves responsibilities for a later tick. No additional scheduler, +activation path, journal, configuration or wire format was introduced. + +Repair takes the original activation lane. A still-owned opening cannot be +classified from a missing view. After that owner releases the lane, canonical +retirement atomically excludes a never-started request or publishes the joined +native refusal. An unjoined native failure remains blocked. Selected open views +replay the original Established event before refresh. Already-fenced views resume +the same closure and Retired publication before remote reads. Finished finite +jobs are reaped from their existing bank; original failures remain observable. +Activation also joins its exact task before returning, so its completion implies +release of the finite-job byte token. A dropped join waiter leaves the same job +owned; a failed task returns its original join error before the derived channel +closure error. + +### Scoped verification + +The old loop fails the new lost-acceptance case after two real periodic ticks; +its original Pending request never settles. The initial build selected zero +cases with a short exact filter; that is excluded from evidence. The corrected +fully qualified case ran and failed. Before/after source bytes and the complete +636-path baseline set are retained with the original failing executable. + +Five focused cases now pass: lost acceptance, lost opening result, cancelled +removal, a still-owned cancelled-waiter opening, and temporary inventory-credit +refusal. The latter retains the request through a refused tick and repairs it +when credit returns. Checked replay keeps the original spec, acceptance time, +settlement event and error Arc. Readback and joined cleanup remain required. + +| Final isolated scope | Result | +| --- | --- | +| Complete host library / public node / fleet example | 29 / 92 / 110 passed; none filtered or ignored. | +| Complete application integration | 37 passed; none filtered; 16 documented manual cases ignored. | +| Runtime reader integration scope | 14 passed; 186 filtered; one documented RustFS case ignored. | +| Distinct native cases | 282 passed; five new periodic repair cases. Focused repetitions are excluded. | +| Host/runtime all-target/all-feature Clippy and API docs | Passed with warnings denied. | +| Static gates | Format/diff, boundaries/layout, 110 Rust snippets, 1173 Markdown links, 28 SQL/peer assertions and 567 validator links passed. | +| Linux ARM complete fleet example | 110 passed; none filtered or ignored; 82.35 seconds, with the documented descriptor headroom. | + +All final native commands used all features and locked dependencies. An earlier +complete pass before tightening the activation task join is archived separately; +its results cannot qualify the final source. + +The first two-CPU/four-GiB Linux run inherited a 1,024-descriptor soft limit and +failed while creating a source Cell in the existing 128-reader long-partition +case: 109 passed, one failed with OS error 24. A parent-production control also +failed at that same source-bootstrap boundary, with SQLite CannotOpen. The +unchanged parent executable passes its selected case when only the soft limit +is raised to the existing 524,288 hard limit. The unchanged final executable +then passes all 110 cases under that same headroom. Original failures are +retained; the low-limit run remains rejected. No source, Cell count, memory +ceiling, deadline, qualification profile or expected assertion changed. + +Both controlled runs check their entire source sets and executable hashes +before/after. Linux used the pinned repository Rust image, Rust/Cargo 1.97.1, +Debian 12 aarch64, default features, locked dependencies, two CPUs, four GiB and +256 PIDs. Its final executable SHA256 is +`46aadad9901d79c0e4476751da92b7d4085b30ce87ae64690b9d8d1aec3e751f`. +The 636-path parent-production control uses its own manifest and frozen binary. +Both containers were removed only after terminal results and evidence retention. + +All final native commands compare the active/isolated complete 637-path +Rust/Cargo/lock set and every byte before and after execution. Source manifest: +`14669eae88f753561e4873eaa40798fc5625f5eccaaaad0786f82303c23cf274`. +Before/final logs, sources, comparison driver and executable hashes are retained +under `/tmp/cellule-reader-reconciliation-evidence/` and the matching +`/tmp/cellule-reader-reconciliation-final-*` files. + +### Remaining scope + +Parent `d5eebffe05b6d9514861d85f54d0d1f7c404e083` passed workspace/MSRV, +qualification contracts and both capacity campaigns at inspection. Compose +qualification and fresh CI for this change remain required. The historical x86 +publication-hint failure still has no established cause; this repair is not +claimed to fix it. + +Local producer repair does not reconcile failed processes, certify replacement +redundancy or supply current remote authority. Complete role observation, +reader/follower evacuation, finalization without executor self-join and every +remaining W1–W10 exit assertion remain required. SettleRoles and Finalize remain +refused by the host executor. + +## October 2 2026 canonical advertisement verification checkpoint + +Canonical advertisement decoding previously verified signatures and called the +storage encoder for exact-byte comparison. That encoder verified the same +immutable signatures again. Decode now checks shape and signatures, then uses +one private canonical serializer for byte equality. The storage encoder verifies +before calling that same serializer. Directory scope, liveness and signature +checks remain independent; there is no cached authorization or changed wire +format, signing domain, lease budget or recruitment scheduling path. + +A deterministic thread-local test counter measures verification passes without +wall-clock thresholds or parallel-test interference. The regression first failed +on the old decoder with two passes instead of one. After the fix, legacy, +schema-2 and schema-3 decode each take one signature-set pass and reproduce the +original bytes. Additional cases retain exact signature errors for corrupted +identity and understood placement signatures, reject noncanonical whitespace +and reject oversized input before verification. The first broad run rejected a +new fixture that inserted an unknown root field; it now corrupts the actual +`identity.signature`. The typed error assertion was retained. + +### Scoped verification + +| Evidence | Result | +| --- | --- | +| Complete isolated runtime library | 510 passed; none filtered; three documented provider cases ignored; 89.40 seconds. | +| Complete runtime fleet integration | 25 passed; none filtered; one documented provider case ignored. | +| Runtime reader integration scope | 14 passed; 186 filtered; one documented RustFS case ignored. | +| Complete host library / public node / fleet example | 29 / 92 / 105 passed; none filtered or ignored. | +| Complete application integration | 37 passed; none filtered; 16 documented manual cases ignored. | +| Distinct native cases | 812 passed; three new canonical codec cases. Diagnostics and second-platform repetitions are excluded. | +| Host/runtime all-target/all-feature Clippy and API docs | Passed with warnings denied. | +| Linux ARM application integration | 37 passed; none filtered; 16 documented manual cases ignored; 22.28 seconds. | +| Static gates | Format/diff, boundaries/layout, 110 Rust snippets, 1173 Markdown links, 28 SQL/peer assertions and 567 validator links passed. | + +All final native commands used the isolated snapshot, all features and locked +dependencies. Before and after each command, the comparison driver checked the +complete active/isolated 635-path Rust/Cargo/lock set and every source byte. +Manifest SHA256: +`234d7c78aee2c5ce3aa257d242e7415200a63b01956fa470172f37c8fd768741`. +Logs, before/after source manifests, the original failing unit executable and +Linux executable are archived in `/tmp/cellule-signature-evidence/`. Final logs +and the source comparison driver are `/tmp/cellule-signature-final-*` and +`/tmp/cellule-signature-verify.py`. + +Linux verification used the pinned repository Rust image +`sha256:0e2bcaef56d041a486784e54104a81aebe0da44bd03019bd70bc0401e42e4a97`, +Rust/Cargo 1.97.1, default features, locked dependencies, two CPU credits and +four GiB. All manifest hashes were checked before and after its complete target. +The Linux integration executable SHA256 is recorded with the original logs; +its container was removed after retaining the evidence. + +### Publication-hint diagnosis and remaining work + +The original x86 contract failure remains open. Twenty byte-checked isolated +macOS repetitions and one two-CPU Linux ARM repetition of the original failing +case passed before this change. Those are negative reproductions, not proof of +a cause. The Linux ARM full-target pass after the change likewise cannot +establish the x86 failure's cause. Ranked hypotheses remain hint coalescing, +synchronous verification cost, membership/lease rejection and delayed bounded +queue progress. The confirmed duplicate verification is removed; no assertion, +profile or expected evidence was weakened. + +Previous head `4dc71df0df4117ca48c1c6aa313f6fbdf8e3f0c2` passed workspace/MSRV, +contract, both capacity campaigns, website, decoder fuzz and fast/negative TLC. +Its Compose smoke and routing comparisons were in progress at inspection. +Root-capture head `284dd8c` subsequently completed both routing comparisons and +the aggregate routing job successfully. Broad TLC/simulator remain skipped. +Earlier failures stay in their own execution records. Fresh CI for this source, +full role/current-authority observation, replacement/failed-owner evidence, +maintenance settlement/finalization and the remaining W1–W10 work are required. + +## October 2 2026 reader continuation coherence checkpoint + +Managed reader continuation now fingerprints all native reader observations, +including entries outside the returned page. The fingerprint binds the exact +manager/session, sorted receipt positions and incarnations, canonical lifetime +counts, reader admission closure, snapshot attachment, manager closure and node +admission. Each reader is captured once and that same observation is copied if +selected for the page. A changed node admission during capture rejects the page. +The original activation lane, 10,000-view bound, 128-row page limit and one-MiB +native reservation remain in use; there is no new task, provider call or cache. + +Previously the continuation depended only on the manager's topology UUID and +session. A retained peer can close, detach or refresh shared native state outside +that manager lane. The new public regression first failed on the old code: +continuation still succeeded after both peer closure and detachment outside the +first page. It now rejects those mixed scans. Two further cases observe native +query start/completion and an exact-root refresh beyond the returned page. They +check successful fresh traversal, SQL value and position, retained peer handles +and empty native resource ledgers after shutdown. The existing inventory case +also requires restart after cordon. + +Fingerprints describe intervals, not an atomic whole-node state. The native query +case deliberately verifies that an open lifetime count can return to its old +value after completion, restoring the fingerprint. Matching hashes cannot turn +an open reader into joined. The irreversible closed/detached/zero predicate, +current authority, durable enrollment, replacement policy and complete host +facility joins remain separate requirements. + +### Scoped verification + +| Evidence | Result | +| --- | --- | +| Lifetime CAS unit / runtime reader scope | 1 / 14 passed; 509 / 186 filtered; one documented RustFS case ignored. | +| Complete host library / public node / fleet example | 29 / 92 / 105 passed; none filtered or ignored. | +| Complete application integration | 37 passed; none filtered; 16 documented manual cases ignored. | +| Distinct scoped cases | 278 passed; three new continuation cases. Selected diagnostics are excluded. | +| Host/runtime all-target/all-feature Clippy and API docs | Passed with warnings denied. | +| Static gates | Format/diff, boundaries/layout, 110 Rust snippets, 1173 Markdown links, 28 SQL/peer assertions and 567 validator links passed. | + +All final native commands ran in the isolated snapshot with all features and +locked dependencies. The active and isolated complete 635-path Rust/Cargo/lock +sets and every byte were checked before and after each command. Manifest SHA256: +`3e46afe3bca79ffc8b45a4d00ef04e691a21f25f4bc11d00f1ed1f1552b58300`. +The original failing regression is `/tmp/cellule-reader-pagination-before.log`; +it was a scoped workstation run. Final logs, source comparisons and evidence are +`/tmp/cellule-reader-pagination-final-*`, +`/tmp/cellule-reader-pagination-source.sha256`, +`/tmp/cellule-reader-pagination-verify.py` and +`/tmp/cellule-reader-pagination-evidence/`. + +### Previous-head CI failure retained + +Head `9f7d3c6ee86850d6df79de476293a3a1198d92a0` passed workspace/MSRV, +website, decoder fuzz and fast/negative TLC. Its capacity and Compose campaigns +were still in progress at inspection. The +[qualification contract job](https://github.com/crabbuild/cellule/actions/runs/36999783770/job/110814598431) +failed: 36 application cases passed, one failed and 16 documented manual cases +were ignored. `publication_hints_reach_readers_beyond_the_activation_concurrency` +timed out at publication 3 after 2.000884718 seconds, receiving 14 of 19 healthy +hints with five missing sessions, no held peer and one intentionally pending +transport. The workspace pass and isolated passes do not establish its cause. + +The original failed-job log is +`/tmp/cellule-fleet-9f7d3c6-contract-failed.log`; SHA256: +`38a9f7767a428bd6bd780ce320196822abe6cf9f9c31ff081861c146b7c61670`. +The failure remains open. No bound, profile or expected evidence was weakened. +Fresh CI for the continuation change and complete W1–W10 work, including role +settlement/finalization and publication-hint diagnosis, remain required. + +## October 2 2026 native reader lifetime checkpoint + +`CellReadReplica::lifecycle_observation()` reads the original admission CAS word +and shared snapshot state. It reports the last installed receipt, irreversible +admission closure, snapshot attachment and retained snapshot/operation guards. +Accepted queries, native SQL and refresh work stay visible after their callers +cancel. Retained peer handles share the same closure. Closed admission, detached +state and zero guards establish local joining; the count is neither a query count +nor a count of handle copies. + +Bounded host reader pages now return these observations through `entries()`. +Callers use `entry.receipt()` for the previous position fields. Pagination, +topology checks, the 128-row limit and one-MiB native reservation remain in the +canonical manager path. The runtime reader module moves to `replica/mod.rs` so +its focused observation module follows the workspace layout contract. + +The public regression cancels an accepted native query, detaches shared state, +and observes its retained lifetime through both the peer and managed inventory. +Joining waits for that original callback; retained handles then reject new +queries, including after host shutdown. Existing regressions also observe old +snapshots, cancelled refresh opens, stalled authority reads and concurrent close +waiters. The private CAS case checks that closed zero cannot acquire another +operation or snapshot guard. Resource and error assertions remain required. + +### Scoped verification + +| Evidence | Result | +| --- | --- | +| New lifetime CAS unit case | 1 passed; 509 filtered. | +| Runtime reader integration scope | 14 passed; 186 filtered; one documented RustFS case ignored. | +| Complete host library / public node / fleet example targets | 29 / 89 / 105 passed; none filtered or ignored. | +| Complete application integration target | 37 passed; none filtered; 16 documented manual cases ignored. | +| Distinct scoped cases | 275 passed; two new cases. Diagnostic repetitions are excluded. | +| Host/runtime all-target/all-feature Clippy and API docs | Passed with warnings denied. | +| Static gates | Format/diff, boundaries/layout, 110 Rust snippets, 1173 Markdown links, 28 SQL/peer assertions and 567 validator links passed. | + +Every command used the isolated snapshot, all features and locked dependencies. +The complete active and isolated 635-path Rust/Cargo/lock sets and every byte +were checked before and after each command. Manifest SHA256: +`f820e068a457b6fddbd2b2ff70d06e4796afce9c6617b0ec84ac3dc104c7da93`. +Logs and source comparison drivers are `/tmp/cellule-reader-lifetime-final-*`, +`/tmp/cellule-reader-lifetime-source.sha256` and +`/tmp/cellule-reader-lifetime-{app-,}verify.py`. + +These local observations still require independent current authority, durable +producer retirement, replacement policy and full host drain evidence. They do +not enable SettleRoles or Finalize. Complete W1–W10 implementation and fresh +process/provider qualification of this source remain required. + +## October 2 2026 canonical root capture checkpoint + +The entity process driver now waits for canonical object roots after client jobs +and receipt readbacks join. Follower proofs may acknowledge commands before +object publication. One read-only barrier covers the entire original serving +roster under the existing two-second bound. It pins session and endpoint, epoch, +incarnation, code and schema, and preserves original read errors. Missing or +changed authority, an unbounded roster, stalled reads or stalled publication fail. +It does not start another publisher, rotate authority, or use shutdown as proof. + +The driver retains each Cell's latest acknowledged sequence across all windows +and emits actual canonical roots, independently checked minima, complete read +passes, elapsed duration and boot-clock bounds. Clock reads are included in the +duration and separately measured for the centisecond clock comparison. The +arrival windows, rate/concurrency ramp, response and arrival latency, overload +classification, receipt ledgers, original ownership, follower epochs and final +root coverage remain required. A barrier must finish within two seconds. See +the [capacity capture contract](../crates/cellule-app/performance/2026-09-29-write-capacity.md#canonical-root-capture-after-client-work). + +The native regression first failed the old one-read pattern with `published root +does not cover writes`, then passed after the barrier. It uses a real SQL command +and canonical authority, and reconstructs the captured root to verify both the +new table and durable request record. This observation test does not claim a +network follower-proof response. Seven other cases cover every original writer +binding, a shared deadline across four real Cells and read admission, no progress, +original typed read errors, dropped observation, closed/missing/successor +authority and invalid rosters. Each fixture has a private disk ledger and keeps +all zero-resource shutdown assertions. The initial parallel fixture run failed +because the default disk ledger is shared; the assertions were retained. + +### Scoped verification and remaining qualification + +| Evidence | Result | +| --- | --- | +| Complete isolated application integration target | 37 passed, none filtered, 16 documented manual cases ignored; eight new cases; 6.53 seconds. | +| Application all-target/all-feature Clippy | Passed with warnings denied. | +| Qualification verifier tests | 35 passed across entity and scaling verifiers, including seven new barrier cases. | +| Static gates | Format/diff, boundaries/layout, 110 documented Rust snippets, 1173 Markdown links, 28 SQL/peer assertions and 567 validator links passed. | +| Original failed follower artifact | All seventeen windows rechecked; the original root coverage gate still rejects Cells 9–11 at 330 versus acknowledged 331. | +| Historical passing follower artifacts | Their original roots still cover every independent acknowledgement; none supplies the newly required barrier records. | + +The unchanged 634-path Rust/Cargo/lock manifest has SHA256 +`724d50a3f8b1e52fd670211966e938dae994b6f38c94fca289db23c65a202122`. +Active and isolated path sets and every source byte were compared before and +after both Cargo commands. Logs and the comparison driver are +`/tmp/cellule-root-barrier-final-*`, `/tmp/cellule-root-barrier-source.sha256` +and `/tmp/cellule-root-barrier-verify.py`. Original regression and intermediate +fixture failures are retained in the task record. These are native checks; +fresh process/provider evidence for this change remains required. +The native integration binary SHA256 is +`cf83bd2249db7ae27865ff06c45c29231d0d50483dfa3e2b9913a8e0c46c8205`. +Verification used Rust/Cargo 1.97.0, all features, locked dependencies and a +separate Workspace target with `CARGO_INCREMENTAL=0`. + +### Root-capture head CI + +Head `284dd8cc9d70970a33926885337aecb5c087ebc7` passed workspace/MSRV, +contracts, both capacity campaigns, Compose smoke, website, decoder fuzz and +fast/negative TLC. Broad TLC and simulator were skipped. Both routing +comparisons remained in progress at inspection. These results qualify that head, +not the subsequent reader-lifetime change or the full fleet plan. + +Both capacity artifacts ran synthetic merge +`7147e9f3e08e69f26c4698a6fae4b49d368518de`, whose parents are +`191409685b001a82bd02780def45102b4fc2f164` and the root-capture head. +Driver/node binary SHA256 is +`f9695b5596bf2955178e07d06d1b6896dcb5a423a275bc96e7681929000fa0cd`. +The current independent verifier rechecked every returned field against each +original report, including all raw hashes, receipts, resources, original root +coverage and the new timing records. + +| Campaign | Repeat windows | Acknowledged writes | Barrier elapsed µs | +| --- | --- | --- | --- | +| [Follower capacity](https://github.com/crabbuild/cellule/actions/runs/36996826665/job/110805347208) | 18 / 18 / 18 | 6228 / 6230 / 6233 | 8560 / 8740 / 9053 | +| [Object capacity](https://github.com/crabbuild/cellule/actions/runs/36996826665/job/110805347410) | 17 / 19 / 19 | 5385 / 7400 / 7403 | 7619 / 9261 / 8642 | + +Each repeat captured all twelve original Cells in one pass under the unchanged +two-second limit. Follower repeats verified 6218/6222/6225 follower-proof +responses and 11316/11716/11720 network appends. Their one-pass captures do not +exercise waiting; the controlled native regression covers that case. These +passes do not establish the causes of historical publication-hint failures. + +Artifact 11222033521 retains 432 entries; ZIP SHA256: +`ba871f08d332bc74b7d449818c7dfce6959b568e653b34e6bf0e510f10c951bd`. +Artifact 11222293106 retains 435 entries; ZIP SHA256: +`89029d4efe01160eb7e4d3dfb6de9ef2f23fecd435964c7bd56010d63a445807`. +Archives and independent inspection reports are +`/tmp/cellule-fleet-284-{follower,object}-capacity*`. + +### Previous head CI + +PR head `aa12ad905aadd4189f83b4561334efb935a521e3` passed workspace/MSRV, +contracts, both capacity campaigns, smoke, website, decoder fuzz and fast/negative +TLC. Broad TLC and simulator were skipped; both routing comparisons remained in +progress at inspection. This does not establish the causes of earlier failures. + +Follower-capacity [job 110794579951](https://github.com/crabbuild/cellule/actions/runs/36993417592/job/110794579951) +ran three fresh-provider repeats on synthetic merge source +`096dfb2ff42016519d756c8a302ea22c90a515ea` (parents `191409685b001a82bd02780def45102b4fc2f164` +and `aa12ad9`). All four driver/node binary records match SHA256 +`ef8ba2ae24a51cb728c937e126e0b94e64ed7110afd06bf4e4cbf2c72fb96ac7`. +The repeats verified 10/9/9 windows, 2118/1443/1435 acknowledged writes, +2104/1424/1420 follower-proof responses and 4106/2826/2780 network appends. +Every recorded raw TSV hash was checked, and the current independent window and +root-coverage verifier rechecked each repeat. Artifact 11220794009 is retained +with all 351 entries; ZIP SHA256 is +`316c295e8e6f33b024354d77d2bff655aa75dbb57a1675dfc2f6bdbab3da3371`. +The archive and inspection are `/tmp/cellule-fleet-aa12-follower-capacity*`. +These historical passes cannot qualify the new barrier. Full W1–W10 work, +including maintenance role settlement/finalization, remains active. + +## October 2 2026 live intent refresh checkpoint + +`CellNode::refresh_fleet_intent` reads current physical intent and the original +Established boot through the existing atomic enrollment journal contract. Startup +now retains that exact enrollment record instead of a confirmation flag. Acceptance +time, immutable spec and establishment evidence cannot be replaced by another +Established request. Startup confirmation and live refresh share one application +path for the monotonic role gate, with lifecycle-before-startup lock ordering. + +The application drives refresh before membership/lease renewal. Cordon/drain +preserve existing owners and close new writer/reader/follower admission. Older or +contradictory replies fail, Active replies cannot clear a local cordon, and replies +after shutdown cannot reopen the node. Deadlines and cancellation drop only the +read waiter; journal errors retain their original source. These failures grant +no renewal authority. The framework starts no additional supervision task. + +The reference heartbeat calls this path before canonical directory CAS and guard +renewal. Its actual operational sample advertises Draining even without a Cordon +RPC. The real twelve-writer regression checks canonical advertisement equality, +unchanged original boot evidence, closed new-role admission, successful queries +to every existing actor, and joined shutdown with empty native resource ledgers. +Six further tests cover live intent, competing/delayed replies, shutdown, +deadline/backend errors, cancellation/sticky cordon and alternate boot requests. +The capture nonce uses the existing workspace `try_update` API; this removes the +Rust 1.99 deprecation without changing overflow behavior or the 1.97 minimum. + +### Scoped qualification + +The unchanged manifest contains 632 Rust/Cargo/lock paths with SHA256 +`5a6e98c4c10dd1ad9b6d8842d814db55be7f78819344d4d7989a0f66bce05deb`. +The driver compares both the complete path set and every byte in the active and +isolated checkouts before and after each command. + +| Evidence | Result | +| --- | --- | +| Complete host library target | 29 passed; none filtered/ignored. | +| Complete public host node target | 88 passed; none filtered/ignored. | +| Complete fleet example target | 105 passed; none filtered/ignored; 69.28 seconds. | +| Isolated complete application integration target | 29 passed; none filtered; 16 documented manual cases ignored. | +| Distinct scoped cases | 251 passed, including seven new cases. Diagnostic repetitions are not added. | +| Isolated standalone `balance` command | Exit zero; eight releases/activations/retirements and original receipt checks, shared in-flight maximum two, restore maximum 2,550,136,832 bytes, three joined/retired boots, two receivers, controller epoch one, final counts 4/4/4. | +| Host all-target/all-feature Clippy | Passed with warnings denied. | +| Host/runtime API docs | Passed with warnings denied. | +| Static gates | Boundaries/layout, 110 Rust snippets, 1172 Markdown links, 28 SQL/peer assertions and 567 validator links passed. | + +Logs and the comparison driver are `/tmp/cellule-live-intent-final-*`, +`/tmp/cellule-live-intent-source.sha256` and `/tmp/cellule-live-intent-verify.py`. +The earlier selected startup/scenario runs are diagnostics, not added coverage. +The isolated CLI binary, source manifest, output and metadata are retained at +`/tmp/cellule-live-intent-cli-evidence/`; binary SHA256 is +`60a3225495cf482b823fd531f369b1ac4e459813ffeaa166da07358273159925`. +Its three intermediate blockers (IncompleteObservation, StaleObservation and +MovementBudget) do not replace the complete final counts or two equilibrium +passes with scheduling enabled. This is a native in-process reference profile, +not process/provider qualification. Verification used Rust/Cargo 1.97.0 with +all features, locked dependencies, `CARGO_INCREMENTAL=0` and separate Workspace +targets. Intermediate failed documentation edits remain in the task record. +Production intent supervision, complete role observation, replacement policy, +maintenance finalization and the full W1–W10 qualification remain unfinished. + +### Exact parent CI failures retained + +Parent `c40761eb323e480a3b08cac57ebd97d7f8a715ec` workspace +[job 110784161845](https://github.com/crabbuild/cellule/actions/runs/36990132970/job/110784161845) +failed two publication-hint cases at their original two-second bound. Publication +three received 9/19 and 10/19 healthy hints, respectively. Selected native tests +and a full isolated application run pass; they do not establish the x86 Linux +failure's cause. The existing deadlines, healthy-peer sets and assertions remain. +Original logs are `/tmp/cellule-fleet-c407-workspace*.log` and reproduction logs +are `/tmp/cellule-fleet-c407-*-repro.log`. + +The follower-capacity +[job 110784162740](https://github.com/crabbuild/cellule/actions/runs/36990133107/job/110784162740) +failed its published-root coverage assertion after seventeen rate windows. The +artifact identifies entities 9, 10 and 11: captured root sequence 330 versus +last acknowledged sequence 331. Its publication telemetry also records successful +sequence-331 publications for these Cells. The driver source captures roots before +requesting node stop. A premature root probe is therefore an inference to +investigate, not an established diagnosis or permission to weaken the gate. +The original ZIP (artifact 11218763624, SHA256 +`34cfc1610501fa3fd545cd0497b93d467fae1722a965ae79a1ea3576695f9fe7`) +and all 143 files are retained under +`/tmp/cellule-fleet-c407-follower-capacity*`; the exact mismatch report is +`/tmp/cellule-fleet-c407-root-mismatch.json`. Its process binary SHA256 is +`5716668c0c0e136b0250bfe7cffac4445dc5855bb91ffe3170ff55a5d5f090cf`. + +At the recorded capture the same head passes MSRV, contracts, object capacity, +Compose smoke, website, fuzz and fast/negative TLC. Leased/object routing jobs +110784163035 and 110784163093 in +[run 36990133064](https://github.com/crabbuild/cellule/actions/runs/36990133064) +remain confirmed in progress. Broad TLC and simulator are skipped. These parent +checks do not qualify this new source; the full goal remains active. + +## October 2 2026 writer profile observation and count convergence checkpoint + +The reference observer now captures native pages through `CellNode::fleet_snapshot` +using the original full journal barrier, physical endpoint, fresh nonce and +exclusive deadline. It captures all seven categories and repeats actor topology. +Fresh Cell authority is checked for every catalog-provisioned Cell and reread +before completion; stale actor rows are removed from both count and pressure +inputs. Both canonical advertised-session scans include expired records and +unexpected boots. Directory follower-log discovery includes expired/fenced +leader obligations. The complete journal roster is confirmed after collection. + +Complete counts are supported only for the private constructor's closed profile: +three managed boots, twelve catalog-backed SQL writers, and no reader/follower +installation or enrollment. Unbound pages alone supply no absence proof. +Missing or changed authority, Pending/role enrollment, native role installation, +unknown boots or unresolved logs prevent complete counts. Role-enabled production +observation, replacement policy and maintenance finalization remain unfinished. + +Capacity samples now renew the original signed boot through canonical directory +CAS. The original signing key and executable identity remain pinned; the local +guard advances only after CAS confirmation. Real classifier sequences and sample +times advance naturally. Fenced, missing or expired boots cannot be recreated. +Withdrawal permits validated same-boot heartbeat successors while preserving +original establishment evidence and requiring joined shutdown plus its exact +permanent directory tombstone. + +The new `balance` command drives actual residence and post-batch sample barriers. +Starting at 12/0/0, bounded batches move eight distinct Cells to 4/4/4. +Two further complete passes retain scheduling enabled and allocate no new work. +All eight original command outcomes and SQL values survive, successor authority +is checked, and old handles are fenced. New five-minute receipts cover this +longer scenario; overload/restart retain their one-minute receipts, profiles, +shared bounds and all required two-move assertions. Exit joins all three runtimes, +checks every resource ledger and retires all three exact boot obligations. + +### Focused qualification + +| Source and evidence | Recorded value | +| --- | --- | +| Parent | `ac739bcdad0232296d3b03a7bd34492873aab224`. | +| Qualified Rust/Cargo manifest | 632 sources/manifests/all lockfiles; SHA256 `6bb973f54342c10a296154111273f77396a385601be2816220bcee6b8633c49f`. Active and isolated source sets both match it after execution. | +| Environment | Rust/Cargo 1.97.0; all features, locked dependencies, `CARGO_INCREMENTAL=0`; separate Workspace targets for active and isolated checkouts. | +| Complete fleet example | 98 passed, none filtered or ignored, in 76.80 seconds; seven cases are new. Selected repetitions are not added. | +| New regression cases | Complete writer coverage and canonical renewal; unknown advertised boot; Pending role/revision change; stale current owner; fenced guard and retirement; a real canonical release preserving eleven unchanged pressure candidates; real-time count convergence/equilibrium. | +| Host all-target Clippy / API docs | Clippy and host/runtime API documentation passed with warnings denied. | +| Static gates | Format/diff, boundaries/layout, 110 Rust snippets, 1,172 Markdown links, 28 SQL/peer assertions and 567 validator links passed. | +| Isolated public CLI | `cargo run -p cellule-host --example fleet_operations --all-features --locked -- balance` exited zero. Eight releases, activations, retirements and receipt checks; max inflight two; max restore credit 2,550,136,832 bytes; final counts 4/4/4; three joined nodes and boot retirements. | +| Original CLI executable | SHA256 `b8a813dc120d67c98b080345f6b521a9b2dadba06fef6c1f09369fe60cff8dcd`; executable, raw log, source manifest and metadata archived at `/tmp/cellule-reference-observation-final-cli-evidence/`. | + +Logs and manifests are `/tmp/cellule-reference-observation-*`. This is a +real-time in-process reference qualification, without process crash, distributed +provider, sustained workload or mixed-binary evidence. The full W1–W10 objective +remains active and the PR remains draft. + +### Intermediate failures and corrections + +The first full run passed 94 cases and failed three. Two new fault tests wrongly +expected successful shutdown after deliberately fencing a live writer; the native +owner retained `SharedDrainError(Fenced)`. Cleanup now preserves that original +source across repeated joins, requires non-Stopped status and an unresolved boot +record after failed drain, and still checks zero native resources. The separate +fenced-guard/withdrawal case uses an empty receiver and confirms no advertisement +CAS or boot recreation after fencing. These tests do not manufacture completion. + +An intermediate observer discarded all source rows after any topology change, +losing independent pressure candidates. It now disables complete counts and keeps +only unchanged generation/position/cost/blocker rows from both native captures. +A deterministic regression releases one real Cell and requires the other eleven +to remain candidates. Subsequent full runs pass the original overload and +controller-restart assertions. Earlier generic overload/retirement failures lack +sufficient diagnostics to establish their individual causes; detailed pass and +retained-attempt context is now preserved on failure. + +A new blanket test assertion prohibited IncompleteObservation even during an +accepted movement batch. That misrepresented the coverage contract: transition +passes can be incomplete and must block count planning. The final test instead +requires complete final counts and two complete equilibrium passes with scheduling +still enabled. An obsolete diagnostic executable retained the earlier assertion; +its terminal 96-pass/one-fail result is recorded separately and is not qualification. +The final rebuilt executable passes all 98 cases. A last review preserves the +original movement error Arc in the count command instead of formatting it away; +the complete suite and isolated CLI were rebuilt and rerun after that change. Intermediate API/test constructor +errors and a needless borrowed journal remain in the logs; failed edit scripts +remain in the task record. Production pressure/residence/deadline bounds and +original assertions are unchanged. + +### Parent CI at the recorded capture + +Exact parent `ac739bc` workspace/MSRV, contracts, smoke, both capacity campaigns, +website, fuzz and fast/negative TLC pass. Broad TLC and simulator are skipped. +Workspace [run 36983372396](https://github.com/crabbuild/cellule/actions/runs/36983372396) +is terminal success. Compose [run 36983372019](https://github.com/crabbuild/cellule/actions/runs/36983372019) +was still running its leased/object routing comparisons at the captured read; +that timeout/status is not a terminal result. These parent checks do not qualify +the new source. New-head CI and the full plan's remaining gates are required. + +## October 2 2026 request bound native page checkpoint + +`CellNode::fleet_snapshot` captures one original native category through the +existing two-job fleet action bank. Requests pin the full head/registry, +physical boot, nonce, subject, original native continuation, limit and exclusive +deadline. The journal checks current full versions and endpoint intent in one +transaction before and after capture. Retries preserve the original request and +interval; changed revisions fail instead of restamping a page. Applications +still own authenticated transport and its encoding. + +Actor, managed reader, reader producer, inbound follower, managed follower +producer and supervisor pages retain their original native owners, allocation +tokens and error references. Missing owners return explicit Unbound coverage. +The envelope reports actual lifecycle before/after capture, shared mode and +local log identity. A strict runtime binding read preserves poisoned-lock +failure. Reader producer pages now expose their original journal scope and +physical node; persisted formats remain unchanged. + +Dropped waiters leave accepted native reads and blocking jobs retained until +join. Effects, inspections and page requests share the original two-job and +retained-byte bounds; shutdown joins their original tasks. Tests hold actual +pre/post authorization calls, cancel waiters, race a registry update after +native capture, reject a foreign boot, inspect the real actor and unchanged +Cell authority, and require all retained credit to return after shutdown. +The independent SQLite client case checks read authorization, deadline, +registry/boot changes and controller-head revision without publishing effects. + +| Source and focused evidence | Recorded value | +| --- | --- | +| Parent | `bb08c902e574b8ec40e4580a2a783806abca10e2`. | +| Rust/Cargo manifest | 628 sources/manifests/all lockfiles; SHA256 `9c41f0ee52ae80ddb5929e6158ed6e625377d236d1e7c768c6ec14554554fea6`. | +| Environment | Rust/Cargo 1.97.0; all features, locked dependencies, `CARGO_INCREMENTAL=0`; target `$HOME/Workspace/crabbuild-target/cellule-f9383af7-fleet-operations`. | +| Complete host library / public node | 29 / 88 passed; none filtered or ignored. | +| Selected independent SQLite snapshot authorization | One passed; 90 filtered; none ignored. The complete example suite was not rerun for this checkpoint. | +| Distinct scoped cases | 118 passed, including eight new cases; preliminary repetitions are not added. | +| Clippy | Host all-target Clippy passed with warnings denied. | +| API documentation | Host/runtime all-feature API documentation passed with warnings denied. | +| Static gates | Format/diff, boundaries/layout, 110 Rust snippets, 1,172 Markdown links, 28 SQL/peer assertions and 567 validator links passed; document gates were repeated after adding this record. | + +Original logs and source manifest are `/tmp/cellule-native-snapshot-*`. +Intermediate compilation caught test comparisons against a non-PartialEq +error, the endpoint test expected the wrong error wrapper, and Clippy caught +an oversized shared job enum. Final tests preserve exact wrapped fencing; +boxing the immutable request fixes the enum without suppressing the lint. + +Complete authenticated aggregate observation, current remote authority, +replacement policy, stable traversal and W6–W7 finalization remain unfinished. +The reference observer still reports incomplete coverage; this native page +path cannot enable complete balancing or maintenance on its own. + +### Previous checkpoint terminal CI + +CI on parent `bb08c90` is terminal: workspace, MSRV, contracts, smoke, both +capacity campaigns, website, fuzz, fast/negative TLC and object-only routing +passed. Broad TLC and simulator were skipped. The leased routing +[job 110747490689](https://github.com/crabbuild/cellule/actions/runs/36978501812/job/110747490689) +failed the unchanged performance gate for `leased/local_command/c1`, after +all eight functional executions passed. Cause remains unestablished. Original +run metadata and logs are `/tmp/cellule-fleet-bb08c90-compose-run.json` and +`/tmp/cellule-fleet-bb08c90-leased-routing.log`. Uploaded artifact +`cell-routing-leased-36978501812-1` has ID 11215841243 and SHA256 +`8a542e5db183e14399216e00b41222daa66efdc192cd99c307d4661c68a09821`. +These prior results do not qualify the new checkpoint or complete the plan. + +## October 2 2026 retained supervisor observation checkpoint + +`CellNode::fleet_durability_supervisor` captures the existing supervisor owner +without waiting on its join lane, provider I/O or native retirement. It exposes +unstarted/running/returned/joined states, an unobserved task exit, request-stop +failure, and the original bounded rotation bank, including automatic claims. +Supervisor and request-stop failures remain separate original references. +Unavailable bank capture is an explicit error; missing, idle, canceled or joined +work supplies no role-absence or safe-finalization proof. + +The owner charges four KiB to the shared retained-byte ledger before provider +work starts. `CellRuntime::try_reserve_node_metadata_bytes` permits bounded +lifecycle metadata before lease admission and after fencing. It authorizes no +native work or role; `try_reserve_node_bytes` still checks the node lease, and +terminal drain closes both allocation paths. Capture uses the installed charge +after new byte admission closes. Shutdown releases that owner and its charge. + +The original task join is committed before stopping its request bank. A failed +stop cannot leave a consumed JoinHandle available for a second poll. Tests join +real successful and panicked supervisor tasks with a poisoned original bank, +repeat the join, and compare the concrete original source addresses and separate +error Arcs. Public tests capture unresolved retirement and recruitment during +drain, preserve completed original rotation/proof references, and distinguish +unbound, wrong-type, invalid-time and exhausted admission without creating work. +Runtime tests require metadata/native allocations to share the same capacity, +preserve fencing, reject zero/full/closed allocation and release exact credit. + +The first complete node run passed 83 cases and failed the existing provider +installation before a lease with `Fenced`. The final metadata path preserves +that startup contract; the original test is unchanged. Earlier public-test +builder omissions and a trait-object vtable identity assertion are retained in +the intermediate logs. Concrete downcast source addresses avoid duplicate +vtable identities while keeping the original-source assertion. + +| Source and focused evidence | Recorded value | +| --- | --- | +| Parent | `9faedb991363cbe2d34dfe90690c8e87cc5e605c`. | +| Rust/Cargo manifest | 623 sources/manifests/all lockfiles; SHA256 `f1f06c9f28802aef58710332984425116ab4f01681f903c10533d6c651ab6340`. All indexed blobs match the tested manifest. | +| Environment | Rust/Cargo 1.97.0, all features, locked dependencies, `CARGO_INCREMENTAL=0`, separate Workspace targets for the active and isolated checkouts. | +| Complete host library / public node / fleet example | 26 / 84 / 90 passed; none filtered or ignored. | +| Isolated runtime lease selection | Five passed, 196 filtered; none ignored. Includes the new shared metadata/native credit and fencing case. | +| Isolated complete application integration | 29 passed; none filtered; 16 documented manual cases ignored. | +| Distinct scoped cases | 234 passed, including ten new cases. Repetitions and intermediate runs are not added. | +| Clippy / API documentation | Runtime/host/application all-target Clippy and runtime/host API docs passed with warnings denied. | +| Static gates | Format/diff, boundaries/layout, 110 Rust snippets, 1,172 local Markdown links, 28 SQL/peer assertions and 567 validator links passed. Document gates were rerun after adding the final record. | + +Both drivers compare every file and the complete path set after each command. +Logs, drivers and manifest are `/tmp/cellule-supervisor-observation-final-*`; +intermediate evidence is `/tmp/cellule-supervisor-observation-*`, including +`pre-startup-fix`. The full plan and authenticated aggregate observer remain +incomplete; this checkpoint is a native-owner observation prerequisite. + +### Previous checkpoint terminal qualification + +The corrected `9faedb9` Linux ARM64 campaign is terminal. With the pinned image +below, two CPUs, four GiB, zero swap and 8,192 descriptors, run 1 passed 89 +example cases and failed `measured_overload_moves_real_cells_after_durable_controller_reconstruction` +with `real movement did not settle both attempts`; its cause is unestablished. +Runs 2–5 each passed all 90 cases, in 28.83, 31.62, 57.06 and 51.62 seconds. +The 128-reader and canonical renewal cases passed all five runs, including +beyond the original 30-second expiry. Every before/build/after source manifest +matches the previous 619-path `fbc5eb81…` source. The original binary SHA256 is +`6ba713e78de3c0de84d6d6b997681b7a6736199bd94255bba47dccad5ee51a14`. +All raw logs, build JSON, exact executable, limits, hashes and terminal no-OOM +state are `/tmp/cellule-host-linux-evidence-reader-pressure-fix/`. +The first failure prevents claiming this entire campaign passes. + +Exact-head `9faedb9` workspace +[job 110731568907](https://github.com/crabbuild/cellule/actions/runs/36973266806/job/110731568907) +passes 89 example cases and fails the controller-restart requirement of two +retained lost release replies: one reply and one attempt remain, with one +retirement reported. Cause remains unestablished. Its Compose smoke +[job 110736670337](https://github.com/crabbuild/cellule/actions/runs/36973266798/job/110736670337) +fails `reader-6 made no progress replacement reader replacement` during scale +verification. Both jobs checked out PR merge revision +`3b3bb0a13eae209a845ded4e4ef9316fdc9a930d`, merging `9faedb9` into +`191409685b001a82bd02780def45102b4fc2f164`; the smoke artifact records that +exact source. Original logs are `/tmp/cellule-fleet-9faedb9-{workspace,smoke}.log`; +the original smoke artifact is `/tmp/cellule-fleet-9faedb9-compose.zip`. +Its replacement interval was 14,247,556 microseconds. Reader lane 6 started +three calls in that interval, each returned `behind` at approximately five +seconds, and none supplied an `ok` result wholly inside the interval. This is +failure evidence; the reason for those responses remains unestablished. The raw +artifact SHA256 is `5ab774d86a259c548dae1107118bd806d8685e32ce1fe5c1e985b8b698099900`; +extracted records and the derived summary are `/tmp/cellule-fleet-9faedb9-compose/` +and `/tmp/cellule-fleet-9faedb9-compose-reader-loss-summary.json`. +Contracts, MSRV, both capacity jobs, website, fuzz smoke and fast/negative TLC +pass; broad TLC and the simulator are skipped. Leased routing job 110736670549 +completed success at 2026-10-02 07:19:32 UTC; the object-only comparison was +still running at the 07:22 UTC capture. These results do not certify the new +source or complete W1–W10. The new commit requires its own CI; the PR stays draft. + +## October 1 2026 reader fixture admission and lease checkpoint + +The 128-view pagination case now has explicit native admission headroom and +caller-driven canonical directory renewal. It still requires 128 original +requests, two byte-bounded pages, all 128 installed native views, restored value +readback and joined zero ledgers. The receiver's local guard starts from its +actual published advertisement and advances only after directory refresh confirms. +Expired or withdrawn boot evidence fails; renewal cannot recreate that boot. +No additional background scheduler is introduced. + +The original two-GiB native ceiling plus 256 MiB of retained credit caused the +fixture to depend on opening speed. A real 127-view capture measured +1,598,029,824 native bytes and 24,969,216 retained bytes, zero jobs and zero disk +reservations. Their combined utilization exceeds the classifier's 600-permille +recovery threshold. Holding that workload through the real dwell produced +`Constrained` and the original `Capacity("node pressure")` refusal. The clean +regression also reproduced this refusal during the opening loop. The large +fixture now reserves four GiB of native credit on each runtime; its 128 slots, +256-MiB retained ceiling and 192-KiB enrollment charges are unchanged. Ordinary +memory budgets and production profiles/classifier thresholds are unchanged. + +Before the last Pending request, the case requires Normal pressure across at +least 1,500 ms of actual sample timestamps. The first corrected full run exposed +another original fixture limit: at 31.98 seconds its unrenewed 30-second +advertisement expired and opening returned `Fenced`. The large case now renews +both original signed boots during opening, the dwell and readback. A new +canonical test keeps those boot identities and their guard live beyond the +original two-second advertisement expiry, then explicitly fences the guard. +No expiry assertion or timing bound is relaxed. + +| Source and focused evidence | Recorded value | +| --- | --- | +| Parent | `353904eed67a7e673fc3bc18dcc390e8527473a4`, including the separately authored control-plane plan/audit. Both were preserved identically when advancing the checkout. | +| Rust/Cargo manifest | 619 sources/manifests/all lockfiles; SHA256 `fbc5eb815cfc35d883bc56f84451e8fa52f196ad4934f9070fe2b3090538d835`. | +| Complete host library / public node / fleet example | 20 / 81 / 90 passed; none filtered or ignored. | +| Isolated complete application integration | 29 passed; none filtered; 16 documented manual cases ignored. | +| Distinct scoped cases | 220 passed, including the new canonical renewal case. Repetitions are not added. | +| Clippy / API documentation | Host/application all-target Clippy and host API docs passed with warnings denied. | +| Static gates | Format/diff, boundaries/layout, 110 Rust snippets, 1,172 local Markdown links, 28 SQL/peer assertions and 567 validator links passed; the Rust/Markdown gates were rerun after adding this evidence. | + +Both drivers recheck every source and the complete path set after each command. +Logs, drivers and manifest are `/tmp/cellule-reader-pressure-final-*`. +Original diagnostic/refusal logs are `/tmp/cellule-reader-pressure-*`; the +intermediate complete example failure is retained under `pre-heartbeat`. +An intermediate renewal-helper compilation error and its corrected selected +case are retained under `/tmp/cellule-reader-heartbeat-*`. Diagnostic logging +was confined to the isolated reproducer and is absent from this change. + +Five full Linux ARM64 baseline runs used two CPUs, four GiB and zero swap with +the pinned Rust image described below. All failed, including original operating +system file descriptor exhaustion; additional controller/timing failures remain +unexplained. They do not isolate the x86 CI causes. The operating system descriptor +limit was not recorded for that original campaign. Every build/run source check +matches the original 618-file manifest +`9eb1e0bef426cfd02bc3212e12d058fe7d123f7a0bafaa731a2f554d60d96fbf`; original binary SHA256 is +`0c281c09e65e562a774ecc409a207f750c3e49514d060d9e8b49a588611c69ad`. +Source hashes, all five logs, build output, original binary and terminal container +configuration/state are retained in `/tmp/cellule-host-linux-evidence-489c5a9/`. +The corrected source ran separately with an explicit 8,192-descriptor limit; +its terminal results are recorded in the October 2 checkpoint above. + +### Current CI limitations + +Parent `353904e` workspace +[job 110723513586](https://github.com/crabbuild/cellule/actions/runs/36970580571/job/110723513586) +passes 88 example cases and fails the same large reader admission refusal. +Original log is `/tmp/cellule-fleet-36970580571-workspace.log`. +Previous `489c5a9` workspace +[job 110716061446](https://github.com/crabbuild/cellule/actions/runs/36968076991/job/110716061446) +fails the reader refusal and the controller-restart expectation of two retained +lost release replies. Its contract +[job 110716061458](https://github.com/crabbuild/cellule/actions/runs/36968076984/job/110716061458) +times out on publication 3 with 10 of 19 healthy hints, nine missing peers and +one intentionally stalled transport. Original logs are +`/tmp/cellule-fleet-489c5a9-{workspace,contract}.log`. The latter two causes remain +unestablished. Local passes cannot qualify those failures or complete W1–W10. +The new commit requires its own CI; the implementation PR remains draft. + +## October 1 2026 follower producer inventory checkpoint + +`CellNode::fleet_follower_enrollments_page` captures every retained original +epoch through the existing supervisor's provider. It includes all selected +member requests and unknown acceptance, original signed-attempt/proof digests, +native dispatch/delivery/closure flags, and original retirement/error Arcs. +Signed ensembles, provider CAS tokens and native proofs are hashed in place +rather than deep-copied. The [producer inventory](../crates/cellule-host/src/durability/enrollment/inventory/mod.rs) +reserves one MiB before copying fixed-size follower metadata. Its 1–32-epoch +pages use the producer's existing 32-epoch bound. Continuations bind every +epoch, pending attempt, shared mode and protocol state; missing keys or changed +progress require restarting the scan. + +Capture does not await provider/journal/native calls or acquire the async +protocol lane. A busy lane with zero rows exposes preparation before an original +request exists. An idle lane or zero rows cannot prove a joined supervisor, +empty native lanes or fleet settlement. Typed component lookup now preserves +wrong-type and poisoned-lock failures; unbound `None`, exhausted credit and a +closed runtime cannot be treated as empty coverage. This remains advisory +interval observation, with separate current authority and complete role +envelopes required. + +Three new public scenarios use the same SQLite journal, signed directory and +two actual native follower stores. They cover paused preparation, partial member +acceptance, original immutable requests/proof identities, exact page charge and +release, bounds/missing/stale continuations, memory refusal, and original failed +member/error Arcs during requested rotation. The latter pauses the next native +retry before comparing two captures, preserving the original response without +a scheduling race. The existing deadline case additionally requires a closed +runtime to refuse a page while its unresolved original epoch remains visible +through retained completion. Three new unit cases cover cursor validation, +bounded fixed metadata/signing scratch, and native/proof/error progress hashing. +Public component assertions distinguish missing, unowned and wrong-type owners. + +| Source and focused evidence | Recorded value | +| --- | --- | +| Baseline | `8cff2c3e6e516ebd5b4434e31fd0cad83c24a509` | +| Rust/Cargo manifest | 618 sources/manifests/all lockfiles; SHA256 `9eb1e0bef426cfd02bc3212e12d058fe7d123f7a0bafaa731a2f554d60d96fbf`. | +| Environment | Rust/Cargo 1.97.0, all features, locked dependencies, `CARGO_INCREMENTAL=0`, separate Workspace targets for the active and isolated checkouts. | +| Complete host library target | 20 passed; none filtered/ignored. | +| Complete public host node target | 81 passed; none filtered/ignored. | +| Complete fleet example target | 89 passed; none filtered/ignored. | +| Isolated complete application integration target | 29 passed; none filtered; 16 documented manual cases ignored. | +| Distinct scoped cases | 219 passed, including six new cases; repetitions/intermediate runs are not added. | +| Host/application all-target Clippy and host API docs | Passed with warnings denied. | +| Static gates | Format/diff, boundaries/layout, 110 Rust snippets, local Markdown links, 28 SQL/peer assertions and 567 validator links passed. | + +Both verification drivers compare the complete unchanged Rust/Cargo path set +and manifest after every command. Final reproduction, manifest and logs are +`/tmp/cellule-follower-inventory-final-*`. The active link gate also includes the +independently edited control-plane plan/audit; the final isolated documentation +snapshot excludes them. Intermediate module/helper/ownership/macro compilation +failures are archived under `/tmp/cellule-follower-inventory-*`. An intermediate +full example run passed 87 cases and failed the added page assertion after runtime +closure. The final case retains the original proof assertions and requires +`RuntimeClosed`; the separate live rotation case proves original response +identity before closure. The earlier 219-case run before the deterministic retry +gate is archived as `pre-retirement-gate` and is not added to the final count. + +### Publication-hint reproduction and baseline CI + +Baseline `8cff2c3` workspace +[job 110698431351](https://github.com/crabbuild/cellule/actions/runs/36962309533/job/110698431351) +failed `pending_reader_activation_retains_a_new_publication_hint` at its +two-second post-publication bound: 28 application cases passed and 16 manual +cases were ignored. The log does not identify the occurrence or missing peers. +The workspace stopped before the host example target. Cause remains unestablished. + +Fresh isolated baseline builds passed one initial selected run, 100 serial and +100 concurrent repetitions, plus ten complete macOS application runs. A copied +source snapshot from the baseline Git archive (SHA256 +`186c9b451dc62fac9c7d9b39014cb0a94cb37517a3bdd00f161d6eac93ce11db`) +also passed twenty complete Linux ARM64 application runs, each 29 passed and +16 manual cases ignored. The pinned Rust image was +`rust@sha256:0e2bcaef56d041a486784e54104a81aebe0da44bd03019bd70bc0401e42e4a97`, +with two CPUs, four GiB memory and zero swap in the shared eight-CPU Colima VM. +Linux binary SHA256 was +`fe3f3c993893686b1c14b0adff85a1ce929382b85ded5fe7d366b7f4e08cb8f5`. +These repetitions do not explain the original x86 CI failure or qualify the new +source. Linux post-run source hashes were not retained. The final new-source +application run above has its own unchanged 618-file manifest. + +Diagnostic drivers/logs are `/tmp/cellule-hint-repeat*`, +`/tmp/cellule-hint-full-repeat*` and `/tmp/cellule-hint-linux-evidence/`; the latter +includes every original run, build, binary identity and container limits/state. +The dedicated Linux container exited zero without OOM and was removed after +evidence capture. Its build/Cargo volumes remain for reproduction. Initial host +bind mounts failed because the Docker VM cannot follow the Workspace volume's +host symlink; copying the pinned archive into the isolated container resolved +that environment limitation. + +The regression now retains partial received-peer state in its timeout assertion +and reports the publication occurrence, elapsed time, missing/held peers and +pending transports. Its two-second bound, five-second periodic tick, healthy-peer +set, duplicate assertions and joined shutdown remain unchanged. No recruitment +algorithm or qualification profile was changed; the failure remains open pending +cause evidence. + +Baseline `8cff2c3` passes MSRV, both capacity campaigns, contracts, website, +fuzz smoke, fast/negative TLC and +[Compose smoke and both routing comparisons](https://github.com/crabbuild/cellule/actions/runs/36962309460). +Leased job 110698430939 completed success at 2026-10-02 04:23:26 UTC; +object-only job 110698430922 at 04:41:35 UTC. Broad TLC/simulator were skipped. +These green comparisons do not establish the cause of the historical routing +failures below, and new-source CI remains required. + +Complete authenticated role/current-authority envelopes, supervisor and failed +boot evidence, replacement policy, SettleRoles/Finalize, cross-session receiver +recovery, remaining primitive/fault matrices, convergence, complete maintenance +and receiver-loss scenarios, process/provider/mixed-binary qualification and +rollout/runbooks remain required. This checkpoint does not complete W1–W10. + +## October 1 2026 reader producer inventory checkpoint + +`ReadReplicaManager::fleet_reader_enrollments_page` supplies bounded original +reader requests and progress while acceptance, native opening or result +publication awaits its original reply. The canonical activation lane still owns +all mutations; brief index and progress locks replace the record lock previously +held over I/O. Native execution errors are retained before publication awaits, +independently of journal errors. No alternate opening or retirement path is added. + +Every page reserves one MiB before copying, allows 1–128 rows, scans at most +10,000 obligations, and preserves original specifications, acceptance, source, +events and errors. Its fixed-width continuation binds boot/scope, mode, job +counts and all retained row progress. Changes require a fresh scan. Accepted +preparation before a request exists remains visible as running work. Finished +jobs without their original response and handles held by another join owner +remain explicitly unknown. Returned responses do not discard retained epilogue +joins. Unbound managers provide no coverage; closed or exhausted runtime ledgers +fail without synthesizing an empty inventory. + +Six new public example cases use the shared SQLite journal and actual native +readers. They cover paused/lost acceptance, paused establishment, preparation +before any journal row, stable multiple pages, progress/retirement/mode changes, +invalid cursors/bounds, ledger admission refusal, and exact page charge/release. +Existing native VFS pause and cordon/refusal cases now inspect progress before +releasing their original owner. The public host inventory case additionally +checks unbound/invalid enrollment capture. Six new unit cases cover cursor +encoding, live/joining jobs, a real panicked owner without its response, returned +protocols, variable byte admission and distinct original error Arcs. + +Variable payloads stop copying before the byte boundary and return a continuation, +while every remaining original row still enters the topology hash. An oversized +first row fails rather than being skipped. The large native fixture preserves +128 original requests across pages while the final acceptance is paused (127 +installed views plus one Pending request). After that owner resumes, it confirms +all 128 native views and reads their restored value through receipt-bound queries. +Joined shutdown leaves native/retained bytes, SQL jobs and disk reservations at +zero. The fixture explicitly admits 128 slots, 256 MiB of retained credit and a +two-GiB native ceiling; ordinary cases retain their original eight slots, +16 MiB of retained credit and 128-MiB receiver ceiling. Production profiles are +unchanged. + +| Source and focused evidence | Recorded value | +| --- | --- | +| Baseline | `58721227e56dac3ebcda6b74b4ed2a514f8cc42b` | +| Rust/Cargo manifest | 615 Rust sources, Cargo manifests and all lockfiles; sorted SHA256/path lines. SHA256 `bee6e0b5bdbe8c136830f152462bc7ad006840462ea906374c76bb03c25e9112`. | +| Environment | Rust/Cargo 1.97.0, all features, locked dependencies, `CARGO_INCREMENTAL=0`, this checkout's Workspace target directory. | +| Complete host library target | 17 passed; none filtered/ignored. | +| Complete public host node target | 81 passed; none filtered/ignored. | +| Complete fleet example target | 86 passed; none filtered/ignored. | +| Distinct scoped cases | 184 passed, including twelve new cases; intermediate runs are not added. | +| Host all-target Clippy/API docs | Passed with warnings denied. | +| Static gates | Format/diff, boundaries/layout, 110 Rust snippets, documentation links and 28 SQL/peer assertions plus 567 validator links passed. | + +The verification driver checked the unchanged source manifest after every +command. Reproduction, manifest and logs are `/tmp/cellule-reader-inventory-*`. +Initial test compilation used nonexistent role accessors, ordered Digest instead +of its bytes, and reached a too-narrow private join helper; those now use the +canonical source identity, byte ordering and the enrollment family's join helper. +An intermediate example run passed 84 cases and failed an assumption that a +Stopped runtime could admit an inventory page. The final case observes joined +empty manager inventory before runtime closure, then requires `RuntimeClosed`. +The active checkout's intermediate link gate encountered a concurrently edited +control plane document's unfinished audit link. An isolated intermediate snapshot +passed all static gates, and the final active-checkout pipeline passed after that +independent document appeared. Its link count includes those separate edits; +an isolated final snapshot qualifies this checkpoint's committed documentation. +The first large fixture reached the unchanged 16-MiB retained credit limit, +which cannot hold 128 obligations at 192 KiB each; its explicit large-scenario +credit now matches that intended workload. A subsequent publication capture timed +out without native phase diagnostics; a diagnostic rerun passed. The final +scenario pauses acceptance, so capture timing is independent of the last native +open, and still requires all 128 views and readbacks after resume. These +intermediate failures remain archived; their runs are not added to the final +count. Moving example module entries beside their child tests exposed an initial +manifest driver assumption about deleted paths. The corrected driver enumerates +all current tracked/nonignored sources, excludes explicitly deleted entries, +and rediscovers the complete path set after each command. No broad, process or +provider suite ran locally. + +### Baseline CI and unresolved qualification + +The original `5872122` workspace job +[110679369745](https://github.com/crabbuild/cellule/actions/runs/36956132510/job/110679369745) +failed after 79 example cases passed: the follower failure case observed no +execution error immediately after its 80 ms drain waiter deadline. That deadline +bounds the waiter, not completion of retained native retirement. The test now +awaits the original failure and retirement observation within the fixture's +three-second capture bound, keeping the 80 ms deadline and every original +closure/error/authority/registry/resource assertion. The original CI log does +not establish which native phase was pending. Its terminal failure remains +archived at `/tmp/cellule-fleet-5872122-workspace.log`; the new source needs CI. + +Baseline MSRV, both capacity campaigns, contracts, website, fuzz smoke, +fast/negative TLC and Compose smoke passed. Broad TLC/simulator were skipped. +Both routing jobs finished failure after all eight functional executions in each +mode passed. The unchanged median-of-four gate requires p95/p99 at most 110% +and throughput at least 90%. Original failing comparison rows: + +| Case | Throughput ratio | p95 ratio | p99 ratio | +| --- | ---: | ---: | ---: | +| `leased/local_query/c1` | 0.98936 | 1.07740 | **1.18368** | +| `object_only/local_query_expired_bursts/c16` | 1.00000 | 1.00871 | **1.15413** | + +Leased [job 110680072679](https://github.com/crabbuild/cellule/actions/runs/36956132515/job/110680072679) +completed at 2026-10-02 03:11:43 UTC; object-only +[job 110680072513](https://github.com/crabbuild/cellule/actions/runs/36956132515/job/110680072513) +completed at 03:12:18 UTC. The comparison artifacts preserve raw rows/windows, +logs, manifests and frozen binaries: + +| Artifact | ID | ZIP SHA256 | +| --- | --- | --- | +| `cell-routing-leased-36956132515-1` | 11207545900 | `0d86d9cb16c83846e0432468b80425bcdd01f6fc02d9bb2e45bb3cb8b291f7f7` | +| `cell-routing-object_only-36956132515-1` | 11206579342 | `accb887edde0b11f11f9e02338a7e685847216a78be20a1608a6ca71f0fc7dcb` | + +Both manifests pin candidate `4b6b091679636b3eeeba985e26f4c9a5164cc44c`, +whose GitHub commit parents are `fbfd84f9c4dfcb6de497072efd9aafa5a6f409cc` +and PR head `58721227e56dac3ebcda6b74b4ed2a514f8cc42b`. The frozen baseline +is `0dc04a658bd99668936f7ec58032d054f6fbc141`. Candidate binary SHA256 is +`5f2914db0810cce008b272fb2e27fb0dbfcd733c1d6bbbed65e1ace77a50f136`, +baseline binary is `08108609e742a3f8091616675152cab3de21f0e021f41588465d60b49a25a83c`, +and both use harness `425b55c168f55ef4b7395c69948027dd77b2bd72836e962daa399120d74149d7`. +The candidate binary matches the prior checkpoint's routing candidate exactly; +this fact alone does not establish the comparison failures' cause. Original ZIPs, +logs and extracted evidence are `/tmp/cellule-fleet-5872122-routing-*`. +No production profile or required evidence was weakened. + +The W1–W10 goal remains active. These advisory producer pages are an input to +complete role observation, not authenticated atomic envelopes, current authority, +replacement policy or finalization proof. Complete failed-session and leader +producer observation, receiver-loss recovery, role actions/settlement, primitive +fault matrices, sustained convergence, complete reference scenarios and +process/provider/mixed-binary qualification and rollout remain required. + +## October 1 2026 durable roster observation checkpoint + +The reconciler now fully traverses retained physical intents and enrollment +records before invoking `FleetObserver::observe`. That adapter receives the +original `FleetRoster`, including Pending work, failed boots, both original +role endpoints and terminal records. Every page uses the canonical codec's +128-row/one-MiB bounds, exact version and strict continuation; each aggregate +is capped at 10,000 rows. The full journal head and registry are rechecked +before/after traversal and after native capture. A lost read or changed version +returns no partial roster. Expired deadlines reject another request before its +adapter future is constructed; dispatched native journal jobs keep their +ordinary retained ownership and join. + +Count balancing now also requires exact established boot advertisements, +current physical intents, no Pending enrollment and ownership rows matching +signed counts. An adapter's completeness flag cannot override these checks. +Missing failed boots, unknown live boots, duplicate boot rows, replaced sessions +and omitted writers disable count balancing. Partial pressure relief keeps the +existing source/receiver gates. Every retained row's original status, evidence, +inputs and timestamps enters the roster digest and the planner's v4 input +producer domain. Persisted record codecs and signed peer formats are unchanged. + +The reference collector consumes this same roster and continues to report +incomplete native-role coverage. Driver fixtures now explicitly establish their +synthetic boot rows before testing counts; native boot qualification remains +in the separate public startup scenarios. No original assertion or profile was +weakened. New cases qualify multiple pages with 132 intents/130 enrollments and +all statuses, reconstruction, independent commits/lost acceptance replies, +controller-head changes, cursor/version errors, aggregate bounds, source error +preservation, deadlines, failed boot preservation, signed coverage, omitted +writers, enrollment during capture and Pending enrollment with pressure relief. + +| Source and focused evidence | Recorded value | +| --- | --- | +| Baseline | `73751090f289f4d302852605f311039adf83d393` | +| Rust/Cargo manifest | 610 tracked/nonignored Rust sources, Cargo manifests and all lockfiles; sorted SHA256, two spaces, relative path lines. SHA256 `4892861ee8518547b233e6bbc5d454131112ab85729cdd6ebd48ba9062cfdefc`. | +| Toolchain/environment | Rust/Cargo 1.97.0; all features, locked dependencies, `CARGO_INCREMENTAL=0`, this checkout's Workspace target directory. | +| Complete host library target | 11 passed; none filtered or ignored. | +| Complete public host node target | 81 passed; none filtered or ignored. | +| Complete fleet example target | 80 passed; none filtered or ignored. | +| Distinct scoped cases | 172 passed, including 17 new cases. Intermediate runs are not added to this total. | +| Host lints/API docs | All targets Clippy and host API documentation passed with warnings denied. | +| Static gates | Format, diff, crate boundaries/module layout, 110 Rust snippets, 1151 Markdown links, 28 SQL/peer assertions and 567 validator links passed. | + +The final pipeline checked the unchanged manifest after every command. Logs, +manifest and reproduction driver are `/tmp/cellule-fleet-roster-*`. An initial +compile attempted a private advertisement method and used an obsolete journal +method name; both now use the canonical public APIs. An intermediate pipeline +omitted five nested lockfiles from its manifest selection; the final pipeline +includes all lockfiles and reran every selected gate. No broad/process/provider +suite executed locally, and these scoped checks do not prove fleet qualification. + +CI for baseline `7375109` passes workspace/MSRV, both capacity campaigns, +contracts, website, fuzz smoke, fast/negative TLC and Compose smoke. Broad TLC +and deterministic simulator were skipped. Leased job 110671204057 is now +terminal failure (2026-10-02 02:31:11 UTC), after all eight functional runs passed. +The unchanged median-of-four gate requires p95/p99 at most 110% and throughput +at least 90%. `forwarded_command/c1` p99 ratio was 1.25613; +`forwarded_command/c16` p95/p99 ratios were 1.15368/1.11897; +`local_query_expired_bursts/c16` p99 ratio was 1.34762. Throughput passed for +these rows. Object-only job 110671203955 completed success at 02:35:43 UTC. +Artifact `cell-routing-leased-36953458504-1`, ID 11205313244, ZIP SHA256 +`dd328ccd12d670f1bcb2f05d3af441ab85abb26dd8a4577d1b524e24b880e724` +preserves the original evidence under `/tmp/cellule-fleet-7375109-routing-*`. +Its candidate source `55ae5c21290319e996b6c4b7506abf6ffdf1d00a` is the +synthetic merge of `7375109` into `fbfd84f9c4dfcb6de497072efd9aafa5a6f409cc`, +compared with frozen baseline `0dc04a658bd99668936f7ec58032d054f6fbc141`. +The cause is unestablished. New source requires independent CI evidence; +earlier comparison failures remain recorded below. + +The complete W1–W10 goal remains active. Next, match bounded native actor, +managed/pending reader, cold follower and leader-enrollment observations against +this roster, using authenticated request-bound envelopes and fresh exact +ordinary authority for every obligation, including failed sessions. Replacement +policy, role evacuation/finalization, remaining primitives/fault matrices, +sustained convergence, complete reference scenarios, process/provider and +mixed-binary qualification, rollout and operator runbooks remain required. +Roster traversal, boot coverage and matching writer counts do not establish +SettleRoles, Finalize, or completion of the plan. + +## October 1 2026 managed follower producer checkpoint + +The host now binds follower production to the shared journal through +`install_fleet_node_durability_provider`. An application-owned read-only provider +supplies one opaque signed attempt and exact boot-bound transport/authority. +The existing retained supervisor owns preparation, every member's Pending +acceptance, its single original-token enrollment CAS, and original establishment +publication. It neither constructs a second scheduler nor reselects on unknown +results. Ship configuration is delivered only after all publication replies +confirm. Before acceptance, canonical gate/shipper validation and the shared +one-MiB reservation fail closed; at most 32 epochs remain inventoried. + +Unknown CAS outcomes reconcile only the original signed source/epoch/member set +or its original-token conditional refusal. Joined nonexecution uses a new atomic +journal exclusion: exact Pending becomes Refused, or an absent key becomes a +terminal refusal tombstone. A late acceptance returns that original row without +opening a native role. This also replaces the managed reader's absence-based +cleanup. Existing opened-reader cases still require Retired; the two explicit +nonexecution cases now require Refused and terminal replay. Independent SQLite +clients qualify the shared transaction domain, immutable timestamps, lost replies +and reconstruction. No persisted format or native qualification profile changed. + +The managed authority receives complete native retirement observations and +requires every original member fence before canonical closure and durable +retirement. Native closure is recorded before fallible evidence construction; +lost publication retries the original events without repeating confirmed close. +Shutdown preserves its original errors and retries instead of caching a terminal +transient failure. Undelivered enrollment cleanup joins after the supervisor, +using the original limits and retained durability object; runtime drain closes +delivered epochs. Caller cancellation/deadlines leave inventory and byte charges +owned until settlement. Local completion captures original native and journal +errors independently; removed history proves no fleet fact. + +Nine new public example scenarios use signed directory authority and two actual +native follower stores. They cover every Pending-before-CAS boundary, partial +acceptance and receiver cordon, lost acceptance/establishment/close/retirement +replies, cancelled/deadline drain waiters, failed member fences, invalid shipper +limits, and an acknowledged SQL mutation. The latter confirms nonzero contiguous +object coverage, every original member watermark, byte-identical root restore, +counter state and the original stored `sys_requests` response. One public host +case rejects an unmanaged provider before it can start on a configured fleet +boot. Three new runtime cases check refusal codecs, startup-held versus confirmed +maintenance state, and token-bound/domain-separated evidence replay. + +The final scoped pipeline passed with Rust/Cargo 1.97.0, all features, the lockfile, +`CARGO_INCREMENTAL=0`, and this checkout's Workspace target directory. Its +Rust/Cargo manifest stayed identical after every command. + +| Source and focused evidence | Recorded value | +| --- | --- | +| Baseline | `cc7aa3063556b03f8c165f1dc74def0aae60a88e` | +| Rust/Cargo manifest | 607 tracked/nonignored Rust/Cargo paths; sorted SHA256, two spaces, relative path lines. SHA256 `ce6f869aec2f6705a260aea141b3837c0db37e0d77cf07c7db9b956c02bb95c9`. | +| Complete public host node target | 81 passed; none filtered or ignored. | +| Complete fleet example target | 71 passed; none filtered or ignored. Includes nine new public native follower scenarios and two new shared-journal cases. | +| Public runtime durability selection | 24 passed; 176 filtered; none ignored. | +| Runtime node library selection | 101 passed; 408 filtered; none ignored. | +| Runtime fleet operations selection | 66 passed; 443 filtered; none ignored. | +| Runtime admission selection | 5 passed; 504 filtered; none ignored. | +| Distinct scoped cases | 348 passed. Earlier/intermediate executions are not added to this total. | +| Lints | Host/app all targets, runtime library/public runtime target, all features, warnings denied; passed. App process drivers compiled under Clippy; they were not executed locally. | +| Host/runtime API documentation | All features, warnings denied; passed. | +| Static gates | Format, diff whitespace, boundaries/layout, Rust snippet and Markdown link checks, and SQL/peer validators passed. | + +Intermediate attempts exposed obsolete reader settlement assertions, premature +fixture readiness, missing advertised follower capacity, a Cell replica limits +mismatch, one missed test-only authority signature, and two lint issues. Fixtures +now follow startup order and existing capacity/application contracts; opened +readers still require Retired, while proven nonexecution requires Refused and +terminal replay. Production limits, profiles and required evidence are unchanged. +Logs and the exact manifest use `/tmp/cellule-managed-follower-*`; the intermediate +pipeline failures remain in the `intermediate-verification` and +`pre-lint-verification` logs. No process/provider suite ran locally. + +Baseline `cc7aa30` passes workspace/MSRV (36947731399), both capacity campaigns +(36947731308), contracts (36947731662), website (36947731302), decoder fuzz smoke +(36947731443), fast/negative TLC (36947731392), and Compose smoke +[job 110653610813](https://github.com/crabbuild/cellule/actions/runs/36947731323/job/110653610813). +Broad TLC and the deterministic simulator were skipped. Routing +[job 110653610627](https://github.com/crabbuild/cellule/actions/runs/36947731323/job/110653610627) +completed with a failed frozen-binary comparison. All 16 functional executions +passed. `leased/local_command/c16` failed the unchanged median gate: +throughput ratio 0.88824 (required at least 0.90), p95 ratio 1.12470 (required +at most 1.10), and p99 ratio 1.06528 (passed). Reads were 2192 for both. +The cause is unestablished. Artifact `cell-routing-36947731323-1`, ID +11204931295, retains raw rows, windows, logs and frozen binaries. ZIP SHA256: +`140559620e498df8bd134f9902e422cf95c5ba4658846ed15f463af716e32ed2`. +Candidate synthetic merge `51a7ee50ed76d2a37311317d5df6291888714859` +combines PR `cc7aa30` with base `0dc04a658bd99668936f7ec58032d054f6fbc141`. +The newer routing workflow measures its head independently; passing measurements +would not prove this earlier comparison regression resolved. +The new source requires its own CI qualification. The original `6092203` reader +scaling failure and artifact below remain evidence; later green smoke does not +establish that failure's cause. + +The full W1–W10 objective remains active. Complete authenticated revisioned role +observations, failed-process/boot evidence, replacement-policy proof, maintenance +role actions/finalization, remaining primitive/fault matrices (including Cron and +Blob external owners), sustained convergence, full maintenance/receiver-loss +examples, process/provider/mixed-binary qualification and operator rollout/runbooks +remain required. The existing ownership-only collector still reports incomplete +role coverage. This producer checkpoint alone cannot complete SettleRoles, +Finalize, or the full plan. + +## October 1 2026 prepared follower enrollment checkpoint + +Runtime recruitment now shares one canonical selector and conditional write +with a read-only preparation bridge. Opaque selection retains every original +signed follower boot, complete member set and provider-assigned advancing epoch. +An opaque attempt revalidates those exact boots and receiver admission/capacity, +then captures a fresh source CAS token before registry acceptance. Commit uses +that original token and ensemble; it never substitutes a receiver or rebases an +ambiguous effect. Directory instance binding permits clones and rejects other +instances even with identical fleet metadata. Ordinary recruitment preserves +its caller's observed-version CAS behavior. + +Fresh inspection reconciles the original source boot, epoch and complete member +set across activation, coverage and heartbeat changes. A conditional refusal +fence competes with enrollment using the same original token and next generation. +Its checked successor prevents that attempt's delayed CAS. Missing or newer empty +records cannot create this proof; enrollment followed by closure cannot be +reported as refusal. These proofs do not assert native follower fsync, current +receiver authority, fleet journal publication or role retirement. The managed +host producer still needs to journal all members Pending before native dispatch +and consume these retained attempts through settlement. + +Retirement now caches complete checked member responses before authority closure. +After a lost close reply or a cancelled close waiter, strict and ordinary retries +reuse that exact observation and retry the original authority callback. They no +longer send unauthorized retire requests against an already-closed epoch. Final +shutdown proof still requires callback success. Partial best-effort closure and +contradictory receipts preserve their original contracts. + +Eleven new directory cases exercise read-only selection, exact signed boot +metadata, heartbeat rebasing before acceptance, stale dispatch tokens, current +inspection, absent/withdrawn source, scope isolation, expired/replaced followers, +receiver capacity/mode/pressure, outbound enrollment from a cordoned leader, +signer replacement and conditional refusal races. Three new public runtime +cases use actual Cell commands, native follower stores and signed directory +authorization/CAS. They lose closure replies, cancel closure waiters, and cancel +reconciliation after the backend accepted an enrollment/refusal but lost its +reply. Original member observations, persisted native fences, exact-root state +and the original `sys_requests` response remain checked. These in-process faults +do not establish provider/process or complete fleet-role qualification. + +The final scoped pipeline passed against the exact Rust/Cargo manifest below. +The selections establish these contracts and regression evidence; they do not +qualify the full W1–W10 implementation. + +| Source and focused evidence | Recorded value | +| --- | --- | +| Baseline | `60922033dc3ecc9a01f3f15877aa62c3cdabd62c` | +| Rust/Cargo manifest | 601 tracked/nonignored `.rs`, `Cargo.toml` and `Cargo.lock` paths; sorted SHA256, two spaces, relative path lines. SHA256 `787c1c77da2b553be326a312b05cfbc32143a5f98ad5ab45a15abb0bbc7057ee`. | +| Complete public host node target | 80 passed; none filtered or ignored. | +| Public runtime durability selection | 24 passed; 176 filtered; none ignored. Includes the three new signed-directory/native-follower cases. | +| Runtime node library selection | 100 passed; 406 filtered; none ignored. Includes eleven new directory cases. | +| Complete fleet operations example target | 60 passed; none filtered or ignored. | +| Distinct scoped cases | 264 passed. Earlier and intermediate selections are not counted again. | +| Host lint | All targets/features, warnings denied; passed. | +| Runtime lint | Library/public runtime target, all features, warnings denied; passed. | +| Host/runtime API documentation | All features, warnings denied; passed. | +| Static gates | Format, diff whitespace, boundaries/layout, 110 Rust snippets, 1150 Markdown links and 28 SQL/peer assertions with 566 links passed. | + +An intermediate retirement-only run passed five existing cases and failed both +new signed fixtures because their heartbeat lifetime exceeded the existing +30-second contract. The fixtures now use 20 seconds; the limit and verification +assertions are unchanged. Final logs and manifest use +`/tmp/cellule-fleet-enrollment-*`, Rust/Cargo 1.97.0, all features, the lockfile, +`CARGO_INCREMENTAL=0` and this checkout's Workspace target directory. Persisted +formats, signed message bytes and object paths are unchanged. + +Baseline `6092203` passed Rust (36944383297), capacity (36944383244), contract +(36944383282), website (36944383388), fuzz (36944383341), and fast/negative TLC +(36944383262 / 110642975391). Broad TLC and the simulator corpus were skipped. +Compose smoke **failed** at the reader-scaling step in +[job 110643075559](https://github.com/crabbuild/cellule/actions/runs/36944383259/job/110643075559). +Its three initial provider-backed tests passed; the mixed three-node reader +driver then failed at `process_scaling.rs:548` with `ReplicaUnavailable` after +successful reader readiness and distribution. The last retained mixed write is +arrival 184 at about 36.8 seconds, with acknowledged sequence 246/count 216. +This is a symptom, not an established cause. The raw artifact +`cell-reference-compose-36944383259-1` (ID `11201985421`) includes original +driver/node/container logs, TSV arrivals and reads, binary hashes and source +identity. The downloaded ZIP SHA256 is +`2ec63e359d13b01ca86ef2fd16ae3763d53e0517480e32d723772fc14a6c06da`. +The artifact pins synthetic PR merge source +`68943f6c8b0e1bd7fd57e766590779c067983d34`, whose parents are baseline +`0dc04a658bd99668936f7ec58032d054f6fbc141` and PR head `6092203`. +The scaling driver binary SHA256 is +`d8796fd2b9a195c62fe7c4020aadab033a27a1f013b2ad56ee2bc56ef239ded6`. +Local copies are `/tmp/cellule-fleet-6092203-compose-smoke*`. Routing job +`110643075291` is authoritatively Cancelled (completed at 2026-10-02 00:47:34 UTC) +after the subsequent push. +Reader availability and process/provider qualification remain open. No profile, +required evidence or assertion was weakened. + +The full W1–W10 goal remains active. Next work includes managed follower +production before CAS, journal retirement, complete revisioned role observations, +failed-owner/replacement-policy proof, role evacuation/finalization and the +remaining primitive, example, fault, deployment and operator deliverables. + +## October 1 2026 requested host rotation checkpoint + +`CellNode::request_node_log_rotation(epoch)` wakes the existing durability +supervisor without waiting for a frame threshold. Acceptance is linearized with +automatic rotation: duplicate epochs share progress, and an automatic +best-effort rotation already in flight is refused. One pending and one latest +completed request are charged to the node byte ledger. Weak handles can inspect +or recover local progress after a lost caller; they cannot retain the node's +resources after drain. Evicted local history proves neither absence nor success. + +Requested work keeps confirmed retirement through all retries. Its original +epoch, complete member set and object-coverage barrier cannot be weakened by a +timer. Local completion requires every old member's append fence, canonical +authority closure and a newer binding installed through expected-binding +replacement. The runtime checks boot, physical leader, epoch advancement and +the exact old binding under its replacement lock. Foreign replacement inputs +are rejected before construction or any attempt to close their scope. Tuple +identity getters supply configured metadata, not signed current authority. + +The host retains the single supervisor's join outside the cancellable task-group +watcher. Cancelling a caller, watcher or facility join leaves accepted native +retirement/recruitment owned. A replacement that returns during drain is closed +through the existing canonical path; lost close replies retry that same object. +Lease maintenance now waits for required facility joins as well as runtime +closure. Deadlines leave Draining and retain progress; Interrupted never becomes +Completed. The original first and latest source failures remain independent of +eventual success. + +Eight new public host cases use actual Cell commands and native follower stores. +They cover strict retirement and recruitment retries while writes continue, +old/new follower ensembles and ordinary object-proof fallback, duplicate and +stale requests, dropped handles, cancelled shutdown waiters, an automatic +rotation already in flight, bounded history/zero final resource charges, +invalid replacement boot/node/epoch, accepted recruitment across a host deadline, +and lost cleanup close replies across repeated deadlines. Exact-root readback +checks Cell state and the original `sys_requests` result. Independent follower +reopening checks old durable fences. The source authority fixture records and +reconciles callbacks; it does not exercise signed directory CAS or provider/ +process fault behavior. + +Intermediate full-target verification caught a startup regression: reserving +fixed supervisor bookkeeping through the leased runtime rejected provider +installation before a lease existed. Fixed supervisor ownership remains under +the facility/task bounds; accepted request records retain ordinary ledger +admission. The existing pre-lease installation contract now passes unchanged. +Lease withdrawal ordering and generation cleanup were repaired without weakening +the shutdown deadline or expected evidence. + +| Source and focused evidence | Recorded value | +| --- | --- | +| Baseline | `29780a36c477bb283d19f0f8e3fef11e4c249c1a` | +| Rust/Cargo manifest | 599 tracked/nonignored `.rs`, `Cargo.toml` and `Cargo.lock` paths; sorted SHA256, two spaces, relative path lines. SHA256 `8092eea2db8eb158ddea7ae56c19a6549be8da94c11a7c44e39df81a2dd9cabb`. | +| Complete public host node target | 80 passed; none filtered or ignored; includes eight new cases. | +| Public runtime durability selection | 21 passed; 176 filtered; none ignored. Includes atomic rejection of stale bindings, foreign boots/physical nodes and non-advancing epochs. | +| Runtime node library selection | 89 passed; 406 filtered; none ignored. | +| Complete fleet operations example target | 60 passed; none filtered or ignored. | +| Distinct scoped cases | 250 passed. Initial retirement-only and intermediate target runs are not counted again. | +| Host lint | All targets/features, warnings denied; passed. | +| Runtime lint | Library/public runtime target, all features, warnings denied; passed. | +| Host/runtime API documentation | All features, warnings denied; passed. | +| Static gates | Format, diff whitespace, boundaries/layout, 110 Rust snippets, 1150 Markdown links and 28 SQL/peer assertions with 566 links passed. | + +Commands use all features, Rust/Cargo 1.97.0, the lockfile, +`CARGO_INCREMENTAL=0` and the checkout's Workspace target directory. The source +manifest and final logs are retained under +`/tmp/cellule-fleet-requested-rotation-*`. The host adds the already locked +`bytes` crate only to test dependencies. `replace_node_durability` now takes +the exact expected binding; all workspace callers are updated. Persisted IDs, +object paths, LTX formats and signed message bytes are unchanged. + +Baseline `29780a3` passed Rust (36940992641), capacity (36940992651), contract +(36940992648), website (36940992677), fuzz (36940992676), model (36940992681) and +Compose smoke (36940992647 / 110632500524). Routing 110632500048 remained running +when inspected. Optional broad model/simulator campaigns require job-level +evidence; workflow success alone does not establish their execution. Earlier +failure evidence remains below and this new source requires its own CI gates. + +Next, journal follower production before recruitment CAS and consume complete +revisioned role observations, replacement policy and failed-owner recovery +evidence. The new request is a local host primitive: the application still owns +authorization, deadlines, maintenance-node exclusion, enrollment/results and +redundancy policy. The reference collector remains explicitly incomplete; +requested rotation alone cannot complete SettleRoles or Finalize. All remaining +W6–W10 matrices, sustained convergence, the four executable scenarios, +process/provider/mixed-binary qualification and operator rollout/runbooks remain +part of the active goal. + +## October 1 2026 confirmed member retirement checkpoint + +`NodeDurability::shutdown_for_maintenance` now shares the existing shipper drain, +contiguous object-publication barrier and authority close path. It joins every +original member, retains individual source errors and requires every exact +retirement receipt before authority closure. A failed response remains retryable +with the original leader, epoch, complete member set and watermark. The opaque +member proof is cached only after canonical authority closure succeeds. + +Ordinary `shutdown`, `rotate_node_log` and `close_node_log` preserve documented +best-effort retirement. Their successful epoch closure cannot establish complete +member confirmation when a response was lost. Contradictory successful receipts +still block authority closure in both modes. An observation's `confirmed()` +proof describes member append fences only; it does not assert directory closure, +lane deletion, journal settlement, replacement policy or safe physical shutdown. + +Five new public cases use actual SQLite commands, native follower stores and +canonical Cell publication. They cover lost retirement replies after fsync, +joining a delayed healthy sibling before returning another member's failure, +exact-scope retry and cached proof, cancellation before authority close, +best-effort closure that cannot be upgraded into proof, contradictory receipts +in both modes, and incomplete object publication blocking all retirement RPCs. +Every case reopens follower storage, observes persisted Retired fences and +rejects old appends, then restores the exact authority root and reads the counter +and original `sys_requests` response. The source authority fixture records +callbacks; these cases do not qualify authenticated directory CAS, separate +processes or provider failure. + +| Source and focused verification | Recorded value | +| --- | --- | +| Baseline | `9c99d3ac620821fa21e11fe462b1046cca898305` | +| Rust/Cargo manifest | 595 tracked/nonignored `.rs`, `Cargo.toml` and `Cargo.lock` paths; sorted SHA256, two spaces, relative path lines. SHA256 `f2d51309c3797d0a27e57b7dcf5da8459e66835f9911529e368c1e16fcadfe22`. | +| Public durability selection | `cargo test -p cellule-runtime --test runtime --all-features --locked runtime::lifecycle::durability::`: 21 passed, 176 filtered, none ignored; includes five new cases. | +| Node library selection | `cargo test -p cellule-runtime --lib --all-features --locked node::`: 89 passed, 406 filtered, none ignored. | +| Distinct scoped cases | 110 passed. The separate initial five-case retirement run is included in the public selection, not counted again. | +| Runtime lint | `cargo clippy -p cellule-runtime --lib --test runtime --all-features --locked -- -D warnings`: passed. | +| Host compatibility | `cargo check -p cellule-host --lib --test node --all-features --locked`: passed; this is compilation, not host scenario execution. | +| Runtime API documentation | All features, warnings denied; passed. | +| Static gates | Format, diff whitespace, boundaries/layout, 110 Rust snippets, 1150 Markdown links and 28 SQL/peer assertions with 566 links passed. | + +Commands use Rust/Cargo 1.97.0, the lockfile, `CARGO_INCREMENTAL=0` and +`$HOME/Workspace/crabbuild-target/cellule-f9383af7-fleet-operations`. +The source manifest is retained at +`/tmp/cellule-fleet-member-retirement-rust-cargo-manifest.txt`. Persisted IDs, +object paths, receipt fields, LTX formats and signed message bytes are unchanged. +Additional node, lint, host compilation and documentation logs are retained +under `/tmp/cellule-fleet-member-retirement-{node,clippy,host,docs}.log`. + +Baseline `9c99d3a` passed Rust (36937784705), capacity (36937784385), contract +(36937784326), website (36937784644), fuzz (36937784399), model (36937784598) and +Compose smoke (36937784915 / 110622398881). Compose routing +110622399162 remained running when inspected. A model workflow's success does +not establish that optional broad campaigns ran. Earlier failure evidence +remains below; these baseline results do not qualify the new implementation. + +The host's automatic rotation still uses ordinary best-effort closure. Requested +maintenance rotation, durable follower enrollment before recruitment CAS, +complete role observation and journal retirement remain unwired. A member +proof alone must not complete SettleRoles or Finalize. Remaining W6–W10 matrices, +sustained convergence, all scenario deliverables, process/provider/mixed-binary +qualification and operator rollout/runbooks remain part of the active goal. + +## October 1 2026 managed reader producer checkpoint + +`CellNode::install_fleet_reader_enrollment` binds the existing manager to the +shared fleet journal before start and the first activation. Configured fleet +hosts require the owned binding before readiness. Ordinary peer hints and +prepared activation use the same source selection, admission and native opening +path. Both signed physical boots and fresh intent revisions participate in +Pending acceptance; only New starts the first opening. + +The manager retains 32 bounded finite activation jobs and charges job/record +storage to the existing runtime byte ledger. Waiter cancellation cannot cancel +accepted opening or result publication. Exact original source, opening/closure +evidence and independent native/publication errors remain inspectable through +`enrollment_completion`. A lost Established reply replays original evidence +before ordinary refresh. Policy eviction, refresh fencing, explicit removal and +shutdown join canonical closure before Retired. Failed or cancelled retirement +keeps the fenced view and original event for retry. Missing acceptance plus +proof this owner never began opening settles only the local entry; removal does +not create another Pending request. Unobserved establishment and unjoined task +failure remain blocked. + +Nine new public cases use the durable SQLite journal with actual readers, +signed advertisements and the host lifecycle. Seven cover Pending before native +work, a cancelled activation, lost acceptance/Established/Retired replies, +independent backend reconstruction, cancelled removal, a real native VFS stall +across a host shutdown deadline, local cordon refusal with independent retained +errors, and journal cordon winning between intent observation and acceptance. +Two cover required startup binding and shutdown with a missing binding. A typed +counter query checks receipt-bound readback. These are local role fixtures, not +process/provider or complete fleet-observer qualification. + +Intermediate verification caught two relevant issues. An exact-version registry +scan correctly conflicted while the owned producer advanced its version; the +fixture now retries only that typed conflict with a bounded fresh scan. Bulk +shutdown had removed healthy siblings before the last native join, contradicting +the established cancelled-shutdown inventory contract. The final implementation +retains the complete collection until all native joins finish. Closed reader +queries return Fenced; the new fixture initially expected RuntimeClosed and was +corrected to the existing contract. Profiles and production refusal semantics +were not weakened. + +| Final source and focused evidence | Recorded value | +| --- | --- | +| Baseline | `c67c4e5d291e0c908277556788b3237e52af3c4f` | +| Rust/Cargo manifest | 593 tracked/nonignored `.rs`, `Cargo.toml` and `Cargo.lock` paths; sorted SHA256, two spaces, relative path lines. SHA256 `06a97bce23936f849659154cc923dfedc980ca3d0821cc8044734d44dd24b234`. This includes five existing nested Cargo locks omitted by the previous manifest, plus three new Rust modules. | +| Complete public host node target | 72 passed; none filtered or ignored. | +| Complete fleet operations example target | 60 passed; none filtered or ignored. | +| Public runtime reader selection | 14 passed; 177 filtered; one isolated RustFS case ignored. | +| Application reader integration selection | 5 passed; 40 filtered; none ignored. | +| Distinct scoped cases | 151 passed; nine new example cases. | +| Host all-target/all-feature Clippy | Passed with warnings denied. | +| Host/runtime API documentation | All features, warnings denied; passed. | +| Static gates | Format, diff whitespace, boundaries/layout, 110 Rust snippets, 1150 Markdown links and 28 SQL/peer assertions with 566 links passed. | + +Commands use Rust/Cargo 1.97.0, the lockfile, `CARGO_INCREMENTAL=0` and the +checkout's Workspace target directory. Raw final command logs and the source +manifest are retained under `/tmp/cellule-fleet-reader-producer-*` on the execution +host. The host remove/shutdown APIs now return Result so durable retirement +failure reaches the canonical facility drain. Persisted production IDs, object +paths and message formats are unchanged. The example adds an additive typed +counter query and uses freshly compiled example releases. + +Baseline `c67c4e5` passed workspace/MSRV (36933281069 / 110607525964 and +110607525809), object/follower capacity (36933281010 / 110607525081 and +110607524697), contract (36933281080), website (36933281071), fuzz (36933281047), +fast/negative model (36933281034) and Compose smoke (36933281024 / 110607805140). +Broad model/simulator were skipped. Its routing job 110607804555 remained +running when inspected. Earlier `e472441` routing 110587330915 is authoritatively +Cancelled, with smoke passed. Historical failure evidence below remains retained; +later green baselines do not establish the causes of those failures or qualify +this new source. + +Next, wire follower production and complete revisioned role observation, then +consume canonical replacement/failed-process evidence for evacuation and node +finalization. The movement commands still declare incomplete role coverage. +All remaining W6–W10 matrices, sustained convergence, four scenario deliverables, +process/provider/mixed-binary qualification and operator rollout/runbooks remain +part of the active goal. + +## October 1 2026 exact reader enrollment source checkpoint + +`ReadReplicaSource` is opaque metadata prepared through canonical authority and +signed live-boot validation. It carries the exact target, incarnation, code, +schema, owner, epoch, root, physical source node and fleet. Preparation opens no +view and reserves no native resources. `CellReadReplica::open_source` opens the +original pinned root through the same admitted native path as ordinary opening. +Newer publication cannot replace it. Installation and query release still +check authority and the original signed boot identity. Ordinary opening retains +its admission-before-provider-I/O order. + +The host manager prepares selected sources and offers initial `activate_source` +through its existing activation lane. An already installed view is refused +without refreshing it; ordinary authenticated hints retain refresh behavior. +After entering the lane, a retained prepared activation joins opening on manager +closure, closes an uninstalled view and returns RuntimeClosed. The adapter must +retain its accepted future across transport waiter cancellation. Source metadata +cannot by itself authorize an enrollment or establish durable role coverage. + +Three new public cases verify old-root SQL readback after newer publication, +normal refresh to the newer root, stale-source and cordon refusal, manager +initial-only behavior and native-open closure. The last case pauses a real VFS +range read only while the reader's SQL job ledger is charged, closes the manager, +then verifies opening remains owned until release and all charges settle. +The first runtime fixture initially invoked source drain twice after deliberately +releasing it to test fencing; the repeated call returned CellDraining. The final +fixture joins the already-released runtime directly. No production error, +profile or assertion was weakened. + +| Final source and focused evidence | Recorded value | +| --- | --- | +| Baseline | `e472441c99ac7e3c5da95809a874f5138cfe37d0` | +| Rust/Cargo manifest | 585 tracked/nonignored paths; sorted SHA256, two spaces, relative path lines. SHA256 `5da2afb2fc2f7e2cb0c6f52f36f7e6d8d4941ca835702609d94299e61438b8cf`. | +| `cargo test -p cellule-runtime --test runtime --all-features --locked read_replica::` | 14 passed; 177 filtered; one isolated RustFS case ignored. | +| `cargo test -p cellule-host --test node --all-features --locked` | 72 passed; none filtered or ignored. | +| `cargo test -p cellule-host --example fleet_operations --all-features --locked` | 51 passed; none filtered or ignored. | +| `cargo test -p cellule-app --test integration --all-features --locked host::replicas::` | 5 passed; 40 filtered; none ignored. | +| Distinct scoped cases | 142 passed. One runtime and two host cases are new. | +| Selected host/runtime libraries, runtime/node targets and example Clippy | All features, warnings denied; passed. | +| Host/runtime API documentation | All features, warnings denied; passed. | +| Static gates | Format, diff whitespace, boundaries/layout, 109 Rust snippets, 1149 Markdown links and 28 SQL/peer assertions with 566 links passed. | + +Commands use Rust/Cargo 1.97.0, `CARGO_INCREMENTAL=0` and this checkout's +Workspace target directory. Local fixtures use in-memory CAS and actual SQLite; +the ignored RustFS case supplies no provider evidence. The opaque type and API +inventory are additive; persisted IDs, descriptors and peer formats are unchanged. + +Baseline CI passed workspace/MSRV (36927118058), object/follower capacity +(36927117222), contract (36927117305), website (36927117280), fuzz (36927117240), +fast/negative TLC (36927117235) and Compose smoke (36927117229 / 110587330500). +Broad TLC/simulator were skipped. Routing (36927117229 / 110587330915) remained +running when inspected. These results qualify the baseline only, do not identify +the cause of earlier failures, and do not qualify this new source. + +Next, wire every reader producer to Pending acceptance before initial activation, +retain finite completion/result ownership, reconcile ambiguous acceptance and +publish checked retirement after joined closure. Wire follower producers and +complete observer coverage before enabling maintenance finalization. The full +plan, all W6–W10 gaps and their qualification requirements remain active. + +## October 1 2026 joined reader closure checkpoint + +`CellReadReplica::close_and_join` fences admission, detaches the current +snapshot across every clone and joins accepted queries, authority reads, +refreshes and native SQLite opens. Snapshot and accepted-operation lifetimes +share one atomic closure gate. A native open returns the complete owned +snapshot, so cancelling its refresh waiter cannot hide an uninstalled view. +Resource fields drop before the lifetime signals completion. Historical +`receipt()` metadata survives detachment without asserting current authority. + +The host reader manager retains each view until joined removal succeeds. +Cancelled remove/shutdown waiters leave ownership inventoried and joinable. +Shutdown fences all views first, then joins at most 16 concurrently through +the existing activation and host drain lanes. A host deadline leaves the node +Draining until accepted native work settles; retained peer clones cannot keep +snapshot charges after successful closure. + +Four new public runtime cases cover retained clones, old queries across a +refresh, cancelled query/close waiters, a cancelled refresh blocked inside an +actual native-open VFS read, and a delayed authority/readiness reply. A public +host case covers cancelled removal/shutdown and a host drain deadline; the +existing reader inventory case also retains peer clones through shutdown. +Fixtures check real SQLite paths and memory, retained-byte, descriptor, worker +and private disk ledgers. Initial fixtures shared the process-wide default disk +budget, so zero reader disk assertions included the source's charges. Separate +node budgets corrected those fixtures; no production threshold was changed. + +| Final source and focused evidence | Recorded value | +| --- | --- | +| Baseline | `135492e75cfb142e4b918ce63d3e5c97ba103453` | +| Repository Rust/Cargo manifest | 585 tracked/nonignored paths; sorted SHA256, two spaces, relative path lines. Manifest SHA256 `23f6d06cc2a3788db971204ea2f2acecc325deecbd926c85d6fd5509247b4811`. | +| `cargo test -p cellule-runtime --test runtime --all-features --locked read_replica::` | 13 passed; 177 filtered; one isolated RustFS case ignored. Four cases are new. | +| `cargo test -p cellule-host --test node --all-features --locked` | 70 passed; none filtered or ignored. One case is new. | +| `cargo test -p cellule-host --example fleet_operations --all-features --locked` | 51 passed; none filtered or ignored. | +| `cargo test -p cellule-app --test integration --all-features --locked host::replicas::` | 5 passed; 40 filtered; none ignored. | +| Distinct scoped cases | 139 passed; the RustFS case remains ignored. | +| `cargo clippy -p cellule-runtime -p cellule-host --lib --test runtime --test node --example fleet_operations --all-features --locked -- -D warnings` | Passed after naming the test fixture's query-gate type. No lint was suppressed. | +| `RUSTDOCFLAGS='-D warnings' cargo doc -p cellule-host -p cellule-runtime --all-features --no-deps --locked` | Passed. | +| Static gates | Format, diff whitespace, boundaries/layout, 108 Rust snippets, 1149 Markdown links and 28 SQL/peer assertions with 566 links passed. | + +Commands use Rust/Cargo 1.97.0, `CARGO_INCREMENTAL=0` and +`$HOME/Workspace/crabbuild-target/cellule-f9383af7-fleet-operations`. +These local fixtures use in-memory CAS and actual SQLite. The ignored RustFS +case supplies no provider evidence. Broad workspace and process/provider +qualification remains assigned to CI and controlled environments. + +This implements a reader closure prerequisite for W3/W7. Reader/follower +enrollment producers, complete observations, replacement-policy evidence, +durable retirement and finalization remain required. The last reader receipt +alone cannot settle a fleet enrollment or establish safe node removal. + +Baseline `135492e` passed workspace/MSRV (36921861844), contracts (36921861824), +website (36921861940), fuzz (36921861753), fast/negative TLC (36921861711), +follower capacity (36921862099 / 110569602571) and Compose smoke +(36921861765 / 110569778663). Broad TLC/simulator were skipped; routing remained +running when inspected. Object capacity (36921862099 / 110569602323) failed +repeat 2: hot rate 2 completed 59 of 60, with one `scheduler_late` arrival, and +the driver reported `hot: no fully served capacity point`. Repeat 1 passed; +repeat 3 did not run. The original log is retained at +`/tmp/cellule-fleet-ci-36921862099-object.log` and the full original artifact at +`/tmp/cellule-fleet-capacity-36921862099`. The qualification profile remains +unchanged. A passing workspace on this baseline does not identify the cause +of the earlier controller-restart failure or qualify this new reader source. + +## October 1 2026 boot enrollment and startup barrier checkpoint + +Configured fleet hosts hold the runtime's existing writer, reader and follower +gate before a lease can expose acquisition. `with_fleet_startup_intent` validates +the retained physical row before runtime construction. Confirmation reads the +exact Established boot and current intent through the new required atomic +`FleetEnrollmentJournal::load_boot` contract. It records checked intent while +leaving admission held. `start()` checks required facilities/task supervision +before removing the hold. Cordon/drain and pressure remain independent and +sticky; changing admission preserves the original measurement timestamp. + +A maintenance reboot enters `NodeState::Maintenance`: `is_ready()` stays false, +while `is_management_ready()` exposes authorized action and inspection paths. +The executor remains bound to this host's physical node, scope and session. +Lifecycle/startup locking rejects a delayed proof older than a newer confirmed +intent and rejects reads completing after shutdown. Applications still own +authorization and delivery of intent transitions after startup confirmation. + +The executable now accepts Pending before actual signed canonical directory +creation, publishes checked Established evidence, derives the lease guard from +that advertisement, confirms the boot and starts the host. Existing ambiguous +acceptance cannot repeat an unobserved creation. Independent SQLite reopening +adopts a lost Established reply with the original evidence and time. Boot +advertisements publish zero receive capacity until readiness; all samples are +the actual runtime classifier output. + +Application boot owners remain retained through joined runtime shutdown, +guard fencing, canonical directory withdrawal and registry retirement. A +permanent exact-session tombstone adopts an already committed withdrawal; +absence/expiry alone cannot settle an obligation. Replay checks the original +spec and Established evidence and preserves a committed retirement. Both real +movement commands require all three boot rows Retired at the final registry +revision in addition to their existing resource-ledger checks. + +Seven public startup cases cover missing/Pending/foreign rows, contradictory +same-revision modes, wrong Active boots, lost acceptance/Established replies, +required-component checks, racing cordon, maintenance reboot management, +delayed stale reads/shutdown and canonical withdrawal/retirement replay. Tests +use actual local hosts, directory signing/validation and SQLite transactions. +Initial fixtures incorrectly used a 60-second advertisement lifetime, mismatched +directory image, a timestamp preceding the runtime sample and a stale registry +version. Existing validators rejected them. A closed-journal assertion expected +an IO error; the final test verifies its original `RuntimeClosed` source. Those +fixture corrections did not alter any production qualification profile. + +| Final source and focused evidence | Recorded value | +| --- | --- | +| Baseline | `5fc341bc31b098b3bacee3b7a8e98aca4f4c1d07` | +| Complete repository Rust/Cargo manifest | 588 tracked/nonignored paths; sorted SHA256, two spaces, relative path lines. Manifest SHA256 `fbb3715e14316c2943866ab4c2a07a6b1d5517a90ed8db5b1cd850e9c320aa7d`. | +| Runtime admission library selection | 4 passed; 491 filtered, none ignored. Two cases are new. | +| Complete public host node target | 69 passed; none filtered or ignored. | +| Complete executable example target | 51 passed; none filtered or ignored. Seven cases are new. | +| Distinct scoped cases | 124 passed. | +| Host/runtime library, public node and example Clippy | All features, warnings denied; passed. | +| Host/runtime API documentation | All features, warnings denied; passed. | +| Actual executable commands | Overload and controller-restart: each released/activated/retired two Cells, checked two receipts, joined three nodes and retired three boots. Peak restore charge 2,550,136,832 bytes under the unchanged 8-GiB/two-move budget. Replacement also adopted two lost replies at epoch 2 and joined two expired unused reservations. | +| Static gates | Format, diff whitespace, module/boundaries, 108 Rust snippets, 1148 Markdown links, and 28 SQL/peer assertions with 565 links passed. | + +Commands use Rust/Cargo 1.97.0, the lockfile, `CARGO_INCREMENTAL=0` and the +checkout's Workspace target directory. This finite local reference has no +production heartbeat provider, complete reader/follower enrollment producers, +complete observer, process-fault campaign or distributed journal qualification. +The full plan remains active: role evacuation/finalization, remaining primitive +and failure matrices, continuous-load convergence, runnable maintenance/receiver +loss and W9–W10 deployment/operational qualification remain required. + +Baseline CI `5fc341b` passed MSRV (36914686533 / 110545650679), contract +(36914686490), fuzz (36914686514), website (36914686782), fast/negative TLC +(36914686630), object/follower capacity (36914686675) and Compose smoke +(36914686644 / 110546093110). Broad TLC/simulator jobs were skipped. Routing +(36914686644 / 110546092620) was still running when recorded. +Workspace (36914686533 / 110545650944) failed the real controller-restart example +at its strict two-lost-release assertion: 43 passed, one failed. Original log is +retained at `/tmp/cellule-fleet-ci-36914686533-workspace.log`. The failing report +did not identify which condition failed. The final source retains every assertion +and the same three-second profile and now includes the complete bounded report, +lost-reply count and original node completion in failure output. Local complete +example runs pass, but they do not establish the CI failure's cause or resolution. +This source requires its own CI; earlier routing failure evidence remains below. + +## October 1 2026 public primitive maintenance checkpoint + +Five new public cases exercise `release_maintenance_cell_at` with actual typed +Effect and Activity claims. They wait for actor quiescence, prove release stays +pending while the exact live lease validates and authority remains Serving, +and check new claims and ordinary reads are refused. Native acknowledgement, +Activity validation/extension/completion and exact-root release use the existing +registry, actor, worker and publication paths. No production API changed. + +Effect cases cover normal acknowledgement, a dropped maintenance waiter and +normal lease expiry. An authenticated destination publication precedes source +release. Ordinary receiver activation preserves the original command resolution +and destination inbox receipt. A late acknowledgement loses its lease. Unclaimed +work resumes through the registered native Effect driver. The expiry case keeps +the existing retry backoff, verifies Ready state, reclaims with a new token and +attempt 2, and delivers to the same inbox without applying the command twice. + +Activity cases cover extension/completion and normal expiry. A settled completion +replays as Duplicate on the receiver; an expired token is rejected before and +after normal reclamation. The retry preserves activity ID and definition digest, +increments its attempt and replaces its token. The registered native Activity +driver completes an unclaimed blocking activity. Running Workflow waits retain +their run/state, accept a signal on the receiver and deduplicate its replay. +Preserved due timers advance through the registered Tick after restoration. +Every receiver and source joins with zero retained bytes. + +Initial fixture synchronization incorrectly awaited an advisory cached work +sample during the separate maintenance preflight read; those runs timed out. +The final cases synchronize on installed quiescence and directly validate the +lease, bounded pending release and authority state. Initial expiry assertions +also incorrectly expected reclaimed work ahead of an already-ready Activity and +an immediately claimable Effect despite its existing backoff. Correcting those +fixtures preserves canonical ordering/backoff and checks actual reclaim and +deduplication. No runtime invariant, profile or threshold was changed. + +| Source and focused evidence | Recorded value | +| --- | --- | +| Baseline | `e7aa6923bce13d4cc97e069016040a5e1e356fae` | +| Complete repository Rust/Cargo manifest | 586 paths, sorted from tracked and nonignored files; each line is SHA256, two spaces, relative path. Manifest SHA256 `c42597897d612a37a112bab937f6785c725ebf0aebf2bbeb1220d25584ddfff2`. | +| Public maintenance selection, primitives + protocol | 9 passed: six primitives (43 filtered), three protocol (29 filtered); none ignored. Five cases are new. | +| Complete public client target | 30 passed; two manual throttled-provider measurements ignored; none filtered. | +| Public Workflow API selection | 7 passed; 42 filtered, none ignored. | +| Distinct selected cases across these commands | 41 passed; maintenance/client/workflow selections overlap in the five new cases. | +| Selected primitives/protocol Clippy | All features, warnings denied; passed. | +| Static gates | Format, diff whitespace, boundaries/layout, 108 Rust snippets, 1144 Markdown links and 28 SQL/peer assertions with 564 links passed. | + +Commands used Rust/Cargo 1.97.0, all features and the lockfile, with +`CARGO_INCREMENTAL=0` and this checkout's Workspace target directory. Fixtures +use in-memory authority and local runtimes. They do not qualify distributed +providers, process interruption, continuous traffic, Cron dispatch or external +Blob owners. The full W6–W10 and earlier enrollment/observation gaps remain open. + +CI for the baseline above now passed workspace/MSRV (36912160616), contract +(36912160634), website (36912160311), fuzz (36912160309), TLC fast/negative +(36912160379) and object capacity (36912160550 / 110537198649). Follower capacity, +Compose smoke and routing remained running when recorded. Broad TLC/simulator +jobs were skipped. These results belong to that pushed baseline; the prior +routing failure remains recorded below and this new test source requires its +own CI. + +## October 1 2026 busy demand and host acceptance checkpoint + +The actor exposes a separate `maintenance_cost` from validated per-Cell LTX +limits. It uses the same conservative native/cache/descriptor/job envelope and +configured disk ceiling as receiver admission, independently of the measured +worker sample. Mutations still invalidate ordinary cost, timestamps and stability; +the configured envelope supplies no readiness or release proof. Blob owners have +no busy envelope and refuse release before foreground closure. + +The driver selects this envelope only for the exact physical node and boot of a +retained, unexpired Evacuating maintenance operation. The pure planner also +requires an explicit maintenance demand and a draining donor. Fresh authenticated +collection/source/receiver evidence, receiver projection and the shared two-move/ +8-GiB budget remain required. Local execution, publication, external lease and +unknown primitive conditions are deferred to the canonical source barrier; +foreign reader/follower/facility blockers are retained. Ordinary pressure/count +movement still requires settled worker samples. A Draining advertisement or +pressure alone cannot force busy work to move. + +Planner-input domain v3 hashes both costs (presence and every admission dimension), +role, code/schema identity and individual blocker classes alongside the existing +capture interval, generation, source boot and root. Retained older digests remain +opaque. Focused digest cases check each new field and expired collection refusal. + +The public host case exercises actual accepted SQL work under Cordon and explicit +action 8, closes new foreground admission, joins the original durable receipt, +and restores it on a second public host through ordinary exact-root acquisition. +It checks normal completion, a dropped action waiter and lost result publication. +Release replay preserves the exact original proof after the receiver owns the +Cell. Both hosts join with zero active Cells and retained bytes. These fixtures +use in-memory authority and local lease guards; they do not qualify a distributed +lease provider or complete role evacuation. + +The driver model uses real SQLite journal transactions with signed synthetic +observations/effects. It plans busy rows with no measured cost/time/stability, +charges the distinct peak cost, retains permits after a lost action-8 reply and +adopts the exact stored result without issuing ordinary Release or repeating the +effect. Negative cases cover draining/pressure without maintenance, Blob roles, +foreign follower blockers, missing peak cost and missing published position. +This model is distinct from the public host and leased-node example evidence. + +Initial digest fixtures used invalid empty module/schema inventories and zero +placement totals; the existing validators refused them. Correcting the fixtures +supplied valid signed inventories and capacities. The lost-response fixture first +expected an OutcomeUnknown blocker after a transport error; the actual contract +retains the unresolved dispatch phase and original transport failure. Its corrected +assertion checks phase 13, absent release proof, refused retirement, retained +permits and exact later adoption. No production gate or qualification profile was +weakened. A private-method fixture assertion and unsupported Default initializer +were also corrected before execution. + +| Source and evidence | Recorded value | +| --- | --- | +| Baseline | `5d98eb0e7fa961c5420022c1da07e8cad16e3433` | +| Implementation and main sync | `df55295`; merged `origin/main` (`0dc04a6`) as `869f0e4`. Only the website checker conflicted; the upstream feature-group implementation was retained. The merge introduced no Rust-source delta. | +| Source manifest | 579 Rust/Cargo files; SHA256 `0ffe6084e1f2bcc0f65b70695b42e37892cb7f2733b442ba9889db8167885ef2` | +| Runtime library placement / actor inventory | 16 passed (477 filtered) / 7 passed (486 filtered). | +| Runtime public placement / ownership inventory | 10 passed (16 filtered) / 5 passed (182 filtered). | +| Runtime maintenance selection | 37 passed: 30 library (462 filtered), four Queue (43 filtered), three runtime (184 filtered). One isolated RustFS library case ignored; no provider evidence. Two selected cases overlap the placement/ownership commands. | +| Host digest library | 3 passed; none ignored or filtered. | +| Complete executable example test target | 44 passed; none ignored or filtered. | +| Host fleet node selection | 31 passed; 38 filtered. The new public case exercises three waiter/publication modes. | +| Application local reader selection | 5 passed; 40 filtered, including both cases that timed out in prior workspace CI. This scoped result does not resolve broad CI failure. | +| Selected all-feature runtime/host library/tests/example Clippy | Passed with warnings denied. | +| Host/runtime all-feature API documentation | Passed with warnings denied. | +| Website Rust example checker after merge | All examples compiled across 12 authored guides using the upstream checker. | +| Static gates | Format, diff whitespace, boundaries/layout, 108 Rust snippets, 1143 Markdown links and 28 SQL/peer assertions with 563 links passed. | + +These final checks ran on the merged Rust source. Cargo used Rust 1.97.0, +`CARGO_INCREMENTAL=0` and the checkout target directory recorded in the prior +checkpoint. The tables name focused command selections; broad workspace, process, +provider and mixed-binary campaigns remain separate proof obligations. + +The prior `b8ec303` routing job (36899568202 / 110495509870) is now terminal: +**failed** `object_only/forwarded_command/c16`. Across its four frozen runs, +median candidate p99 was 343.037178 ms versus baseline 305.3395325 ms (ratio +1.1234614), exceeding the unchanged 1.10 limit. Throughput ratio was 0.9706314 +and p95 ratio 1.0280569. The retained manifest identifies baseline `c51dd12` and +synthetic candidate merge `970d874`; this is evidence for that CI source, not +this later checkpoint. Logs and artifact remain at +`/tmp/cellule-fleet-ci-36899568202-routing.log` and +`/tmp/cellule-fleet-routing-36899568202` (remote artifact 11185357976). +Correctness passing inside each benchmark does not erase this performance +failure. No threshold or profile was changed. The updated PR must qualify its +own source and the failure still requires investigation. + +Full W6 remains incomplete: every primitive's public acceptance/fault matrix, +Blob external owners and continuous-load qualification still need coverage. +Enrollment/startup barriers, complete role observations/evacuation, finalization, +remaining failure adoption, sustained convergence, runnable maintenance/receiver- +loss scenarios and process/provider/mixed-binary qualification remain required +by the full plan. The previous canonical busy release checkpoint below retains +its original source and proof limits. + +## October 1 2026 canonical busy release checkpoint + +`CellRuntime::release_maintenance_cell_at` now owns an exact-source maintenance +request independently of its reply waiter. It uses the existing actor movement +permit, shared admission capability, serialized worker, publication barrier and +canonical deactivation/release path. Foreground work closes first. A readiness +read is serialized behind accepted SQL without waiting for the ingress queue to +become idle. Native completion then closes, accepted work/publication joins, and +a second fresh readiness read must pass before canonical release is confirmed. +Lease renewal remains active until confirmation. A newly extended live lease +restores native completion only and waits within the preflight deadline. + +`MaintenanceCellRelease::Refused` proves that this request started no canonical +release and retains any source error. Quiescence remains sticky if installed; +native completion is restored on the same admission capability. The deadline +bounds preflight, not confirmed release. A timed-out request's owned read is +joined; its effect ID cannot classify a newer request. Blob owners remain +unproven and are refused before foreground closure. + +The journal appends explicit phase 13 (`MaintenanceReleasing`) and action 8 +(`ReleaseMaintenance`). Exact Evacuating maintenance identity is required before +dispatch. The host executes that policy through the new runtime method and +preserves definite refusal versus unresolved release. Controller reconstruction, +action replay, atomic absence and retained source inspection keep the exact +release kind; ordinary phase/action values and policies remain unchanged. +Readers must understand the new tags before producers are enabled. Mixed-binary +qualification is still required. + +Public SQL and Queue cases cover accepted mutation settlement, exact final-root +release and ordinary receiver restoration, original outcome resolution, surviving +unclaimed Queue work, native validation/acknowledgment, dropped release waiters, +and deadline refusal followed by native completion and a later release. The SQL +case also exercises overlapping expired and newer maintenance requests. These +in-process cases do not establish every late-read interleaving or provider fault. +Pure action cases cover intent binding, codecs, distinct action keys/results, +unknown/absence permit retention, controller replacement and preserved ordinary +release policy. The executable driver model checks explicit action 8 dispatch. + +The first public run found that an actor guard refused native completions during +all transfers. The guard now retains that refusal for ordinary idle transfers +and routes maintenance completion through the canonical kernel admission check. +The corrected run passes without weakening expected evidence. Earlier fixture +compile errors (private re-export scope and an invalid test control-state variant) +were corrected before the passing run; they are not qualification results. + +| Source and evidence | Recorded value | +| --- | --- | +| Baseline | `b8ec303942d112ce8ca36dc1a6761c5191574d49` | +| Source manifest | 578 Rust/Cargo files; SHA256 `530c407930cbab52a16876136a1cf16ba8e20921e485be6beb02a8431e5865a0` | +| `cargo test -p cellule-runtime --lib fleet::operations --all-features --locked` | 65 passed; 427 filtered. | +| `cargo test -p cellule-runtime --lib --test primitives --test runtime maintenance --all-features --locked` | 35 passed: 29 library (462 filtered), four Queue (43 filtered), two SQL (184 filtered). One isolated RustFS library case ignored; no provider evidence. Eight operation tests overlap the 65-test command. | +| `cargo test -p cellule-host --example fleet_operations --all-features --locked` | 41 passed; none ignored or filtered. | +| `cargo test -p cellule-host --test node node::fleet --all-features --locked` | 30 passed; 38 filtered. These existing public host cases do not qualify the new busy maintenance action. | +| `cargo clippy -p cellule-runtime -p cellule-host --lib --tests --example fleet_operations --all-features --locked -- -D warnings` | Passed after removing an unused lint expectation; no lint or qualification bound was weakened. | +| Static gates | Format, diff whitespace, boundaries/layout, 108 Rust snippets, 1143 Markdown links, and 28 SQL/peer assertions with 563 validator links passed. | + +Cargo commands use Rust 1.97.0, `CARGO_INCREMENTAL=0` and +`$HOME/Workspace/crabbuild-target/cellule-f9383af7-fleet-operations`. Proof is +focused in-process behavior with real SQLite and in-memory authority, plus the +existing local leased-node overload/restart examples. Broader suites belong to +CI or isolated qualification snapshots. + +New demand collection still requires ordinary settled eligibility. Busy demand +planning, public host acceptance/fault coverage, Effect/Activity and remaining +primitive acceptance cases, Blob owners, enrollment/startup intent barriers, +reader/follower evacuation, finalization, sustained convergence and process/provider +qualification remain open. This checkpoint does not complete W6 or the full plan. + +Baseline `b8ec303` CI: MSRV, contract, website, decoder fuzz, TLC fast/negative, +both capacity jobs and smoke passed. Routing remained pending. Workspace tests +failed in `pending_reader_activation_retains_a_new_publication_hint` and +`publication_hints_reach_readers_beyond_the_activation_concurrency`, both timing +out at `crates/cellule-app/tests/host/replicas.rs:285`. The original failed log is +retained at `/tmp/cellule-fleet-ci-36899568285-workspace.log`. Qualification +thresholds are unchanged; the updated PR must qualify its own source in CI. +Skipped broad simulator/TLC jobs supply no evidence. + +## October 1 2026 foreground quiescence checkpoint + +`CellRuntime::quiesce_cell_at` installs a sticky exact-generation foreground +gate without interrupting previously actor-admitted SQL. It checks runtime +boot, generation, incarnation and ownership epoch before closing the shared +admission capability. Unavailable publisher state refuses conservatively. +This API is a local runtime boundary: applications must retain and authorize +maintenance intent before calling it. It supplies no release proof. + +Native Queue/Effect lease commands and validation, and Activity completion, +extension and validation have private registry admission classification. New +claims and foreground work close; original outcome resolution stays available. +All calls use the existing bounded mailbox, worker, fencing, transaction and +durability gates. Duplicate binding errors now preserve the original handler, +preventing an application replacement from inheriting native classification. +Descriptor/release bytes and persisted IDs are unchanged. New background +hydration and compaction stop, while required inventory can refresh. + +The serialized worker reports separate maintenance readiness alongside ordinary +idle readiness. It retains simultaneous live lease classes, refuses malformed +lease/schema/time observations, and leaves expired lease rows unchanged. Pending +durable messages, Effects, timers and Workflow waits can be carried by an exact +root after claims close. Blob stream/upload/pin coverage is explicitly blocked. +The actor exposes optional readiness and its quiescence flag; the planner hashes +these under input domain v2 without rewriting retained attempt digests. + +| Source and evidence | Recorded value | +| --- | --- | +| Baseline | `7be792bf5a71fc7d6ec9b82d615f512631ed4833` | +| Source manifest | 573 Rust/Cargo files; SHA256 `67b06c0b29f8dc6e17d4ec3e008ea8f311f0eb3c594ea742d8eb82e518edd91c` | +| `cargo test -p cellule-runtime --lib --test primitives --test runtime maintenance --all-features --locked` | 23 passed: 21 library (462 filtered), one Queue (43 filtered), one accepted SQL (184 filtered). One library RustFS case ignored because its isolated provider environment was not supplied; it establishes no provider evidence. | +| `cargo test -p cellule-host --example fleet_operations --all-features --locked` | 41 passed; none ignored or filtered. | +| `cargo test -p cellule-host --test node node::fleet --all-features --locked` | 30 passed; 38 filtered. | +| `cargo clippy -p cellule-runtime -p cellule-host --lib --tests --example fleet_operations --all-features --locked -- -D warnings` | Passed. | +| Static gates | Format, boundaries/layout, 108 Rust snippets, 1143 Markdown links, 28 SQL/peer assertions and 563 validator links passed. | + +The Queue case checks all source identity mismatches before closure, sticky +replay, refusal of new sends/claims/info and raw callbacks, exact validation, +acknowledgment, live-to-settled inventory refresh and joined shutdown with zero +retained bytes. It also attempts duplicate application replacement bindings and +verifies unchanged release bytes plus the original native handlers. The SQL +case closes admission while an accepted handler is blocked, then joins its +commit, resolves its original outcome and shuts down with zero retained bytes. + +The first public Queue run failed because the test's raw query requested a +zero-byte reservation; the existing byte validator correctly refused it before +admission. Correcting that fixture to a valid reservation produced the passing +run above. No production bounds or expected evidence were weakened. + +Commands use Rust 1.97.0, the checkout target directory recorded below and +`CARGO_INCREMENTAL=0`. These are local focused checks with real SQLite and +in-memory authority. Effect/Activity maintenance public acceptance cases, +Blob owner coverage, journal-bound busy release, enrollment/startup barriers, +complete reader/follower evacuation and finalization, continuous-load +maintenance and process/provider qualification remain open. The full plan is +still incomplete; this checkpoint does not certify W6 or complete maintenance. + +On baseline `7be792b`, workspace, MSRV, contract, website, decoder fuzz, both +capacity jobs and smoke passed. Routing remained pending at inspection. The +older object-capacity failure remains recorded below; baseline successes do not +qualify the new source, whose full gates must run in CI after push. + +## October 1 2026 maintenance admission checkpoint + +Maintenance Cordon now runs through the public node-owned action executor. +Exact journal acceptance binds the physical node, boot, operation and retained +intent before closing the existing writer, new-reader and new-follower role +gate. Existing Cell owners remain available; this step releases no role and +takes no shutdown lane. Unsupported maintenance role, finalization and +inspection effects fail before acceptance or inventory work. + +The driver dispatches Cordon from Requested, commits Cordoned only after a +checked durable result, then commits BeginEvacuation on a later pass. Lost +replies, unconfirmed publication and Unknown retain the phase and intent. +Maintenance has a bounded deadline share and a separate original error in +`maintenance_failure`; an ambiguous timeout rereads the journal and epoch. +Stopping optional scheduling still allows these intent steps. Deadline expiry +keeps the cordon, prevents new maintenance allocations, and does not discard +accepted work. New movement deadlines also respect the operation deadline. + +Retained intents now overlay signed placement samples before donor selection. +This fixes normal-pressure maintenance donors being skipped when the roster is +partial. Fresh partial observations can allocate settled evacuation through the +existing planner and shared two-attempt budget. Empty permits still leave the +operation Evacuating: complete role inventory, busy-work quiescence and joined +shutdown/withdrawal evidence are required before completion. + +A rerun exposed an immediate publication-retry race: the watch result could +arrive before its task exited, so the next dispatch joined the old unconfirmed +receipt. Result delivery now joins the owned task's short exit sequence. A +subsequent dispatch can retry the retained publication; dropped waiters still +leave the task owned. The original failing run was 29 of 30 host cases; the +final suite and ten exact-case repetitions below pass after this repair. + +| Source and evidence | Recorded value | +| --- | --- | +| Baseline | `b61768f95fe159f07db975f68950c0588ab563cc` | +| Source manifest | 568 Rust/Cargo files; SHA256 `66e9bf466b87fd4c3c9f0ab745a726c4a505d916337fa6ba1ada2ef4e95afe44` | +| Journal SQL | Unchanged SHA256 `6fd5c72f5e8f4bd245797dbc6f433baca0cece4d519aa954f7526b2ae61ad668` | +| `cargo test -p cellule-host --test node node::fleet --locked` | 30 passed; 38 filtered. Four new public maintenance cases cover closed sticky admission, existing-owner query readback, duplicates, lost reply/waiter, wrong boot, stale intent and unsupported effects. Shutdown joins and checks Stopped with zero actor/retained charges. | +| `cargo test -p cellule-host --example fleet_operations --locked` | 41 passed; none ignored or filtered. Six new driver cases use real SQLite with synthetic effects; both real-node overload/controller-restart cases also pass. | +| Exact immediate-retry case repeated ten times | Each passed; 67 filtered per run. No delay added to the retry. | +| `cargo clippy -p cellule-host --lib --tests --example fleet_operations --all-features --locked -- -D warnings` | Passed. | +| Static gates | Format, boundaries, module layout, 108 Rust snippets, 1140 Markdown links, 28 SQL/peer assertions and 562 protocol links passed. | + +Commands used Rust 1.97.0, the recorded checkout target and +`CARGO_INCREMENTAL=0`. Public maintenance tests use a local in-memory acceptance +journal with real runtime/SQLite owners. The separate driver tests use durable +SQLite and synthetic endpoint effects. These establish admission and driver +sequencing; they do not establish complete three-node maintenance, enrollment +producer/startup barriers, role evacuation, busy primitive quiescence or +process/provider qualification. The full plan remains incomplete. + +### Baseline CI qualification result + +On baseline `b61768f`, workspace and MSRV +[run 36888176019](https://github.com/crabbuild/cellule/actions/runs/36888176019), +contract, website and decoder fuzz passed. Follower capacity in +[run 36888175986](https://github.com/crabbuild/cellule/actions/runs/36888175986) +and smoke in +[run 36888176067](https://github.com/crabbuild/cellule/actions/runs/36888176067) +also passed; routing was still pending at inspection. + +The same capacity run's object-proof job **failed**: the skewed window at two requests per second per node +completed 59 of 60 planned arrivals. Arrival 49 was `scheduler_late` +(scheduled 8,166,666 us, started 8,464,809 us); the driver reported no fully +served skewed capacity point. Artifact 11175158203 retains the failed evidence +with ZIP SHA256 `34b6aad753434eacd208c1c03ec49952fdea202adb62e435790b3690b807605a`. +This is failed qualification, irrespective of other passing jobs or local +functional tests. Profiles, deadlines and expected evidence remain unchanged; +the updated PR must complete its own CI and the full fleet qualification work. + +## October 1 2026 atomic API lint repair + +Workspace CI `36885649739` failed the warnings-denied lint gate on deprecated +`AtomicU64::fetch_update` calls. All eleven workspace calls now use its renamed +`try_update` API, preserving orderings, update closures, return handling and +counter behavior. A minimal compiler probe and the selected builds below +confirm availability on the declared Rust 1.97 minimum. No lint allowance, +toolchain pin, profile threshold or qualification requirement changed. + +| Source and evidence | Recorded value | +| --- | --- | +| Baseline | Controller checkpoint `ecec878` | +| Source manifest | 565 Rust/Cargo files; SHA256 `5fff33c4ab3ab4da25f73c4280ee62049ee08f5d85d39a18da6d26bc2164939c` | +| `cargo test -p cellule-store --lib --features test-support read_admission --locked` | 18 passed; 147 filtered. | +| `cargo test -p cellule-ltx --lib --features replica environment::tests::prepared_disk_budget --locked` | 8 passed; 98 filtered. | +| `cargo test -p cellule-runtime --lib fleet::operations --locked` | 60 passed; 419 filtered. | +| `cargo test -p cellule-host --example fleet_operations --locked` | 35 passed; none ignored or filtered. Includes both real-node scenarios. | +| `cargo clippy -p cellule-store -p cellule-ltx -p cellule-runtime -p cellule-host --lib --tests --example fleet_operations --all-features --locked -- -D warnings` | Passed on Rust 1.97.0. | + +Commands used the recorded checkout target with `CARGO_INCREMENTAL=0`. +Format, boundary/layout, whitespace and 1140 Markdown links passed. This is +selected local evidence; latest stable workspace lint remains a CI gate. +On baseline PR head `734650d`, both object- and follower-capacity jobs in +run `36885649670` passed with the existing qualification profiles. The earlier +failed object-capacity run remains recorded below. Routing and smoke jobs +were still pending at the last inspection; they establish no passing evidence +for this checkpoint or the full goal. + +## October 1 2026 controller replacement checkpoint + +`fleet_operations controller-restart` now runs real three-node movement with +two source replies lost after durable release. Both unconfirmed attempts retain +their permits and exact inputs. A bounded real-time wait expires the original +controller lease; an independently reopened journal and a different claimant +acquire epoch 2. The old claimant receives the original journal Fenced error +without changing the successor's snapshot. Fresh source inspection adopts the +retained exact release proofs instead of repeating release. + +The driver now joins expired unused receiver credit before first activation +after a proven release. Committed cleanup preserves the release and both fleet +charges; canonical ordinary admission restores the same Cells afterward. +`receiver_resources_settled` exposes this existing committed fact without +changing codecs. Both other nodes verify original acknowledged outcomes and +SQL readback, old source handles fail, and shutdown joins all three nodes and +their resource owners. + +```text +released=2 activated=2 retired=2 receipt_checks=2 max_inflight=2 max_restore_bytes=2550136832 joined_nodes=3 receiver_nodes=2 lost_release_replies=2 controller_epoch=2 expired_receiver_cleanups=2 blocker_count=1 +blockers=[IncompleteObservation] +``` + +| Source and evidence | Recorded value | +| --- | --- | +| Baseline | `734650d35a355506a76775cab2f75015789d8792` | +| Source manifest | 565 Rust/Cargo files; SHA256 `a6795e35b4b07c6bf5113a12d526e51c530bd497b93c3a342fe0fb08670229a7` | +| `cargo test -p cellule-host --example fleet_operations --locked` | 35 passed; none ignored or filtered. | +| `cargo test -p cellule-runtime --lib fleet::operations --locked` | 60 passed; 419 filtered. | +| `cargo run -p cellule-host --example fleet_operations --locked -- controller-restart` | Exit zero with the output above. | +| `cargo clippy -p cellule-runtime -p cellule-host --lib --example fleet_operations --tests --locked -- -D warnings` | Passed. | +| Static gates | Format, boundaries, module layout, 108 Rust snippets, 1140 Markdown links, 28 SQL/peer assertions and 562 protocol links passed. | + +Commands used Rust 1.97.0 and the existing checkout target with +`CARGO_INCREMENTAL=0`. The fixed profile has a three-second controller lease +and 500-ms interval; node leases, prepared-credit expiry and all observations +use actual time. This establishes in-process controller replacement and local +durable reconstruction. It does not establish process crash, receiver-session +loss, complete fleet observations, busy maintenance or provider qualification. +All full-plan gaps remain required. + +Baseline CI `36885649739` passed through tests and website Rust compilation, +then failed workspace Clippy because newer stable Rust deprecated +`AtomicU64::fetch_update`. The contract, MSRV, website, decoder-fuzz and +object-capacity jobs passed. Remaining jobs were still running when inspected. +The renamed API compiled on Rust 1.97.0 in a separate minimal probe; workspace +call sites still need updating before that lint failure is resolved. + +## October 1 2026 atomic absence reconciliation checkpoint + +The public reconciler can now settle an unknown dispatch that never reached +acceptance. `ResolveUnaccepted` proves absence for the exact attempt, effect, +physical node and boot in the same transaction as the head CAS. The revision +advances even when the attempt state otherwise stays identical. This fences +delayed old envelopes before a fresh retry. If acceptance wins the race, +resolution fails and the original work remains available for adoption. + +Both permits and receiver charges remain retained. Expired preparation or +release moves toward independent cleanup without inventing source release or +serving evidence. Accepted source release without proof still cannot be +repeated. This closes the unaccepted-dispatch recovery gap; failed-source and +receiver-session adoption, complete observations, maintenance and the other +full-plan work remain outstanding. + +| Source and evidence | Recorded value | +| --- | --- | +| Baseline | `801bc4f5cf1cbc56385f09f7a1e9442247aa92da` | +| Source manifest | 565 Rust/Cargo files; SHA256 `9ea5990da8e182094387747d1bc80ca447e61a4fe38612a67ecd0f07fdac1fd3` | +| `cargo test -p cellule-runtime --lib fleet::operations --locked` | 60 passed; 419 filtered. | +| `cargo test -p cellule-host --example fleet_operations --locked` | 34 passed; none ignored or filtered. Includes independent acceptance/absence races, lost CAS replies, delayed envelope fencing, reconstructed driver retry and the existing real three-node scenario. | +| `cargo clippy -p cellule-runtime -p cellule-host --lib --example fleet_operations --locked -- -D warnings` | Passed. | +| Static gates | Format, boundaries, module layout, 85 Rust snippets, 1136 Markdown links, 28 SQL/peer assertions and 562 protocol links passed. | + +Commands used the existing checkout target with `CARGO_INCREMENTAL=0` and +Rust 1.97.0. Initial new-test compilation failures and the failure that exposed +the unchanged-state revision bug were fixed before these passing runs. +No persisted format or journal SQL schema changed. These are focused local +results; no additional process/provider qualification is established. + +CI for the baseline passed its contract, MSRV, website and decoder-fuzz jobs. +The workspace job failed in the website Rust example compiler because it passed +only app/runtime externs while newer main documentation also imported store, +types, bytes and object_store. That checker failure needs repair; it does not +establish a passing workspace qualification result. + +### Main sync and example compiler repair + +Main revision `c51dd12` merged without conflicts in `cddb5c9`. The updated +565-file Rust/Cargo manifest is SHA256 +`b7028a45fefa23a2f780c74ef49c301c692ff1e6111d5bacb2d286825626952d`. +After the merge, the runtime operation tests again passed 60 cases (419 +filtered), and the example passed all 34 cases. Clippy for runtime/host +libraries, example and tests passed with warnings denied. Format, boundaries, +module layout, 108 Rust snippets, 1139 Markdown links and the unchanged SQL/peer +validators passed. + +The website compiler now includes all six documented libraries from Cargo's +reported artifacts. An explicit host target separates target libraries from +host build dependencies; both artifact directories remain available for +procedural macros. Earlier intermediate checker runs exposed a wrong `bytes` +crate identity and a missing procedural-macro search path. The final command +`python3 scripts/check-web-rust-examples.py` exited zero and compiled all 12 +authored guides. Checker SHA256: +`6d3d912944910fd784b1672cb8fcce38e92dd9ae42c1796afc64d1d667c34564`. +It used the same checkout target and toolchain. Updated full CI remains +required; the baseline workspace failure stays a recorded failed run. + +## October 1 2026 real three-node overload checkpoint + +The reference executable now supports `overload`, using the same scenario +implementation as its focused test. Three independently leased CellNodes share +strict in-memory object storage and initialize twelve real SQLite Cells. A held +seven-GiB disk admission reservation drives the actor's actual ledger samples +and normal hysteresis dwell to Shedding. The exported reconciler allocates two +charged moves, prepares real receivers, and resumes through an independently +reopened SQLite controller client. Both other nodes acquire successor authority, +resolve the original acknowledged command outcomes, read the restored values, +and reject the old source handles. Retirement follows fresh current inspections. + +The command checks all three runtimes reach Stopped with empty resource ledgers, +then closes both journal clients after joining their accepted jobs. Setup and +scenario errors still join every registered runtime before dropping private +paths. The fixture preserves original scenario errors and reports additional +cleanup errors. It uses the canonical local-owner lookup for restored actors; +the fully resident fast path can legitimately be unavailable during hydration. + +Observed executable output: + +```text +released=2 activated=2 retired=2 receipt_checks=2 max_inflight=2 max_restore_bytes=2550136832 joined_nodes=3 receiver_nodes=2 blocker_count=1 +blockers=[IncompleteObservation] +``` + +This is a finite **admission-pressure batch**, not physical disk utilization, +sustained-overload convergence, or provider/process qualification. Its fixed +ownership-only collector pins the three boot endpoints and signs actual local +classifier/capacity values. It reports incomplete role coverage and disables +count balancing. The actor retains its existing local idle eviction behavior; +reported fleet movement counts cover only the destination-reserved batch. +Pressure is released and new scheduling stopped after that batch is allocated. + +| Source and environment | Recorded value | +| --- | --- | +| Baseline revision | PR foundation `9138a51d3b1b9691c16d21bf81784677ca55445c` | +| Source manifest | 564 Rust/Cargo files; SHA256 `0add021e98e10002f8aacf1dad914d097233ead7bd75f175dfe8e725633c3bed` | +| Journal SQL | Unchanged SHA256 `6fd5c72f5e8f4bd245797dbc6f433baca0cece4d519aa954f7526b2ae61ad668` | +| Toolchain/target | Rust 1.97.0; existing checkout target, `CARGO_INCREMENTAL=0`. Package artifacts were cleaned with Cargo after a failed compile exhausted the mounted volume. That failed compile establishes no test result. | + +| Command | Observed result | +| --- | --- | +| `cargo test -p cellule-host --example fleet_operations --locked` | 31 passed; no ignored/filtered cases. Includes the shared real-node scenario. | +| `cargo run -p cellule-host --example fleet_operations --locked -- overload` | Exit zero with the output above and all three nodes joined. | +| `cargo test -p cellule-app --test integration host::three_node_host_recovers_published_state_after_owner_loss --locked -- --exact` | 1 passed; 43 filtered. | +| `cargo clippy -p cellule-host --example fleet_operations --tests --locked -- -D warnings` | Passed. | +| `cargo clippy -p cellule-app --test integration --locked -- -D warnings` | Passed. | + +The owner-loss fixture now recognizes only the original terminal Fenced error +inside the retained `cell-runtime-drain` source chain, and only when the node +lease was actually fenced. It also checks zero Cell, memory, job, descriptor and +disk charges. Unrelated facility errors still fail. This fixes the assertion +regression found in the foundation PR's workspace and contract CI runs. + +The foundation commit's third object-capacity repeat remains a failed +qualification result: run `36876658271`, artifact +`cell-write-capacity-36876658271-1`, window `capacity-3-hot-2`, arrival 25 was +`scheduler_late` (scheduled 4166666 us; started 4387343 us). Two earlier repeats +passed. No threshold, profile or evidence requirement was weakened; that +campaign must pass on the updated PR before qualification is claimed. + +Before this entry, module/layer gates, 85 Rust snippets, 1136 Markdown links, +28 schema/protocol assertions and 562 protocol links passed. Format and diff +whitespace checks are repeated after the checkpoint. Full W1–W10 completion is +unproven: complete production observers/producers, failed-source/receiver +adoption, busy maintenance, role evacuation/finalization, remaining scenarios, +qualification and operator rollout remain in the full plan. + +## October 1 2026 public reconciler checkpoint + +The host now exports a caller-driven `FleetReconciler` that claims the journal +controller, advances charged attempts before planning, and publishes each phase +by exact head/registry CAS before dispatch. It uses the existing placement +planner and reducer, projects all unresolved receives, and reads committed +incarnation-specific cooldown and post-batch barriers from the journal. +Scheduling stop prevents new permits while existing work continues to settle. + +Fresh inspection binds the entire action, current registry, exact node/boot, +nonce and original capture interval. The node checks current actor and authority +state without starting recovery or consulting a historical Inspect receipt as +current proof. Unknown outcomes retain both fleet permits. Confirmed activation +and retirement require fresh serving evidence; unused receiver resources settle +independently. Finite-action shutdown joins every sibling job before returning +the retained original failure. + +Each charged attempt receives a bounded share of the pass deadline, leaving time +for healthy siblings and planning. The report retains original endpoint errors; +a timeout requires rereading journal state and the controller epoch before +continuing. Journal failures stop the pass. Dispatch counts include timed-out +waiters; confirmed release, activation, recovery and cancellation counts follow +committed transitions. Forward progress requests an immediate next pass; +unresolved failures and cancellations retain periodic retry to avoid hot loops. +The supplied application clock is read directly at every boundary and regression +fails closed; deadlines remain monotonic. + +| Source and environment | Recorded value | +| --- | --- | +| Baseline revision | `e07670e2348231ed401cc7280a47e3ab97596ffe` | +| Working tree | Implementation before PR commit; 560 Rust source, Cargo manifest and lock files. | +| Source manifest SHA256 | `83289340c715081f7c56113255a9f52a94597f83020c23476797823750adf31d` | +| Example `journal/schema.sql` SHA256 | `6fd5c72f5e8f4bd245797dbc6f433baca0cece4d519aa954f7526b2ae61ad668` | +| Toolchain | `rustc 1.97.0 (2d8144b78 2026-07-07)` | +| Target directory | `$HOME/Workspace/crabbuild-target/cellule-f9383af7-fleet-operations` | +| Proof level | Pure reducer, local SQLite transactions, signed synthetic driver observations/effects, and selected real host action/lifecycle tests. No real three-node driver scenario or process/provider qualification. | + +| Command | Observed result | +| --- | --- | +| `cargo test -p cellule-host --example fleet_operations --locked` | 30 passed, none filtered or ignored: 18 journal and 12 driver/model tests. Includes lost replies, healthy-sibling progress after timeout, retained permits, clock regression, competing controllers, stop-new-moves and pressure relief. | +| `cargo test -p cellule-host --test node fleet --locked` | 26 passed, 38 filtered. Includes fresh actor inspection and failure drain joining a sibling inspection. | +| `cargo test -p cellule-host --test node lifecycle --locked` | 11 passed, 53 filtered. | +| `cargo test -p cellule-runtime --lib fleet::operations --locked` | 57 passed, 419 filtered. | +| `cargo clippy -p cellule-host --lib --example fleet_operations --tests --locked -- -D warnings` | Passed. | +| `RUSTDOCFLAGS='-D warnings' cargo doc -p cellule-host -p cellule-runtime --all-features --no-deps --locked` | Passed. | + +Format, layer and module validators, 1136 local Markdown links/anchors, 85 Rust +snippets, 28 schema/protocol assertions and 562 protocol-document links passed +before this checkpoint. `git diff --check` passed. The documentation link gate +is rerun after this entry. Broad suites remain CI work; no provider/process +campaign result is claimed. + +### Remaining work at the public reconciler checkpoint + +Connect complete authenticated node/role observations, enrollment producers and +the startup intent barrier. Implement failed-source/receiver adoption and the +remaining unknown/absence reconciliation rules. The example executable still +supports only `inspect-journal`; its driver tests use simulated effects, not +three independently leased runtimes. Implement and qualify real overload, +maintenance, controller-restart and receiver-loss scenarios, busy-work +quiescence, role evacuation/finalization, and operator rollout. W5–W10 are not +complete, and earlier work packages retain the gaps listed in the full plan. + +## October 1 2026 durable local journal checkpoint + +The [embedding example](../crates/cellule-host/minion/README.md) +now implements `FleetJournal`, `FleetEnrollmentJournal`, and `FleetActionJournal` +against one SQLite file. Each call uses `BEGIN IMMEDIATE`; current head/registry, +intent checks and record publication share the transaction. WAL with FULL +synchronization retains committed state for independently reopened clients. +The library adds no SQLite provider dependency; the dependency belongs to the +embedding example and its dev dependencies. + +Maintenance stores the immutable original request separately from mutable +deadline/session progress. Older operations and cordons survive later work; +return-to-service cannot rewrite the current maintenance head. Action records +include their executing physical node and boot, allowing distinct source and +receiver inspections. Retiring permits commits their exact progress page +atomically. Loading rejects malformed head bytes and missing referenced +operations. Acquisition and recovery results require matching retained input +and evidence rather than reconstructed success from an arbitrary later root. + +Each client admits at most 32 owned blocking jobs. A dropped caller cannot +cancel an accepted transaction. Close stops admission, joins every accepted +job, settles slots and closes the connection. Tests inject both rollback before +commit and lost replies after commit, retaining original errors and proof times. + +| Source and environment | Recorded value | +| --- | --- | +| Baseline revision | `e07670e2348231ed401cc7280a47e3ab97596ffe` | +| Working tree | Uncommitted implementation; 551 Rust source, Cargo manifest, and lock files. | +| Source manifest SHA256 | `e2220e1dbcede1c01a364a983dedb21e73dd38a7d71cb02551cbf994a7a0c43a` | +| Example `journal/schema.sql` SHA256 | `6fd5c72f5e8f4bd245797dbc6f433baca0cece4d519aa954f7526b2ae61ad668` | +| Toolchain | `rustc 1.97.0 (2d8144b78 2026-07-07)` | +| Target directory | `$HOME/Workspace/crabbuild-target/cellule-f9383af7-fleet-operations` | +| Proof level | Local SQLite transactions, independent clients, injected commit replies, adapter reconstruction and bounded close. No process-crash, filesystem-fault, distributed provider or three-node movement qualification. | + +The source manifest uses the reproduction script below. SQL is compiled into +the example via `include_str!`; its separate fingerprint is required because +that script covers Rust and Cargo files only. + +| Command | Observed result | +| --- | --- | +| `cargo test -p cellule-host --example fleet_operations --locked` | 17 passed; none ignored or filtered. Includes racing claims/allocations/acceptance/enrollment, retained permits and cordons, original acquisition/recovery records, lost enrollment replies, atomic history, malformed/missing references, rollback and bounded cancellation-safe close. | +| `cargo clippy -p cellule-host --example fleet_operations --tests --locked -- -D warnings` | Passed for the example and selected test compilation. | +| `cargo run -p cellule-host --example fleet_operations --locked -- inspect-journal ` | Two consecutive executable runs exited zero. Both printed revision 0, registry revision 0, bootstrap false, scheduling false, zero unresolved attempts/restore bytes. Raw retained head/registry bytes matched after reopening. | + +Format, crate-boundary and module-layout validators, 1134 local Markdown +links/anchors, 85 documented Rust snippets, 28 SQL/peer schema assertions and +562 protocol-document links passed before adding this checkpoint. `git diff +--check` passed. Documentation links are checked again after this entry. +Broad suites and provider/process gates were not run for this local checkpoint. + +### Remaining work at the durable local journal checkpoint + +Connect every enrollment producer and startup intent barrier. Implement fresh +observation envelopes, the complete observer and W5 reconciler, and consume this +journal through the public host action path with real independently leased +nodes. Retained inspection results are historical; the existing stable Inspect +key does not establish current serving and must not be counted as fresh evidence. +Complete cross-session recovery and W4 refusal/unknown reconciliation. + +The example currently supports journal inspection only. W6–W10 still require +busy maintenance, role evacuation and finalization, three-node scenarios, +process/fault/provider and mixed-binary qualification, and operator rollout +runbooks. W1 and W8 are not complete; the full implementation goal remains active. + +## October 1 2026 registry contract checkpoint + +W1 now defines bounded retained intent/enrollment records and their codecs. +An enrollment request binds its exact physical nodes, boot sessions, checked +intent revisions, role and original time. Reader roles name the Cell and +published position; follower roles name the leader boot and node-log epoch. +Boot enrollment carries retained mode and cannot open readiness under a cordon. +Unknown enrollment remains Pending after expiry. Only checked completion, +definite refusal, or canonical closure/retirement advances it. Tombstones retain +immutable input and evidence; duplicate results do not refresh capture time. + +`RegistryVersion` carries the shared mutation revision, controlled bootstrap +marker and revision-checked scheduling policy. New registries start stopped. +Enabling requires bootstrap; stopping preserves charged attempts and cordons. +Its allocation gate checks current retained source/receiver rows. A draining +source is eligible only for the current evacuation operation, and a cordoned +receiver cannot receive a new planned Cell. Registry pages include older +cordons and failed-session obligations with at most 128 sorted rows per page. + +Maintenance adoption of a different boot now advances the intent revision. +The previous behavior changed the session while leaving that revision unchanged. +The retained intent update rejects older revisions and prevents an Active boot +default from overwriting a cordon. + +The host now exposes `FleetJournal`, `FleetJournalSnapshot`, and +`FleetEnrollmentJournal` transaction contracts alongside `FleetActionJournal`. +Implementations must share one durable transaction domain for controller CAS, +permits, scheduling, retained intents/operations, enrollment and action evidence. +These are contracts and pure gates; the reference backend and complete observer +are still missing. No new trait implementation is claimed as durable evidence. + +| Source and environment | Recorded value | +| --- | --- | +| Baseline revision | `e07670e2348231ed401cc7280a47e3ab97596ffe` | +| Working tree | Uncommitted implementation; 544 Rust source, Cargo manifest, and lock files. | +| Source manifest SHA256 | `0236712e5d971a8455b09e333ad8ae86191f111860a07ac31e444b49c20b3eb8` | +| Toolchain | `rustc 1.97.0 (2d8144b78 2026-07-07)` | +| Target directory | `$HOME/Workspace/crabbuild-target/cellule-f9383af7-fleet-operations` | +| Proof level | Pure deterministic registry transitions/codecs and existing public local host action regressions. No backend reconstruction, process restart, complete roster, or provider qualification. | + +| Command | Observed result | +| --- | --- | +| `cargo test -p cellule-runtime --lib fleet::operations:: --locked` | 53 passed; 419 filtered out. Ten new registry cases cover intent/reboot revisions, exact enrollment inputs, ambiguous obligations, immutable evidence/time, bootstrap, cursor/page bounds, malformed codecs, stop/resume, and allocation gates. | +| `cargo test -p cellule-host --test node node::fleet --locked` | 22 passed; 38 filtered out. Existing source, receiver and recovery actions still pass through the public host path. These tests retain the earlier in-memory action journal, not the new complete journal contracts. | +| `cargo clippy -p cellule-runtime --lib --test runtime --locked -- -D warnings` | Passed for the selected runtime library and public runtime target. | +| `cargo clippy -p cellule-host --lib --test node --locked -- -D warnings` | Passed for the selected host library and public node target. | +| `RUSTDOCFLAGS='-D warnings' cargo doc -p cellule-runtime -p cellule-host --no-deps --locked` | Both API documentation targets compiled without warnings. | + +Boundary and module-layout validators, document links and Rust fences, +SQL/peer contracts, format checking and `git diff --check` passed. New registry +record kinds 12 through 16 require the plan's reader deployment gate. The +bootstrap marker is an adapter assertion, not proof that a filtered observation +is complete. + +### Remaining work at the registry contract checkpoint + +Implement a strict durable reference journal against all three host contracts, +with racing transactions, lost commit replies and backend reconstruction. +Connect every enrollment producer and startup intent check, then complete +fresh observation envelopes and the W5 reconciler. Cross-session recovery, +remaining W4 inspection/refusal gates, and W6–W10 remain in scope. The full +implementation goal remains active. + +## October 1 2026 receiver and failed-source recovery checkpoint + +The host executor now prepares and activates admitted receivers, records exact +acquisition input, joins unused preparation credit, and inspects current +serving through the canonical actor and authority paths. It owns finite action +work across dropped transport waiters. Retained results keep their original +position when a successor publishes a newer root. + +Failed-source recovery has a separate `Recover` action and `Recovered` outcome. +The runtime's ordinary Idle acquisition and takeover paths confirm a durable +`RecoveryBasis` before ownership CAS and a `RecoveryEvidence` position before +actor admission. A lost input-record reply prevents CAS. A lost recovered- +position reply prevents admission and follows canonical rollback. Inspection +can later combine the retained position with fresh serving evidence; it cannot +fabricate a clean source release. Receiver cleanup remains an independent fact +required before the attempt's charged permit can retire. + +Public fault tests exercise both a failed Serving owner and an unobserved source +release CAS to Idle. The retained follower-tail integration fixture exercises +the same recording boundaries around actual tail materialization. Its signed +advertisement timestamps now precede the simulated operations rather than +being future-dated. Fatal finite-action task failures retain their original +error across repeated shutdown while resource cleanup continues. + +| Source and environment | Recorded value | +| --- | --- | +| Baseline revision | `e07670e2348231ed401cc7280a47e3ab97596ffe` | +| Working tree | Uncommitted implementation; 538 Rust source, Cargo manifest, and lock files. | +| Source manifest SHA256 | `77ae052ba23864a1ea795f1e0f205fecb857cb1e359ef60f4725c21a96eb338b` | +| Toolchain | `rustc 1.97.0 (2d8144b78 2026-07-07)` | +| Target directory | `$HOME/Workspace/crabbuild-target/cellule-f9383af7-fleet-operations` | +| Proof level | Focused local contracts and public leased runtime/host behavior with an atomic in-memory reference action journal. No restart, process, mixed-binary, or provider qualification. | + +All selected commands below exited zero. Together the selected test suites +reported 98 passed and one environment-dependent ignored test. The ignored +RustFS case requires its documented provider environment and supplies no +passing provider evidence. + +| Command | Observed result | +| --- | --- | +| `cargo test -p cellule-host --test node node::fleet --locked` | 22 passed; 38 filtered out. Includes source actions, admitted receivers, recovery, lost journal replies/waiters, and retained task errors. | +| `cargo test -p cellule-runtime --lib fleet::operations:: --locked` | 43 passed; 419 filtered out. Includes five pure recovery contract/codec cases. | +| `cargo test -p cellule-runtime --test runtime runtime::lifecycle::ownership::recovery:: --locked` | 7 passed; 1 ignored; 172 filtered out. Includes actual retained-tail materialization and both recording failure boundaries. | +| `cargo test -p cellule-runtime --test runtime runtime::lifecycle::receiver:: --locked` | 9 passed; 171 filtered out. | +| `cargo test -p cellule-runtime --test runtime runtime::lifecycle::idle::release:: --locked` | 6 passed; 174 filtered out. | +| `cargo test -p cellule-host --test node node::lifecycle:: --locked` | 11 passed; 49 filtered out. | +| `cargo clippy -p cellule-runtime --lib --test runtime --locked -- -D warnings` | Passed for the selected runtime library and public runtime target. | +| `cargo clippy -p cellule-host --lib --test node --locked -- -D warnings` | Passed for the selected host library and public node target. | +| `RUSTDOCFLAGS='-D warnings' cargo doc -p cellule-runtime -p cellule-host --no-deps --locked` | Both API documentation targets compiled without warnings. | + +Boundary and module-layout validators, document links and Rust fences, +SQL/peer contracts, format checking, and `git diff --check` passed. Recovery adds +strict new record kinds, phases, action, and outcome discriminators; the plan's +reader deployment gate remains required. These local checks do not qualify +mixed deployed readers. + +### Remaining work at the recovery checkpoint + +Recovery execution currently targets the original preferred receiver boot. +Complete cross-session recovery and refusal/unknown reconciliation, fresh +observation envelopes, retained intent/enrollment registries, and durable +adapter semantics. Connect the complete observer and reusable W5 reconciler. +W6–W10 still require busy maintenance, role evacuation/finalization, the runnable +reference fleet, fault and process/provider qualification, and operator +runbooks. The full implementation goal remains active. + +## October 1 2026 accepted source action checkpoint + +The runtime now exposes a bounded `AcceptedFleetAction` record and codec. +First acceptance checks the current journal head and exact executing endpoint. +Replay compares all immutable movement inputs, including physical identities, +authority epoch, conservative cost, snapshot, and deadline. The stable action +key remains an index. An accepted record supplies no remote authentication or +Cell authority. + +The host's [`fleet` module](../crates/cellule-host/src/fleet/mod.rs) defines the +application-owned atomic journal contract and a startup-bound finite-action +executor. Its current effect is settled source Release. It journals acceptance, +uses canonical `release_idle_cell_at`, and retains checked evidence through a +dropped waiter or failed publication. A retry republishes the same result. +An existing acceptance without a result becomes Unknown and never authorizes +releasing a newer actor generation. + +Public fault tests exposed a deadline-resumption problem in host drain. The +host now retains one runtime shutdown task and its original result. A timed-out +waiter leaves that task owned; a later drain joins it. Lease withdrawal waits +for successful runtime drain. These are local lifecycle foundations, not the +complete W7 role-evacuation and fleet-finalization protocol. + +| Source and environment | Recorded value | +| --- | --- | +| Baseline revision | `e07670e2348231ed401cc7280a47e3ab97596ffe` | +| Working tree | Uncommitted implementation; 522 Rust source, Cargo manifest, and lock files. | +| Source manifest SHA256 | `d0171432e9321c7e514a9f793d02388048aab6d74690e78fd7b40ad6e63d1f14` | +| Toolchain | `rustc 1.97.0 (2d8144b78 2026-07-07)` | +| Target directory | `$HOME/Workspace/crabbuild-target/cellule-f9383af7-fleet-operations` | +| Proof level | Focused local contracts and public leased CellNode behavior with an atomic in-memory reference journal; no provider or process qualification. | + +Use the source-fingerprint script below to reproduce this manifest. Older +checkpoint hashes identify their earlier source, not the current diff. + +| Command | Observed result | +| --- | --- | +| `cargo test -p cellule-runtime --lib fleet::operations:: --locked` | 30 passed; 419 filtered out. Includes five accepted-action tests and malformed acceptance deadline checks. | +| `cargo test -p cellule-host --test node node::fleet_actions:: --locked` | 7 passed; 38 filtered out. Concurrent duplicates, lost waiter, lost journal reply after successor publication, stale authorization, unknown accepted effect, resumed publication drain, and accepted-query drain with retained lease maintenance. | +| `cargo test -p cellule-host --test node node::lifecycle:: --locked` | 11 passed; 34 filtered out. Existing node drain, scale-down, deadline, lease-withdrawal, and pressure/cordon behavior. | +| `cargo test -p cellule-runtime --test runtime runtime::lifecycle::receiver:: --locked` | 8 passed; 169 filtered out. Prepared resource ownership, refusal, cancellation, exact restoration, lost replies, and shutdown. | +| `cargo test -p cellule-runtime --test runtime runtime::lifecycle::idle::release:: --locked` | 6 passed; 171 filtered out. Exact identity/final-root release and accepted-work barriers. | +| `cargo test -p cellule-runtime --lib cell::actor::tests:: --locked` | 4 passed; 445 filtered out. Includes preservation of unobserved original release errors. | +| `cargo clippy -p cellule-host --lib --test node --locked -- -D warnings` | Passed for the selected host library and public node target. | +| `cargo clippy -p cellule-runtime --lib --test runtime --locked -- -D warnings` | Passed for the selected runtime library and public runtime target. | +| `RUSTDOCFLAGS='-D warnings' cargo doc -p cellule-runtime -p cellule-host --no-deps --locked` | Both API documentation targets compiled without warnings. | + +The test journal linearizes head checking and acceptance under one lock and +retains accepted/results independently of transport waiters. It does not prove +a production backend's conditional writes, process restart durability, a +complete enrollment registry, or a controller's reconciliation behavior. + +### Remaining work at the source action checkpoint + +Integrate prepared receiver and maintenance actions into this host executor. +Persist immutable checked acquisition and recovery basis, complete the distinct +recovered outcome, and implement source inspection/refusal reconciliation. +Then connect the complete observer, durable registries, and reusable W5 driver. +W6–W10, the runnable reference fleet, broad verification, and measured +process/provider qualification remain required before the full objective is +complete. The goal remains active. + +Boundary and module-layout validators, document links and Rust fences, +SQL/peer contracts, format checking, and `git diff --check` passed for this +checkpoint. They supplement the focused evidence; broad workspace and +process/provider gates remain pending. + +## September 30 2026 disk credit checkpoint + +W4 now has an admitted disk-credit boundary. `DiskReservation::into_budget` +transfers an existing reservation into the canonical LTX host's operation +budget. Preparing work retains the full parent envelope; child reservations +use that credit. `finish_preparation` returns unused bytes and retains actual +charges. Later growth competes with ordinary parent admission. + +The public activation test checks the restored payload against the original +bytes, performs another SQLite transaction and capture, and closes the +database. The cancellation test pauses an accepted filesystem job, aborts its +initiating future, and verifies that parent credit stays charged until the job +closes. A runtime test verifies the same credit against `LedgerDiskAdmission`. + +| Source and environment | Recorded value | +| --- | --- | +| Baseline revision | `e07670e2348231ed401cc7280a47e3ab97596ffe` | +| Working tree | Uncommitted implementation; 511 Rust source, Cargo manifest, and lock files in the source fingerprint below. | +| Source manifest SHA256 | `11801d0adf23955539d3be78b341279e6c571286d0e3daf1574caaf2dce1b8f0` | +| Toolchain | `rustc 1.97.0 (2d8144b78 2026-07-07)`; `cargo 1.97.0 (c980f4866 2026-06-30)` | +| Target directory | `$HOME/Workspace/crabbuild-target/cellule-f9383af7-fleet-operations` | +| Proof level | Local focused unit and public library behavior; no provider or process qualification. | + +All commands below exited zero. Cargo commands used the recorded target +directory. Their scope is the selected modules and targets. + +| Command | Observed result | +| --- | --- | +| `cargo test -p cellule-ltx --no-default-features --lib prepared_disk_budget --locked` | 8 passed; 35 filtered out. | +| `cargo test -p cellule-ltx --features replica --lib prepared_disk_budget --locked` | 8 passed; 97 filtered out. | +| `cargo test -p cellule-ltx --features replica --test host host::hooks::activation:: --locked` | 13 passed; 2 ignored; 55 filtered out. Includes both new public disk-credit scenarios. | +| `cargo test -p cellule-runtime --lib fleet::resource::tests --locked` | 10 passed; 434 filtered out. Includes real runtime-ledger credit inheritance. | +| `cargo clippy -p cellule-ltx --features replica --lib --test host --locked -- -D warnings` | Passed for the selected library and public host target. | +| `RUSTDOCFLAGS='-D warnings' cargo doc -p cellule-ltx --features replica --no-deps --locked` | API documentation compiled without warnings. | + +The ignored activation cases are the directory-cache restart diagnostic and +the RustFS sparse-activation diagnostic. Neither supplies evidence for this +checkpoint. The latter requires the environment in the +[LTX example guide](../crates/cellule-ltx/examples/README.md). + +Boundary/layout validators, document link and Rust-fence validators, the +SQL/peer validator, format checking, and `git diff --check` also passed during +this checkpoint. These checks do not replace the plan's broad gates. + +Reproduce the source fingerprint from the workspace root: + +```python +import hashlib +import pathlib +import subprocess + +root = pathlib.Path.cwd() +tracked = subprocess.check_output(["git", "ls-files", "-z"]).split(b"\0") +untracked = subprocess.check_output( + ["git", "ls-files", "--others", "--exclude-standard", "-z"] +).split(b"\0") +paths = sorted({ + p.decode() for p in tracked + untracked + if p and (p.endswith(b".rs") or p.endswith(b"Cargo.toml") or p == b"Cargo.lock") +}) +manifest = "\n".join( + f"{hashlib.sha256((root / p).read_bytes()).hexdigest()} {p}" for p in paths +) + "\n" +print(len(paths), hashlib.sha256(manifest.encode()).hexdigest()) +``` + +### Remaining work at the disk credit checkpoint + +Integrate disk credit into the host's exact-session receiver preparation and +canonical activation. Transfer memory, descriptor, Cell-slot, and job tokens +through the same admission path. Add durable action acceptance and checked +release/acquisition/recovery evidence, then exercise duplicate dispatch, +controller restart, and shutdown through public CellNode actions. W5–W10 and +the full acceptance matrix remain pending. + +## October 1 2026 prepared receiver checkpoint + +W4 now has a trusted local receiver path in +[`cell/actor/receiver.rs`](../crates/cellule-runtime/src/cell/actor/receiver.rs). +`CellRuntime::prepare_receiver` holds a real Cell slot, conservative native +memory and descriptors, the incoming Cell's affine SQL worker permit and +ledger charge, and scoped LTX disk credit. Partial refusal returns its tokens. +Canonical Idle acquisition consumes those tokens without releasing and +reacquiring the envelope. + +The runtime retains at most two local receipts, charged to the ordinary +retained-byte ledger. Opaque caller references are weak: a lost reply can be +looked up by attempt, and keeping a caller reference cannot prevent shutdown. +Accepted activation belongs to a runtime-owned task. Dropping its RPC waiter +cannot cancel takeover. Shutdown cancels unused preparation, joins accepted +acquisition, then follows the existing actor and worker close path. + +The public receiver suite checks exact root and SQL value readback, resolution +of the original acknowledged mutation, duplicate preparation, admission +refusal before release, explicit expiry/cancellation, and resource cleanup. +It also checks a failed restore after ownership CAS, a lost activation waiter +plus a lost CAS response, shutdown during an accepted claim, and restoration +of the current root after another canonical owner commits and releases. + +Local lifecycle hints are not durable results. `Failed` does not prove that +ownership is absent; `Activated` still requires fresh authority and actor +readiness. The host executor must supply authorization, journal acceptance and +results, and checked release/acquisition/recovery evidence. This checkpoint +establishes neither a deployed fleet controller nor process/provider behavior. + +| Source and environment | Recorded value | +| --- | --- | +| Baseline revision | `e07670e2348231ed401cc7280a47e3ab97596ffe` | +| Working tree | Uncommitted implementation; 513 Rust source, Cargo manifest, and lock files. | +| Source manifest SHA256 | `1f7577f4ba6604898295d9680e4f6d18974345c8317660de49e9d400791a2a32` | +| Toolchain | `rustc 1.97.0 (2d8144b78 2026-07-07)`; `cargo 1.97.0 (c980f4866 2026-06-30)` | +| Target directory | `$HOME/Workspace/crabbuild-target/cellule-f9383af7-fleet-operations` | +| Proof level | Focused local runtime library behavior against in-memory CAS, with real SQLite and resource ledgers. No process, provider, mixed-binary, or measured fleet qualification. | + +The fingerprint uses the reproduction script above. The disk credit +checkpoint remains evidence for its earlier source; it does not certify this +later receiver integration. + +All commands below exited zero. Cargo commands used the recorded target +directory and default features; these results cover only the selected targets. + +| Command | Observed result | +| --- | --- | +| `cargo test -p cellule-runtime --test runtime runtime::lifecycle::receiver:: --locked` | 8 passed; 168 filtered out. Includes original mutation resolution after movement and current-root readback after an intervening owner. | +| `cargo test -p cellule-runtime --test runtime runtime::lifecycle::idle::acquire:: --locked` | 2 passed; 174 filtered out. Existing ordinary receiver failure and cordon behavior. | +| `cargo test -p cellule-runtime --lib cell::worker::tests --locked` | 6 passed; 2 ignored; 436 filtered out. Existing shared-ledger, inventory, sparse SQL, and hydration scenarios. | +| `cargo clippy -p cellule-runtime --lib --test runtime --locked -- -D warnings` | Passed for the selected library and public runtime target. | +| `RUSTDOCFLAGS='-D warnings' cargo doc -p cellule-runtime --no-deps --locked` | API documentation compiled without warnings. | +| `cargo check -p cellule-host --lib --test node --locked` | Selected host library and node target compiled with the runtime API changes. This is not execution of host actions. | + +The two ignored worker tests are RustFS interference diagnostics. Their +isolated provider environment was not supplied, and they provide no evidence +for this checkpoint. The workspace's broad feature, process, model, and fault +campaign gates remain outstanding. + +Boundary/layout validators, 85 documented Rust snippets, 1119 local Markdown +links, 28 SQL/peer assertions with 561 validator links, format checking, and +`git diff --check` also passed. These static checks do not establish fleet +relocation or maintenance completion. + +### Remaining W4 work + +Connect the runtime path to authenticated CellNode actions and durable action +acceptance/results. Bind physical node and fleet scope, capture immutable +checked source release and acquisition/recovery evidence, and adopt unknown +outcomes across controller replacement. Complete host shutdown/role barriers +and concurrent ordinary activation/reader admission scenarios. W5–W10 and the +plan's full acceptance matrix remain pending. diff --git a/docs/roadmap.md b/docs/roadmap.md index 212d5906..0f2f1c09 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -40,5 +40,34 @@ concrete application need. Any new capability must integrate with the current registry, owner admission, maintenance budget, and host lifecycle; former APIs from earlier codebases are not part of today's supported surface. +## Fleet operations implementation plan + +The [fleet operations design and implementation plan](fleet-operations-plan.md) +specifies the proposed reconciliation path for overload relief, ownership +balancing, and resumable node maintenance. It includes signed pressure state, +busy-Cell drain, follower obligations, controller recovery, compatibility, +work packages, and acceptance checks. The plan also specifies reservation +ownership, an enrollment barrier for safe finalization, the source map, and the +first executable implementation slice. + +Start with the plan's [implementation brief](fleet-operations-plan.md#implementation-brief) +and [application startup recipe](fleet-operations-plan.md#application-startup-and-adapter-handoff) +for the delivery order and required integration adapters. + +The working tree now contains operation records, schema 3 readers, +shared local admission reasons, host movement actions, a local durable journal, +and a caller-driven reconciler with SQLite-backed sequencing tests. The overload +executable moves two real Cells across three leased nodes, reopens its controller +client and checks original outcomes and restored state. Complete +enrollment/observation wiring, leased-node failure integration, the maintenance +workflow, and measured fleet +qualification remain in progress; these foundations do not establish fleet +operation support by themselves. + +Follow the plan's [execution checkpoints](fleet-operations-plan.md#execution-checkpoints) +for the three delivery milestones: settled Cell movement, complete node +maintenance, then deployment qualification. Each milestone specifies the +public behavior, verification commands, and evidence required for completion. + For proof levels and receipt requirements, read the [delivery evidence guide](../crates/cellule-runtime/docs/delivery.md). diff --git a/scripts/check-module-layout.py b/scripts/check-module-layout.py index cd60c53d..b641433a 100644 --- a/scripts/check-module-layout.py +++ b/scripts/check-module-layout.py @@ -181,11 +181,15 @@ def check_kernel(crate: Path) -> list[str]: if crate.name != "cellule-runtime": return [] problems = [] - for file in (crate / "src/coordination").rglob("*.rs"): + sources = [ + *(crate / "src/coordination").rglob("*.rs"), + *(crate / "src/fleet/operations").rglob("*.rs"), + ] + for file in sources: text = COMMENT.sub("", file.read_text()) for label, pattern in SANS_IO: if pattern.search(text): - problems.append(f"{file.relative_to(ROOT)}: pure coordination kernel uses {label}") + problems.append(f"{file.relative_to(ROOT)}: pure decision kernel uses {label}") return problems