diff --git a/Cargo.lock b/Cargo.lock index 9e435fa..0ad70d3 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -360,6 +360,7 @@ dependencies = [ "serde", "sha1", "sha2 0.11.0", + "tempfile", "thiserror", ] @@ -405,6 +406,7 @@ dependencies = [ "percent-encoding", "rcgen", "reqwest 0.12.28", + "rusqlite", "russh", "serde", "serde_json", diff --git a/config.example.json b/config.example.json index 254fff3..8ea6094 100644 --- a/config.example.json +++ b/config.example.json @@ -11,5 +11,31 @@ "listen": "127.0.0.1:8080", "data_dir": "/var/lib/canopy", "local_disk_limit_bytes": 10737418240, - "max_active_repositories": 100 + "max_active_repositories": 100, + "native_limits": { + "total": { + "processes": 32, + "cpu_units": 48, + "memory_bytes": 17179869184, + "descriptors": 2048 + }, + "maintenance_reserved": { + "processes": 2, + "cpu_units": 8, + "memory_bytes": 2147483648, + "descriptors": 128 + }, + "read": { + "processes": 1, + "cpu_units": 1, + "memory_bytes": 268435456, + "descriptors": 32 + }, + "pack": { + "processes": 1, + "cpu_units": 4, + "memory_bytes": 1073741824, + "descriptors": 64 + } + } } diff --git a/crates/canopy-git-format/Cargo.toml b/crates/canopy-git-format/Cargo.toml index 5dca27f..0542d19 100644 --- a/crates/canopy-git-format/Cargo.toml +++ b/crates/canopy-git-format/Cargo.toml @@ -10,3 +10,6 @@ serde = { version = "1", features = ["derive"] } sha1 = "0.11" sha2 = "0.11" thiserror = "2.0" + +[dev-dependencies] +tempfile = "3" diff --git a/crates/canopy-git-format/src/lib.rs b/crates/canopy-git-format/src/lib.rs index cb344d9..2e08f87 100644 --- a/crates/canopy-git-format/src/lib.rs +++ b/crates/canopy-git-format/src/lib.rs @@ -1,5 +1,7 @@ //! Git object formats and canonical object identifiers. +pub mod pack_index; + use sha1::{Digest, Sha1}; use sha2::Sha256; use std::ops::Deref; @@ -146,13 +148,17 @@ pub enum ObjectHasher { } impl ObjectHasher { pub fn new(format: ObjectFormat, kind: ObjectKind, size: u64) -> Self { - let mut hash = match format { - ObjectFormat::Sha1 => Self::Sha1(Sha1::new()), - ObjectFormat::Sha256 => Self::Sha256(Sha256::new()), - }; + let mut hash = Self::raw(format); hash.update(format!("{} {size}\0", kind.git_name()).as_bytes()); hash } + /// Native pack/index checksum, without the canonical object header prefix. + pub fn raw(format: ObjectFormat) -> Self { + match format { + ObjectFormat::Sha1 => Self::Sha1(Sha1::new()), + ObjectFormat::Sha256 => Self::Sha256(Sha256::new()), + } + } pub fn update(&mut self, bytes: &[u8]) { match self { Self::Sha1(hash) => hash.update(bytes), diff --git a/crates/canopy-git-format/src/pack_index/mod.rs b/crates/canopy-git-format/src/pack_index/mod.rs new file mode 100644 index 0000000..f55d87a --- /dev/null +++ b/crates/canopy-git-format/src/pack_index/mod.rs @@ -0,0 +1,329 @@ +//! Checked, file-backed Git v2 pack indexes with bounded-memory lookup. +//! +//! The caller must keep an index immutable for this handle's lifetime. The index +//! checksum verifies its bytes, not the pack's decoded objects: admission still +//! requires native pack verification and canonical object verification. + +use crate::{ObjectFormat, ObjectHasher, ObjectId}; +use std::{ + fs::File, + io::{self, Read, Seek, SeekFrom}, + path::Path, +}; + +const HEADER: u64 = 8 + 256 * 4; +const PAGE: usize = 64 * 1024; + +fn invalid(message: &'static str) -> io::Error { + io::Error::new(io::ErrorKind::InvalidData, message) +} + +/// Immutable native index. Heap size is independent of the number of objects. +pub struct PackIndex { + file: File, + format: ObjectFormat, + fanout: [u32; 256], + count: u32, + offsets: u64, + large_offsets: u64, + large_count: u64, + checksum: ObjectId, +} + +/// An indexed object; offsets/CRCs are native hints, never publication authority. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct IndexEntry { + pub oid: ObjectId, + pub offset: u64, + pub crc32: u32, +} + +impl PackIndex { + /// Validate signature, layout, native digest, sorted IDs, fanout and offsets. + /// This intentionally rejects unsupported versions rather than guessing. + pub fn open(path: impl AsRef, format: ObjectFormat) -> io::Result { + let mut file = File::open(path)?; + let length = file.metadata()?.len(); + let width = format.bytes() as u64; + if length < HEADER + 2 * width { + return Err(invalid("truncated Git pack index")); + } + let mut header = [0; HEADER as usize]; + file.read_exact(&mut header)?; + if header[..8] != *b"\xfftOc\0\0\0\x02" { + return Err(invalid("unsupported Git pack index version")); + } + let mut fanout = [0; 256]; + let mut previous = 0; + for (bucket, value) in fanout.iter_mut().enumerate() { + let start = 8 + bucket * 4; + *value = u32::from_be_bytes(header[start..start + 4].try_into().unwrap()); + if *value < previous { + return Err(invalid("nonmonotonic Git index fanout")); + } + previous = *value; + } + let count = fanout[255]; + let offsets = HEADER + u64::from(count) * (width + 4); + let large_offsets = offsets + u64::from(count) * 4; + let payload_end = length - 2 * width; + if large_offsets > payload_end || !(payload_end - large_offsets).is_multiple_of(8) { + return Err(invalid("invalid Git pack index length")); + } + let large_count = (payload_end - large_offsets) / 8; + if large_count > u64::from(count) { + return Err(invalid("excess large Git offsets")); + } + file.seek(SeekFrom::Start(length - 2 * width))?; + let mut trailer = [0; 64]; + file.read_exact(&mut trailer[..(2 * width) as usize])?; + let checksum = ObjectId::try_from(&trailer[..width as usize]) + .map_err(|_| invalid("invalid native pack checksum"))?; + let expected_index = ObjectId::try_from(&trailer[width as usize..(2 * width) as usize]) + .map_err(|_| invalid("invalid native index checksum"))?; + file.seek(SeekFrom::Start(0))?; + let mut hash = ObjectHasher::raw(format); + let mut remaining = length - width; + let mut buffer = [0; PAGE]; + while remaining > 0 { + let size = remaining.min(PAGE as u64) as usize; + file.read_exact(&mut buffer[..size])?; + hash.update(&buffer[..size]); + remaining -= size as u64; + } + if hash.finalize() != expected_index { + return Err(invalid("Git pack index checksum mismatch")); + } + let index = Self { + file, + format, + fanout, + count, + offsets, + large_offsets, + large_count, + checksum, + }; + let mut actual = [0_u32; 256]; + let mut previous = None; + for oid in index.ids() { + let oid = oid?; + if previous.is_some_and(|previous| previous >= oid) { + return Err(invalid("Git index object IDs must be unique and sorted")); + } + actual[oid[0] as usize] += 1; + previous = Some(oid); + } + let mut total = 0; + for (count, expected) in actual.iter().zip(index.fanout) { + total += count; + if total != expected { + return Err(invalid("Git index fanout does not match object IDs")); + } + } + // Validate 32-bit offsets sequentially. Large offset references are read + // by position; no array proportional to the object count is allocated. + let mut at = index.offsets; + let mut remaining = u64::from(index.count) * 4; + let mut references = 0_u64; + while remaining > 0 { + let size = remaining.min(PAGE as u64) as usize; + read_at(&index.file, &mut buffer[..size], at)?; + for encoded in buffer[..size].as_chunks::<4>().0 { + let raw = u32::from_be_bytes(*encoded); + index.offset(raw)?; + references += u64::from(raw & 0x8000_0000 != 0); + } + at += size as u64; + remaining -= size as u64; + } + if references != large_count { + return Err(invalid("unused large Git offsets")); + } + Ok(index) + } + + pub fn len(&self) -> u32 { + self.count + } + pub fn is_empty(&self) -> bool { + self.count == 0 + } + pub fn format(&self) -> ObjectFormat { + self.format + } + pub fn pack_checksum(&self) -> ObjectId { + self.checksum + } + + /// Iterate native hash order with one 64 KiB page, including IO errors. + pub fn ids(&self) -> IndexIds<'_> { + self.ids_at(0) + } + + /// Start a bounded sequential read at a checked native ordinal. Immutable + /// metadata shards use this to cover contiguous ranges without rescanning + /// earlier index entries or materializing all IDs. + pub fn ids_from(&self, ordinal: u32) -> io::Result> { + if ordinal > self.count { + return Err(invalid("Git index ordinal out of range")); + } + Ok(self.ids_at(ordinal)) + } + + fn ids_at(&self, ordinal: u32) -> IndexIds<'_> { + IndexIds { + index: self, + next: ordinal, + buffer: Box::new([0; PAGE]), + start: 0, + end: 0, + failed: false, + } + } + + /// Check native membership without reading offsets or CRCs. + pub fn contains(&self, oid: ObjectId) -> io::Result { + Ok(self.position(oid)?.is_some()) + } + + /// Fanout narrows binary search without constructing a heap OID inventory. + pub fn find(&self, oid: ObjectId) -> io::Result> { + let Some(position) = self.position(oid)? else { + return Ok(None); + }; + let mut encoded = [0; 4]; + read_at( + &self.file, + &mut encoded, + self.offsets + u64::from(position) * 4, + )?; + let offset = self.offset(u32::from_be_bytes(encoded))?; + read_at( + &self.file, + &mut encoded, + HEADER + u64::from(self.count) * self.format.bytes() as u64 + u64::from(position) * 4, + )?; + Ok(Some(IndexEntry { + oid, + offset, + crc32: u32::from_be_bytes(encoded), + })) + } + + fn position(&self, oid: ObjectId) -> io::Result> { + if oid.format() != self.format { + return Ok(None); + } + let bucket = oid[0] as usize; + let mut low = if bucket == 0 { + 0 + } else { + self.fanout[bucket - 1] + }; + let mut high = self.fanout[bucket]; + let mut candidate = [0; 32]; + let width = self.format.bytes(); + while low < high { + let mid = low + (high - low) / 2; + read_at( + &self.file, + &mut candidate[..width], + HEADER + u64::from(mid) * width as u64, + )?; + match candidate[..width].cmp(oid.as_ref()) { + std::cmp::Ordering::Less => low = mid + 1, + std::cmp::Ordering::Greater => high = mid, + std::cmp::Ordering::Equal => return Ok(Some(mid)), + } + } + Ok(None) + } + + fn offset(&self, raw: u32) -> io::Result { + let offset = if raw & 0x8000_0000 == 0 { + u64::from(raw) + } else { + let ordinal = u64::from(raw & 0x7fff_ffff); + if ordinal >= self.large_count { + return Err(invalid("Git large offset out of range")); + } + let mut bytes = [0; 8]; + read_at(&self.file, &mut bytes, self.large_offsets + ordinal * 8)?; + u64::from_be_bytes(bytes) + }; + if offset < 12 { + return Err(invalid("Git object offset precedes pack data")); + } + Ok(offset) + } +} + +/// Bounded sequential index reader. An IO error ends the iterator. +pub struct IndexIds<'a> { + index: &'a PackIndex, + next: u32, + buffer: Box<[u8; PAGE]>, + start: usize, + end: usize, + failed: bool, +} +impl Iterator for IndexIds<'_> { + type Item = io::Result; + fn next(&mut self) -> Option { + if self.failed || self.next == self.index.count { + return None; + } + let width = self.index.format.bytes(); + if self.start == self.end { + let records = (self.index.count - self.next).min((PAGE / width) as u32) as usize; + self.end = records * width; + self.start = 0; + if let Err(error) = read_at( + &self.index.file, + &mut self.buffer[..self.end], + HEADER + u64::from(self.next) * width as u64, + ) { + self.failed = true; + return Some(Err(error)); + } + } + let oid = ObjectId::try_from(&self.buffer[self.start..self.start + width]) + .map_err(|_| invalid("invalid Git index object ID")); + self.start += width; + self.next += 1; + Some(oid) + } + fn size_hint(&self) -> (usize, Option) { + let remaining = if self.failed { + 0 + } else { + (self.index.count - self.next) as usize + }; + (0, Some(remaining)) + } +} + +#[cfg(unix)] +fn read_at(file: &File, bytes: &mut [u8], offset: u64) -> io::Result<()> { + std::os::unix::fs::FileExt::read_exact_at(file, bytes, offset) +} +#[cfg(windows)] +fn read_at(file: &File, mut bytes: &mut [u8], mut offset: u64) -> io::Result<()> { + use std::os::windows::fs::FileExt; + while !bytes.is_empty() { + match file.seek_read(bytes, offset) { + Ok(0) => return Err(io::ErrorKind::UnexpectedEof.into()), + Ok(n) => { + offset += n as u64; + bytes = &mut bytes[n..]; + } + Err(error) if error.kind() == io::ErrorKind::Interrupted => continue, + Err(error) => return Err(error), + } + } + Ok(()) +} + +#[cfg(test)] +mod tests; diff --git a/crates/canopy-git-format/src/pack_index/tests.rs b/crates/canopy-git-format/src/pack_index/tests.rs new file mode 100644 index 0000000..4fe1ecc --- /dev/null +++ b/crates/canopy-git-format/src/pack_index/tests.rs @@ -0,0 +1,262 @@ +use super::*; +use std::{ + io::Write, + process::{Command, Stdio}, +}; + +fn checksum(bytes: &[u8], format: ObjectFormat) -> ObjectId { + let mut hash = ObjectHasher::raw(format); + hash.update(bytes); + hash.finalize() +} +fn fixture(format: ObjectFormat, count: usize) -> Vec { + let width = format.bytes(); + let mut bytes = b"\xfftOc\0\0\0\x02".to_vec(); + let mut ids = Vec::new(); + for n in 0..count { + let mut oid = vec![0; width]; + oid[..4].copy_from_slice(&(n as u32).to_be_bytes()); + ids.push(oid); + } + for bucket in 0..256 { + let total = ids + .iter() + .filter(|oid| usize::from(oid[0]) <= bucket) + .count() as u32; + bytes.extend_from_slice(&total.to_be_bytes()); + } + for oid in &ids { + bytes.extend_from_slice(oid); + } + for _ in &ids { + bytes.extend_from_slice(&0_u32.to_be_bytes()); + } + for n in 0..count { + bytes.extend_from_slice(&(12_u32 + n as u32).to_be_bytes()); + } + bytes.extend_from_slice(&vec![7; width]); + bytes.extend_from_slice(checksum(&bytes, format).as_ref()); + bytes +} +fn rehash(bytes: &mut [u8], format: ObjectFormat) { + let last = bytes.len() - format.bytes(); + let digest = checksum(&bytes[..last], format); + bytes[last..].copy_from_slice(digest.as_ref()); +} +fn open(bytes: &[u8], format: ObjectFormat) -> io::Result { + let mut file = tempfile::NamedTempFile::new()?; + file.write_all(bytes)?; + PackIndex::open(file.path(), format) +} + +#[test] +fn validates_empty_and_multi_page_indexes_for_both_formats() -> io::Result<()> { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let empty = open(&fixture(format, 0), format)?; + assert!(empty.is_empty()); + let index = open(&fixture(format, 10_000), format)?; + assert_eq!(index.len(), 10_000); + assert_eq!( + index.pack_checksum(), + ObjectId::try_from(vec![7; format.bytes()]).unwrap() + ); + for (n, oid) in index.ids().enumerate() { + let oid = oid?; + assert_eq!(u32::from_be_bytes(oid[..4].try_into().unwrap()), n as u32); + assert_eq!(index.find(oid)?.unwrap().offset, 12 + n as u64); + } + // Shards start at a native ordinal, including positions crossing the + // iterator's buffer boundary. The end is a valid empty iterator. + for start in [0, 1, 2_049, 4_097, 9_999, 10_000] { + let mut count = 0; + for (n, oid) in index.ids_from(start)?.enumerate() { + let oid = oid?; + assert_eq!( + u32::from_be_bytes(oid[..4].try_into().unwrap()), + start + n as u32 + ); + count += 1; + } + assert_eq!(count, 10_000 - start); + } + assert!(index.ids_from(10_001).is_err()); + assert!(empty.ids_from(1).is_err()); + assert!(index.find(format.zero())?.is_some()); + let missing = ObjectId::try_from(vec![255; format.bytes()]).unwrap(); + assert!(index.find(missing)?.is_none()); + assert!( + index + .find(if format == ObjectFormat::Sha1 { + ObjectFormat::Sha256.zero() + } else { + ObjectFormat::Sha1.zero() + })? + .is_none() + ); + } + Ok(()) +} + +#[test] +fn rejects_truncation_and_checksum_tampering() { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let bytes = fixture(format, 3); + for length in [0, 8, 1031, 1032, bytes.len() - 1] { + assert!(open(&bytes[..length], format).is_err()); + } + let mut altered = bytes.clone(); + altered[HEADER as usize] ^= 1; + assert!(open(&altered, format).is_err()); + let mut altered = bytes; + let last = altered.len() - 1; + altered[last] ^= 1; + assert!(open(&altered, format).is_err()); + } +} + +#[test] +fn rejects_structural_corruption_even_with_recomputed_checksum() { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let original = fixture(format, 3); + for field in [ + 4, + 8, + HEADER as usize, + HEADER as usize + format.bytes(), + HEADER as usize + 3 * (format.bytes() + 4), + ] { + let mut bytes = original.clone(); + match field { + 4 => bytes[7] = 3, // unsupported version + 8 => bytes[8..12].copy_from_slice(&2_u32.to_be_bytes()), // wrong fanout + n if n == HEADER as usize => bytes[n] = 255, // unsorted IDs + n if n == HEADER as usize + format.bytes() => { + // duplicate IDs + let first = bytes[HEADER as usize..HEADER as usize + format.bytes()].to_vec(); + bytes[n..n + format.bytes()].copy_from_slice(&first); + } + n => bytes[n..n + 4].copy_from_slice(&0_u32.to_be_bytes()), // invalid offset + } + rehash(&mut bytes, format); + assert!(open(&bytes, format).is_err(), "accepted mutation {field}"); + } + let mut bytes = original; + let end = bytes.len() - 2 * format.bytes(); + bytes.splice(end..end, [0; 8]); // unreferenced offset + rehash(&mut bytes, format); + assert!(open(&bytes, format).is_err()); + } +} + +#[test] +fn supports_large_offsets_and_rejects_out_of_range_references() -> io::Result<()> { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let mut bytes = fixture(format, 1); + let offsets = HEADER as usize + format.bytes() + 4; + bytes[offsets..offsets + 4].copy_from_slice(&0x8000_0000_u32.to_be_bytes()); + let end = bytes.len() - 2 * format.bytes(); + bytes.splice(end..end, 0x1_0000_0010_u64.to_be_bytes()); + rehash(&mut bytes, format); + let index = open(&bytes, format)?; + assert_eq!(index.find(format.zero())?.unwrap().offset, 0x1_0000_0010); + bytes[offsets..offsets + 4].copy_from_slice(&0x8000_0001_u32.to_be_bytes()); + rehash(&mut bytes, format); + assert!(open(&bytes, format).is_err()); + } + Ok(()) +} + +#[test] +fn lookup_is_safe_under_concurrent_reads() -> io::Result<()> { + let index = std::sync::Arc::new(open( + &fixture(ObjectFormat::Sha256, 10_000), + ObjectFormat::Sha256, + )?); + let threads: Vec<_> = (0..4) + .map(|_| { + let index = std::sync::Arc::clone(&index); + std::thread::spawn(move || { + for oid in index.ids() { + let oid = oid.unwrap(); + assert!(index.find(oid).unwrap().is_some()); + } + }) + }) + .collect(); + for thread in threads { + thread.join().unwrap(); + } + Ok(()) +} + +#[test] +fn reads_stock_git_indexes_and_agrees_with_show_index() -> Result<(), Box> { + fn git(root: &Path, args: &[&str], input: &[u8]) -> io::Result> { + let mut child = Command::new("git") + .arg("-C") + .arg(root) + .args(args) + .env("GIT_CONFIG_NOSYSTEM", "1") + .env( + "GIT_CONFIG_GLOBAL", + if cfg!(windows) { "NUL" } else { "/dev/null" }, + ) + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .spawn()?; + child.stdin.take().unwrap().write_all(input)?; + let output = child.wait_with_output()?; + if !output.status.success() { + return Err(io::Error::other( + String::from_utf8_lossy(&output.stderr).into_owned(), + )); + } + Ok(output.stdout) + } + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let root = tempfile::TempDir::new()?; + git( + root.path(), + &[ + "init", + "--bare", + &format!("--object-format={}", format.as_str()), + ], + b"", + )?; + let mut ids = Vec::new(); + for n in 0..32 { + ids.extend_from_slice(&git( + root.path(), + &["hash-object", "-w", "--stdin"], + format!("body {n}\n").as_bytes(), + )?); + } + let hash = git( + root.path(), + &["pack-objects", "--index-version=2", "objects/pack/pack"], + &ids, + )?; + let name = String::from_utf8(hash)?.trim().to_owned(); + let path = root.path().join(format!("objects/pack/pack-{name}.idx")); + let index = PackIndex::open(&path, format)?; + assert_eq!(index.pack_checksum(), ObjectId::from_hex(&name)?); + let native = git(root.path(), &["show-index"], &std::fs::read(&path)?)?; + for row in String::from_utf8(native)?.lines() { + let mut fields = row.split_whitespace(); + let offset: u64 = fields.next().unwrap().parse()?; + let oid = ObjectId::from_hex(fields.next().unwrap())?; + let crc = u32::from_str_radix(fields.next().unwrap().trim_matches(['(', ')']), 16)?; + assert_eq!( + index.find(oid)?, + Some(IndexEntry { + oid, + offset, + crc32: crc + }) + ); + } + } + Ok(()) +} diff --git a/crates/canopy-object-storage/src/artifact.rs b/crates/canopy-object-storage/src/artifact.rs new file mode 100644 index 0000000..893bcff --- /dev/null +++ b/crates/canopy-object-storage/src/artifact.rs @@ -0,0 +1,296 @@ +//! Authenticated immutable Git artifacts, scoped to a creating operation. +//! +//! Reusing a content digest after collection uses a different operation path, +//! so a delayed provider delete cannot remove a later artifact incarnation. + +use crate::external::{self, MAX_ARTIFACT_BYTES, PART_BYTES}; +use bytes::Bytes; +use object_store::{ObjectStore, path::Path}; +use std::{sync::Arc, time::Duration}; +use tokio::io::{AsyncRead, AsyncReadExt}; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum ArtifactKind { + Pack, + Index, + Metadata, + DirectoryRun, + CatalogNode, + InputBody, + InputRoot, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ArtifactKey { + pub operation: [u8; 16], + /// Parent pack digest for pack/index/metadata; the artifact's own digest + /// for directory runs, catalog nodes and retained requests/roots. + pub binding_digest: [u8; 32], + pub kind: ArtifactKind, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ArtifactDescriptor { + pub size: u64, + pub digest: [u8; 32], + pub manifest_digest: [u8; 32], +} + +#[derive(Debug, thiserror::Error)] +pub enum ArtifactError { + #[error("artifact store failed")] + Store(#[source] object_store::Error), + #[error("artifact input failed")] + Io(#[from] std::io::Error), + #[error("artifact hashing worker failed")] + Task(#[from] tokio::task::JoinError), + #[error("artifact bytes disagree with the authenticated descriptor")] + Corrupt, + #[error("artifact transfer timed out")] + Timeout, + #[error("artifact exceeds the bounded manifest limit")] + TooLarge, +} + +#[derive(Clone)] +pub struct ArtifactStore { + store: Arc, + repository: [u8; 16], +} + +impl From for ArtifactError { + fn from(error: object_store::Error) -> Self { + if external::is_corruption(&error) { + Self::Corrupt + } else { + Self::Store(error) + } + } +} +impl ArtifactStore { + pub fn new(store: Arc, repository: [u8; 16]) -> Self { + Self { store, repository } + } + + pub fn repository(&self) -> [u8; 16] { + self.repository + } + /// In-process capability identity. Separately constructed providers are + /// never assumed equivalent from a display name or a matching repository. + /// Clone the service-owned store when carrying a local verification proof. + pub fn same_binding(&self, other: &Self) -> bool { + self.repository == other.repository && Arc::ptr_eq(&self.store, &other.store) + } + + pub fn path(&self, key: ArtifactKey, digest: [u8; 32]) -> Result { + let root = format!( + "repos/{}/git-packs/{}/{}", + hex::encode(self.repository), + uuid::Uuid::from_bytes(key.operation), + hex::encode(key.binding_digest) + ); + match key.kind { + ArtifactKind::InputBody | ArtifactKind::InputRoot if key.binding_digest != digest => { + Err(ArtifactError::Corrupt) + } + ArtifactKind::InputBody | ArtifactKind::InputRoot => { + let kind = if key.kind == ArtifactKind::InputBody { + "bodies" + } else { + "roots" + }; + Ok(Path::from(format!( + "repos/{}/git-inputs/{}/{kind}/{}", + hex::encode(self.repository), + uuid::Uuid::from_bytes(key.operation), + hex::encode(digest) + ))) + } + ArtifactKind::Pack if digest != key.binding_digest => Err(ArtifactError::Corrupt), + ArtifactKind::Pack => Ok(Path::from(format!("{root}/pack"))), + ArtifactKind::Index => Ok(Path::from(format!("{root}/index/{}", hex::encode(digest)))), + ArtifactKind::Metadata => Ok(Path::from(format!( + "{root}/metadata/{}", + hex::encode(digest) + ))), + ArtifactKind::DirectoryRun | ArtifactKind::CatalogNode + if key.binding_digest != digest => + { + Err(ArtifactError::Corrupt) + } + ArtifactKind::DirectoryRun | ArtifactKind::CatalogNode => { + let kind = if key.kind == ArtifactKind::DirectoryRun { + "directory" + } else { + "nodes" + }; + Ok(Path::from(format!( + "repos/{}/git-catalogs/{}/{kind}/{}", + hex::encode(self.repository), + uuid::Uuid::from_bytes(key.operation), + hex::encode(digest) + ))) + } + } + } + + /// Read exactly one frame and compare its verified digest before publication. + /// A canceled writer can leave unregistered staging; its operation owns it. + pub async fn put( + &self, + key: ArtifactKey, + size: u64, + digest: [u8; 32], + input: &mut (impl AsyncRead + Unpin), + ) -> Result { + if size > MAX_ARTIFACT_BYTES { + return Err(ArtifactError::TooLarge); + } + let path = self.path(key, digest)?; + let family = match key.kind { + ArtifactKind::InputBody | ArtifactKind::InputRoot => "git-inputs", + ArtifactKind::DirectoryRun | ArtifactKind::CatalogNode => "git-catalogs", + _ => "git-packs", + }; + let stage = Path::from(format!( + "repos/{}/{family}/{}/staging/{}", + hex::encode(self.repository), + uuid::Uuid::from_bytes(key.operation), + uuid::Uuid::new_v4() + )); + let mut upload = external::Upload::new(Arc::clone(&self.store), stage).await?; + let result = async { + let mut hash = blake3::Hasher::new(); + let mut remaining = size; + let mut parts = Vec::with_capacity(size.div_ceil(PART_BYTES as u64).max(1) as usize); + loop { + let length = remaining.min(PART_BYTES as u64) as usize; + let mut bytes = vec![0; length]; + tokio::time::timeout(Duration::from_secs(120), input.read_exact(&mut bytes)) + .await + .map_err(|_| ArtifactError::Timeout)??; + let (next, part, bytes) = tokio::task::spawn_blocking(move || { + hash.update(&bytes); + let part = *blake3::hash(&bytes).as_bytes(); + (hash, part, Bytes::from(bytes)) + }) + .await?; + hash = next; + parts.push(part); + upload.write(bytes).await?; + remaining -= length as u64; + if remaining == 0 { + break; + } + } + if hash.finalize().as_bytes() != &digest { + return Err(ArtifactError::Corrupt); + } + let manifest_digest = upload.publish_hashed(&path, size, &parts).await?; + Ok(ArtifactDescriptor { + size, + digest, + manifest_digest, + }) + } + .await; + let cleanup = upload.cleanup().await; + match result { + Ok(descriptor) => { + cleanup?; + Ok(descriptor) + } + Err(error) => { + if let Err(cleanup) = cleanup { + tracing::warn!(error = %cleanup,"artifact staging cleanup failed"); + } + Err(error) + } + } + } + + pub async fn read( + &self, + key: ArtifactKey, + descriptor: ArtifactDescriptor, + ) -> Result { + if descriptor.size > MAX_ARTIFACT_BYTES { + return Err(ArtifactError::TooLarge); + } + let path = self.path(key, descriptor.digest)?; + let manifest = external::open_hashed( + self.store.as_ref(), + &path, + descriptor.size, + descriptor.manifest_digest, + ) + .await?; + if descriptor.size == 0 + && (descriptor.digest != *blake3::hash(b"").as_bytes() + || manifest.part_digest(0) != Some(*blake3::hash(b"").as_bytes())) + { + return Err(ArtifactError::Corrupt); + } + Ok(ArtifactRead { + store: Arc::clone(&self.store), + path, + manifest, + descriptor, + offset: 0, + hash: Some(blake3::Hasher::new()), + }) + } +} + +/// One bounded authenticated part at a time. Errors/cancellation poison the +/// reader, preventing continuation with an incomplete checksum history. +pub struct ArtifactRead { + store: Arc, + path: Path, + manifest: external::Manifest, + descriptor: ArtifactDescriptor, + offset: u64, + hash: Option, +} +impl ArtifactRead { + pub fn descriptor(&self) -> ArtifactDescriptor { + self.descriptor + } + pub async fn next(&mut self) -> Result, ArtifactError> { + if self.offset == self.descriptor.size { + return Ok(None); + } + let mut hash = self.hash.take().ok_or(ArtifactError::Corrupt)?; + let bytes = external::read( + self.store.as_ref(), + &self.path, + &self.manifest, + self.descriptor.size, + self.offset, + ) + .await?; + let expected = self + .manifest + .part_digest(self.offset / PART_BYTES as u64) + .ok_or(ArtifactError::Corrupt)?; + let (next, valid, bytes) = tokio::task::spawn_blocking(move || { + let valid = blake3::hash(&bytes).as_bytes() == &expected; + hash.update(&bytes); + (hash, valid, bytes) + }) + .await?; + if !valid { + return Err(ArtifactError::Corrupt); + } + let end = self.offset + bytes.len() as u64; + if end == self.descriptor.size && next.finalize().as_bytes() != &self.descriptor.digest { + return Err(ArtifactError::Corrupt); + } + self.hash = Some(next); + self.offset = end; + Ok(Some(bytes)) + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/canopy-object-storage/src/artifact/tests.rs b/crates/canopy-object-storage/src/artifact/tests.rs new file mode 100644 index 0000000..516c120 --- /dev/null +++ b/crates/canopy-object-storage/src/artifact/tests.rs @@ -0,0 +1,369 @@ +use super::*; +use object_store::{ObjectStoreExt, memory::InMemory}; + +type Result = std::result::Result>; + +fn key(body: &[u8]) -> ArtifactKey { + ArtifactKey { + operation: [2; 16], + binding_digest: *blake3::hash(body).as_bytes(), + kind: ArtifactKind::Pack, + } +} + +#[tokio::test] +async fn catalog_artifacts_bind_their_own_digest_and_isolate_retired_incarnations() -> Result { + let store: Arc = Arc::new(InMemory::new()); + let artifacts = ArtifactStore::new(Arc::clone(&store), [1; 16]); + let body = b"immutable catalog fixture"; + let digest = *blake3::hash(body).as_bytes(); + for kind in [ + ArtifactKind::DirectoryRun, + ArtifactKind::CatalogNode, + ArtifactKind::InputBody, + ArtifactKind::InputRoot, + ] { + let old = ArtifactKey { + operation: [2; 16], + binding_digest: digest, + kind, + }; + let new = ArtifactKey { + operation: [3; 16], + ..old + }; + let mut wrong = old; + wrong.binding_digest[0] ^= 1; + assert!(matches!( + artifacts.path(wrong, digest), + Err(ArtifactError::Corrupt) + )); + let descriptor = artifacts + .put(old, body.len() as u64, digest, &mut body.as_slice()) + .await?; + let next = artifacts + .put(new, body.len() as u64, digest, &mut body.as_slice()) + .await?; + let path = artifacts.path(old, digest)?; + assert!(path.as_ref().contains(match kind { + ArtifactKind::InputBody | ArtifactKind::InputRoot => "/git-inputs/", + _ => "/git-catalogs/", + })); + assert_eq!( + descriptor, + artifacts + .put(old, body.len() as u64, digest, &mut body.as_slice()) + .await? + ); + store.delete(&external::part(&path, 0)).await?; + store.delete(&path).await?; + assert!(artifacts.read(old, descriptor).await.is_err()); + let mut read = artifacts.read(new, next).await?; + assert_eq!(read.next().await?.ok_or("part")?.as_ref(), body); + assert!(read.next().await?.is_none()); + } + Ok(()) +} + +#[tokio::test] +async fn roundtrip_preserves_frames_and_replays_create_only_artifacts() -> Result { + let store: Arc = Arc::new(InMemory::new()); + let artifacts = ArtifactStore::new(Arc::clone(&store), [1; 16]); + for size in [0, 1, PART_BYTES, PART_BYTES + 1] { + let body = vec![27; size]; + let key = key(&body); + let mut framed = body.clone(); + framed.extend_from_slice(b"next frame"); + let mut input = framed.as_slice(); + let descriptor = artifacts + .put(key, size as u64, key.binding_digest, &mut input) + .await?; + assert_eq!(input, b"next frame"); + assert_eq!( + descriptor, + artifacts + .put(key, size as u64, key.binding_digest, &mut body.as_slice()) + .await? + ); + let mut reader = artifacts.read(key, descriptor).await?; + assert_eq!(reader.descriptor(), descriptor); + let mut offset = 0; + while let Some(bytes) = reader.next().await? { + assert!(bytes.len() <= PART_BYTES); + assert_eq!(bytes.as_ref(), &body[offset..offset + bytes.len()]); + offset += bytes.len(); + } + assert_eq!(offset, size); + } + let objects = store.list_with_delimiter(None).await?; + assert_eq!(objects.common_prefixes.len(), 1); + // Recursive listing confirms every staging artifact was reclaimed. + let mut listing = store.list(None); + while let Some(object) = std::future::poll_fn(|cx| listing.as_mut().poll_next(cx)).await { + assert!(!object?.location.as_ref().contains("/staging/")); + } + Ok(()) +} + +#[tokio::test] +async fn invalid_input_does_not_publish_and_size_limits_precede_reads() -> Result { + let store: Arc = Arc::new(InMemory::new()); + let artifacts = ArtifactStore::new(Arc::clone(&store), [1; 16]); + let body = b"pack bytes"; + let key = key(body); + assert!(matches!( + artifacts + .put( + key, + MAX_ARTIFACT_BYTES + 1, + key.binding_digest, + &mut body.as_slice() + ) + .await, + Err(ArtifactError::TooLarge) + )); + assert!(matches!( + artifacts + .put(key, body.len() as u64, [0; 32], &mut body.as_slice()) + .await, + Err(ArtifactError::Corrupt) + )); + let wrong_input = b"wrong pack"; + assert!(matches!( + artifacts + .put( + key, + wrong_input.len() as u64, + key.binding_digest, + &mut wrong_input.as_slice() + ) + .await, + Err(ArtifactError::Corrupt) + )); + assert!(matches!( + artifacts + .put( + key, + body.len() as u64 + 1, + key.binding_digest, + &mut body.as_slice() + ) + .await, + Err(ArtifactError::Io(_)) + )); + let listing = store.list_with_delimiter(None).await?; + assert!(listing.objects.is_empty() && listing.common_prefixes.is_empty()); + Ok(()) +} + +#[tokio::test] +async fn corrupt_part_fails_before_yield_and_poisoned_reader_cannot_resume() -> Result { + let store: Arc = Arc::new(InMemory::new()); + let artifacts = ArtifactStore::new(Arc::clone(&store), [1; 16]); + let body = vec![28; PART_BYTES + 1]; + let key = key(&body); + let descriptor = artifacts + .put( + key, + body.len() as u64, + key.binding_digest, + &mut body.as_slice(), + ) + .await?; + let path = artifacts.path(key, descriptor.digest)?; + store + .put( + &external::part(&path, 0), + Bytes::from(vec![0; PART_BYTES]).into(), + ) + .await?; + let mut reader = artifacts.read(key, descriptor).await?; + assert!(matches!(reader.next().await, Err(ArtifactError::Corrupt))); + assert!(matches!(reader.next().await, Err(ArtifactError::Corrupt))); + // A retry must reject an existing wrong part rather than trusting AlreadyExists. + assert!( + artifacts + .put( + key, + body.len() as u64, + key.binding_digest, + &mut body.as_slice() + ) + .await + .is_err() + ); + Ok(()) +} + +#[tokio::test] +async fn collision_is_checked_before_manifest_publication() -> Result { + let store: Arc = Arc::new(InMemory::new()); + let artifacts = ArtifactStore::new(Arc::clone(&store), [1; 16]); + let body = b"native pack"; + let key = key(body); + let path = artifacts.path(key, key.binding_digest)?; + store + .put( + &external::part(&path, 0), + Bytes::from(vec![0; body.len()]).into(), + ) + .await?; + assert!( + artifacts + .put( + key, + body.len() as u64, + key.binding_digest, + &mut body.as_slice() + ) + .await + .is_err() + ); + assert!(matches!( + store.head(&path).await, + Err(object_store::Error::NotFound { .. }) + )); + Ok(()) +} + +#[tokio::test] +async fn existing_manifest_corruption_is_not_overwritten() -> Result { + let store: Arc = Arc::new(InMemory::new()); + let artifacts = ArtifactStore::new(Arc::clone(&store), [1; 16]); + let body = b"native index"; + let mut key = key(b"parent pack"); + key.kind = ArtifactKind::Index; + let digest = *blake3::hash(body).as_bytes(); + let descriptor = artifacts + .put(key, body.len() as u64, digest, &mut body.as_slice()) + .await?; + let path = artifacts.path(key, digest)?; + let corrupt = Bytes::from(vec![0; 48]); + store.put(&path, corrupt.clone().into()).await?; + assert!(artifacts.read(key, descriptor).await.is_err()); + assert!( + artifacts + .put(key, body.len() as u64, digest, &mut body.as_slice()) + .await + .is_err() + ); + assert_eq!(store.get(&path).await?.bytes().await?, corrupt); + Ok(()) +} + +#[tokio::test] +async fn operation_and_repository_namespaces_prevent_delete_aba() -> Result { + let store: Arc = Arc::new(InMemory::new()); + let artifacts = ArtifactStore::new(Arc::clone(&store), [1; 16]); + let other_repository = ArtifactStore::new(Arc::clone(&store), [3; 16]); + let body = b"same immutable bytes"; + let old = key(body); + let mut new = old; + new.operation = [4; 16]; + let descriptor = artifacts + .put( + old, + body.len() as u64, + old.binding_digest, + &mut body.as_slice(), + ) + .await?; + let next = artifacts + .put( + new, + body.len() as u64, + new.binding_digest, + &mut body.as_slice(), + ) + .await?; + assert_eq!(descriptor, next); + assert_ne!( + artifacts.path(old, descriptor.digest)?, + artifacts.path(new, next.digest)? + ); + assert!(other_repository.read(new, next).await.is_err()); + // An old collector's delayed deletes address only the old incarnation. + let old_path = artifacts.path(old, descriptor.digest)?; + store.delete(&old_path).await?; + store.delete(&external::part(&old_path, 0)).await?; + let mut reader = artifacts.read(new, next).await?; + assert_eq!(reader.next().await?.unwrap().as_ref(), body); + assert!(reader.next().await?.is_none()); + Ok(()) +} + +#[tokio::test] +async fn whole_digest_and_empty_part_are_authenticated() -> Result { + let store: Arc = Arc::new(InMemory::new()); + let artifacts = ArtifactStore::new(Arc::clone(&store), [1; 16]); + let body = b"index bytes"; + let mut key = key(b"parent pack"); + key.kind = ArtifactKind::Index; + let digest = *blake3::hash(body).as_bytes(); + let mut descriptor = artifacts + .put(key, body.len() as u64, digest, &mut body.as_slice()) + .await?; + descriptor.manifest_digest = [0; 32]; + assert!(artifacts.read(key, descriptor).await.is_err()); + let empty = key_for_empty(); + let descriptor = artifacts + .put(empty, 0, empty.binding_digest, &mut b"".as_slice()) + .await?; + let path = artifacts.path(empty, descriptor.digest)?; + store.delete(&external::part(&path, 0)).await?; + assert!(artifacts.read(empty, descriptor).await.is_err()); + Ok(()) +} + +#[tokio::test] +async fn final_part_is_withheld_when_whole_digest_disagrees() -> Result { + let store: Arc = Arc::new(InMemory::new()); + let artifacts = ArtifactStore::new(Arc::clone(&store), [1; 16]); + let body = b"index bytes"; + let mut key = key(b"parent pack"); + key.kind = ArtifactKind::Index; + let mut descriptor = artifacts + .put( + key, + body.len() as u64, + *blake3::hash(body).as_bytes(), + &mut body.as_slice(), + ) + .await?; + let original = artifacts.path(key, descriptor.digest)?; + descriptor.digest = [0; 32]; + let wrong = artifacts.path(key, descriptor.digest)?; + store.copy(&original, &wrong).await?; + store + .copy(&external::part(&original, 0), &external::part(&wrong, 0)) + .await?; + let mut reader = artifacts.read(key, descriptor).await?; + assert!(matches!(reader.next().await, Err(ArtifactError::Corrupt))); + assert!(matches!(reader.next().await, Err(ArtifactError::Corrupt))); + Ok(()) +} + +#[tokio::test] +async fn empty_manifest_must_certify_the_empty_part() -> Result { + let store: Arc = Arc::new(InMemory::new()); + let artifacts = ArtifactStore::new(Arc::clone(&store), [1; 16]); + let key = key(b""); + let mut descriptor = artifacts + .put(key, 0, key.binding_digest, &mut b"".as_slice()) + .await?; + let path = artifacts.path(key, descriptor.digest)?; + let mut manifest = b"CANOPY02".to_vec(); + manifest.extend_from_slice(&0_u64.to_le_bytes()); + manifest.extend_from_slice(&[0; 32]); + descriptor.manifest_digest = *blake3::hash(&manifest).as_bytes(); + store.put(&path, manifest.into()).await?; + assert!(matches!( + artifacts.read(key, descriptor).await, + Err(ArtifactError::Corrupt) + )); + Ok(()) +} + +fn key_for_empty() -> ArtifactKey { + key(b"") +} diff --git a/crates/canopy-object-storage/src/external.rs b/crates/canopy-object-storage/src/external.rs index 63557bf..303c245 100644 --- a/crates/canopy-object-storage/src/external.rs +++ b/crates/canopy-object-storage/src/external.rs @@ -9,8 +9,10 @@ use std::future::poll_fn; use std::{future::Future, sync::Arc, time::Duration}; pub const PART_BYTES: usize = 8 * 1024 * 1024; +pub const MAX_PARTS: u64 = 65_536; +pub const MAX_ARTIFACT_BYTES: u64 = PART_BYTES as u64 * MAX_PARTS; const MAGIC: &[u8; 8] = b"CANOPY01"; -const LFS_MAGIC: &[u8; 8] = b"CANOPY02"; +const HASHED_MAGIC: &[u8; 8] = b"CANOPY02"; pub struct Manifest { meta: ObjectMeta, @@ -34,13 +36,20 @@ pub fn part(path: &Path, index: u64) -> Path { Path::from(format!("{path}.parts/{index:016x}")) } +#[derive(Debug, thiserror::Error)] +#[error("invalid external body manifest or part")] +struct IntegrityError; + +/// Classify our integrity failures without treating provider/network failures +/// as corrupt input. Callers preserve their public transfer error contracts. +pub fn is_corruption(error: &object_store::Error) -> bool { + matches!(error, object_store::Error::Generic { source, .. } if source.is::()) +} + fn invalid() -> object_store::Error { object_store::Error::Generic { store: "Canopy external body", - source: Box::new(std::io::Error::new( - std::io::ErrorKind::InvalidData, - "invalid external body manifest or part", - )), + source: Box::new(IntegrityError), } } @@ -95,22 +104,47 @@ impl Upload { self.publish_manifest(path, size, manifest).await } - pub async fn publish_lfs( + pub async fn publish_hashed( &mut self, path: &Path, size: u64, digests: &[[u8; 32]], ) -> object_store::Result<[u8; 32]> { - if u64::try_from(digests.len()).ok() != Some(size.div_ceil(PART_BYTES as u64).max(1)) { + if size > MAX_ARTIFACT_BYTES + || u64::try_from(digests.len()).ok() != Some(size.div_ceil(PART_BYTES as u64).max(1)) + { return Err(invalid()); } - let mut manifest = LFS_MAGIC.to_vec(); + let mut manifest = HASHED_MAGIC.to_vec(); manifest.extend_from_slice(&size.to_le_bytes()); for digest in digests { manifest.extend_from_slice(digest); } let root = *blake3::hash(&manifest).as_bytes(); - self.publish_manifest(path, size, manifest).await?; + if self.active.is_some() || self.parts != size.div_ceil(PART_BYTES as u64).max(1) { + return Err(invalid()); + } + copy_parts(self.store.as_ref(), &self.stage, path, size).await?; + // Conditional copy collisions may name wrong or incomplete bytes. Check + // each destination part before creating its complete manifest. + for (ordinal, expected) in digests.iter().enumerate() { + let offset = ordinal as u64 * PART_BYTES as u64; + let length = size.saturating_sub(offset).min(PART_BYTES as u64) as usize; + let (_, bytes) = bounded( + self.store.as_ref(), + &part(path, ordinal as u64), + GetOptions::default(), + length, + ) + .await?; + let digest = tokio::task::spawn_blocking(move || blake3::hash(&bytes)) + .await + .map_err(|_| invalid())?; + if digest.as_bytes() != expected { + return Err(invalid()); + } + } + self.create_manifest(path, manifest).await?; Ok(root) } @@ -124,9 +158,17 @@ impl Upload { return Err(invalid()); } copy_parts(self.store.as_ref(), &self.stage, path, size).await?; + self.create_manifest(path, manifest).await + } + + async fn create_manifest( + &mut self, + path: &Path, + manifest: Vec, + ) -> object_store::Result<()> { match timed(self.store.put_opts( path, - manifest.into(), + manifest.clone().into(), PutOptions { mode: PutMode::Create, ..Default::default() @@ -134,7 +176,20 @@ impl Upload { )) .await { - Ok(_) | Err(object_store::Error::AlreadyExists { .. }) => Ok(()), + Ok(_) => Ok(()), + Err(object_store::Error::AlreadyExists { .. }) => { + let (_, existing) = bounded( + self.store.as_ref(), + path, + GetOptions::default(), + manifest.len(), + ) + .await?; + if existing.as_ref() != manifest.as_slice() { + return Err(invalid()); + } + Ok(()) + } Err(error) => Err(error), } } @@ -218,19 +273,22 @@ pub async fn open( Ok(Manifest { meta, bytes }) } -pub async fn open_lfs( +pub async fn open_hashed( store: &dyn ObjectStore, path: &Path, size: u64, root: [u8; 32], ) -> object_store::Result { + if size > MAX_ARTIFACT_BYTES { + return Err(invalid()); + } let length = usize::try_from(size.div_ceil(PART_BYTES as u64).max(1)) .ok() .and_then(|parts| parts.checked_mul(32)) .and_then(|bytes| bytes.checked_add(16)) .ok_or_else(invalid)?; let (meta, bytes) = bounded(store, path, GetOptions::default(), length).await?; - if &bytes[..8] != LFS_MAGIC + if &bytes[..8] != HASHED_MAGIC || bytes[8..16] != size.to_le_bytes() || blake3::hash(&bytes).as_bytes() != &root { diff --git a/crates/canopy-object-storage/src/lib.rs b/crates/canopy-object-storage/src/lib.rs index 82539de..279aeb6 100644 --- a/crates/canopy-object-storage/src/lib.rs +++ b/crates/canopy-object-storage/src/lib.rs @@ -1,5 +1,6 @@ //! Immutable object parts and content-addressed large Git blobs. +pub mod artifact; pub mod blob; pub mod external; diff --git a/crates/canopy-server/Cargo.toml b/crates/canopy-server/Cargo.toml index 0741f9a..cd65512 100644 --- a/crates/canopy-server/Cargo.toml +++ b/crates/canopy-server/Cargo.toml @@ -28,6 +28,7 @@ http-body = "1" object_store = "0.14.1" percent-encoding = "2" reqwest = { version = "0.12", default-features = false, features = ["json", "rustls-tls"] } +rusqlite = { version = "0.34", features = ["bundled", "hooks"] } serde = { version = "1", features = ["derive"] } serde_json = "1" sha1 = "0.11" diff --git a/crates/canopy-server/examples/benchmark.rs b/crates/canopy-server/examples/benchmark.rs index df43af5..0af45a1 100644 --- a/crates/canopy-server/examples/benchmark.rs +++ b/crates/canopy-server/examples/benchmark.rs @@ -66,6 +66,7 @@ impl Fixture { ssh: None, data_dir: self.directory.join(name), store_prefix: self.prefix.clone(), + native_limits: canopy_server::native_resources::NativeLimits::default(), local_disk_limit_bytes: 1536 * 1024 * 1024, max_active_repositories: self.active_limit, } diff --git a/crates/canopy-server/src/branch_rules/mod.rs b/crates/canopy-server/src/branch_rules/mod.rs index cc8f1b3..cc64000 100644 --- a/crates/canopy-server/src/branch_rules/mod.rs +++ b/crates/canopy-server/src/branch_rules/mod.rs @@ -280,13 +280,38 @@ pub(crate) fn policies_allow( } fn policy_statement(update: &RefUpdate) -> SqlStatement { + policy_statement_for(update, None) +} +/// The final publisher supplies only MAC-verified catalog ancestry evidence. +/// Current check-context versions/reporters and branch rules stay in this SQL. +pub(crate) fn policy_statement_with_ancestry(update: &RefUpdate, ancestry: bool) -> SqlStatement { + policy_statement_for(update, Some(ancestry)) +} +fn policy_statement_for(update: &RefUpdate, ancestry: Option) -> SqlStatement { let old = update.expected.as_ref().and_then(|old| old.oid); + let ancestry_sql = if ancestry.is_some() { + "?4" + } else { + "(?2 IS NULL OR coalesce(?2 = ?3, 0) OR EXISTS (SELECT 1 FROM commit_ancestry WHERE ancestor = ?2 AND descendant = ?3))" + }; + let mut parameters = vec![ + SqlValue::Text(update.name.clone()), + old.map_or(SqlValue::Null, |oid| SqlValue::Blob(oid.to_vec())), + update + .new_oid + .map_or(SqlValue::Null, |oid| SqlValue::Blob(oid.to_vec())), + ]; + if let Some(value) = ancestry { + parameters.push(SqlValue::Integer(i64::from(value))); + } SqlStatement { - sql: "SELECT b.deny_deletions, b.fast_forward, NOT EXISTS (SELECT 1 FROM branch_required_checks q LEFT JOIN check_contexts c ON c.name = q.context LEFT JOIN check_runs r ON r.number = (SELECT number FROM check_runs WHERE oid = ?3 AND context = q.context AND context_version = c.version ORDER BY number DESC LIMIT 1) WHERE q.reference = b.reference AND (c.enabled IS NOT 1 OR r.state IS NOT 'success' OR r.reporter IS NOT c.reporter)), (?2 IS NULL OR coalesce(?2 = ?3, 0) OR EXISTS (SELECT 1 FROM commit_ancestry WHERE ancestor = ?2 AND descendant = ?3)), b.require_pull_request FROM branch_rules b WHERE b.reference = ?1 AND b.enabled = 1".into(), - parameters: vec![SqlValue::Text(update.name.clone()), old.map_or(SqlValue::Null, |oid| SqlValue::Blob(oid.to_vec())), update.new_oid.map_or(SqlValue::Null, |oid| SqlValue::Blob(oid.to_vec()))], + sql: format!( + "SELECT b.deny_deletions, b.fast_forward, NOT EXISTS (SELECT 1 FROM branch_required_checks q LEFT JOIN check_contexts c ON c.name = q.context LEFT JOIN check_runs r ON r.number = (SELECT number FROM check_runs WHERE oid = ?3 AND context = q.context AND context_version = c.version ORDER BY number DESC LIMIT 1) WHERE q.reference = b.reference AND (c.enabled IS NOT 1 OR r.state IS NOT 'success' OR r.reporter IS NOT c.reporter)), {ancestry_sql}, b.require_pull_request FROM branch_rules b WHERE b.reference = ?1 AND b.enabled = 1" + ), + parameters, } } -fn decode_policy(sets: &[SqlResultSet]) -> cellule_runtime::Result> { +pub(crate) fn decode_policy(sets: &[SqlResultSet]) -> cellule_runtime::Result> { let set = sets .first() .ok_or(Error::Command("missing branch policy result"))?; diff --git a/crates/canopy-server/src/directory/mod.rs b/crates/canopy-server/src/directory/mod.rs index bdf057a..3e7933b 100644 --- a/crates/canopy-server/src/directory/mod.rs +++ b/crates/canopy-server/src/directory/mod.rs @@ -214,6 +214,10 @@ pub struct DirectoryCell { } impl DirectoryCell { + pub(crate) fn matches_repository_scope(&self, repository: &CellTarget) -> bool { + directory_target(repository.tenant(), repository.application()) + .is_ok_and(|target| target == self.target) + } pub fn new( application: &ApplicationHandle, target: CellTarget, diff --git a/crates/canopy-server/src/git_cache/artifacts.rs b/crates/canopy-server/src/git_cache/artifacts.rs new file mode 100644 index 0000000..e461dfe --- /dev/null +++ b/crates/canopy-server/src/git_cache/artifacts.rs @@ -0,0 +1,74 @@ +//! Direct authenticated downloads into an unpublished native workspace. +use super::*; +use crate::packs::{metadata::MetadataError, sources::NativePackDescriptor}; +use canopy_object_storage::artifact::{ArtifactKind, ArtifactStore}; + +struct Writer { + file: File, + _cache: Arc, +} +impl GitCache { + /// The isolated verifier calls this exactly once on its fresh private cache. + /// Reserve the complete pair before creating files or reading the provider. + /// No second pack copy or blob-as-artifact wrapper is involved. + pub(crate) async fn download_native( + self: &Arc, + store: &ArtifactStore, + descriptor: NativePackDescriptor, + ) -> Result<(), MetadataError> { + descriptor + .validate(store.repository(), self.object_format) + .map_err(|_| MetadataError::Integrity)?; + if self.objects.is_some() { + return Err(MetadataError::Integrity); + } + let cache = Arc::clone(self); + tokio::task::spawn_blocking(move || { + let size = descriptor + .pack + .size + .checked_add(descriptor.index.size) + .ok_or(MetadataError::Limit)?; + cache.reservation()?.try_grow(size)?; + Ok::<_, MetadataError>(()) + }) + .await??; + for (kind, artifact, extension) in [ + (ArtifactKind::Pack, descriptor.pack, "pack"), + (ArtifactKind::Index, descriptor.index, "idx"), + ] { + let cache = Arc::clone(self); + let mut writer = tokio::task::spawn_blocking(move || { + let path = cache.git_dir().join(format!( + "objects/pack/pack-{}.{}", + hex::encode(descriptor.git_checksum), + extension + )); + Ok::<_, MetadataError>(Writer { + file: File::create_new(path)?, + _cache: cache, + }) + }) + .await??; + let mut input = store + .read( + descriptor.key(kind).map_err(|_| MetadataError::Integrity)?, + artifact, + ) + .await?; + while let Some(bytes) = input.next().await? { + writer = tokio::task::spawn_blocking(move || { + writer.file.write_all(&bytes)?; + Ok::<_, MetadataError>(writer) + }) + .await??; + } + tokio::task::spawn_blocking(move || { + writer.file.sync_all()?; + Ok::<_, MetadataError>(()) + }) + .await??; + } + Ok(()) + } +} diff --git a/crates/canopy-server/src/git_cache/cleanup.rs b/crates/canopy-server/src/git_cache/cleanup.rs new file mode 100644 index 0000000..62e1fa6 --- /dev/null +++ b/crates/canopy-server/src/git_cache/cleanup.rs @@ -0,0 +1,75 @@ +//! Deferred reclamation keeps admission and alternates until the native fence +//! can be acquired. Brief inherited descriptors must not leak charges forever. + +use super::*; +use tokio::sync::{OwnedSemaphorePermit, Semaphore}; + +pub(super) struct Cleanup { + pub(super) path: PathBuf, + pub(super) reservation: Option, + pub(super) objects: Option>, +} + +impl Cleanup { + pub(super) fn defer(self) { + static SLOTS: OnceLock> = OnceLock::new(); + let slots = SLOTS.get_or_init(|| Arc::new(Semaphore::new(512))); + let (Ok(runtime), Ok(slot)) = ( + tokio::runtime::Handle::try_current(), + Arc::clone(slots).try_acquire_owned(), + ) else { + tracing::error!(path = %self.path.display(), "Git cleanup queue unavailable; admission retained until process restart"); + return; // Drop quarantines the still-owned reservation and alternate. + }; + runtime.spawn(self.reap(slot)); + } + + async fn reap(mut self, slot: OwnedSemaphorePermit) { + let mut delay = std::time::Duration::from_millis(10); + loop { + match crate::native_git::idle_fence(&self.path.join("repo.git")) { + Ok(fence) => { + // The blocking job owns all charges even if its async waiter + // is canceled while filesystem removal is still running. + let _ = tokio::task::spawn_blocking(move || { + let _slot = slot; + let _fence = fence; + match fs::remove_dir_all(&self.path) { + Ok(()) => {} + Err(error) if error.kind() == io::ErrorKind::NotFound => {} + Err(error) => { + tracing::error!(path = %self.path.display(), error = %error, "Git cleanup failed; admission retained until process restart"); + return; + } + } + // Successful removal permits normal field teardown. + self.reservation.take(); + self.objects.take(); + }).await; + return; + } + Err(error) if error.kind() == io::ErrorKind::WouldBlock => { + tokio::time::sleep(delay).await; + delay = (delay * 2).min(std::time::Duration::from_secs(5)); + } + Err(error) => { + tracing::error!(path = %self.path.display(), error = %error, "Git fence acquisition failed; admission retained until process restart"); + return; + } + } + } + } +} + +impl Drop for Cleanup { + fn drop(&mut self) { + // Runtime shutdown, full queue or filesystem failure cannot make bytes + // uncharged or invalidate an orphan worker's borrowed alternate. + if let Some(reservation) = self.reservation.take() { + std::mem::forget(reservation); + } + if let Some(objects) = self.objects.take() { + std::mem::forget(objects); + } + } +} diff --git a/crates/canopy-server/src/git_cache/maintenance.rs b/crates/canopy-server/src/git_cache/maintenance.rs index d8da15e..09028b8 100644 --- a/crates/canopy-server/src/git_cache/maintenance.rs +++ b/crates/canopy-server/src/git_cache/maintenance.rs @@ -8,46 +8,12 @@ fn worker_error(error: GitHttpError) -> CacheError { io::Error::other(error).into() } -// Native receive-pack writes v2 indexes. Reject unsupported layouts rather than -// guessing an inventory and incorrectly skipping authoritative hydration. -pub(crate) fn index_ids( - path: &Path, - format: crate::ObjectFormat, -) -> io::Result> { - let data = fs::read(path)?; - if data.get(..8) != Some(b"\xfftOc\0\0\0\x02") || data.len() < 1032 { - return Err(io::Error::other("unsupported Git pack index")); - } - let count = u32::from_be_bytes(data[1028..1032].try_into().unwrap()) as usize; - let width = if format == crate::ObjectFormat::Sha1 { - 20 - } else { - 32 - }; - let end = 1032_usize - .checked_add( - count - .checked_mul(width) - .ok_or_else(|| io::Error::other("pack index overflow"))?, - ) - .ok_or_else(|| io::Error::other("pack index overflow"))?; - let ids = data - .get(1032..end) - .ok_or_else(|| io::Error::other("truncated Git pack index"))?; - ids.chunks_exact(width) - .map(|id| { - id.try_into() - .map_err(|_| io::Error::other("invalid pack OID")) - }) - .collect() -} - // Follow physical pack order to retain delta-base locality while verifying large // histories. Hash order forces needless repeated decompression of distant bases. fn index_order(path: &Path, format: crate::ObjectFormat) -> io::Result> { let data = fs::read(path)?; - let ids = index_ids(path, format)?; - let count = ids.len(); + let validated = crate::git_format::pack_index::PackIndex::open(path, format)?; + let count = validated.len() as usize; let width = format.bytes(); let offsets = 1032_usize .checked_add( @@ -164,19 +130,11 @@ impl GitCache { let path = cache .git_dir() .join(format!("objects/pack/pack-{}.idx", hex::encode(hash))); - let data = fs::read(&path)?; - let width = cache.object_format.bytes(); - if data.len() < width * 2 - || &data[data.len() - 2 * width..data.len() - width] != hash.as_ref() - { + let index = crate::git_format::pack_index::PackIndex::open(&path, cache.object_format)?; + if index.pack_checksum() != hash { return Err(io::Error::other("index/pack binding mismatch").into()); } - let ids = index_ids(&path, cache.object_format)?; - cache - .packed - .write() - .map_err(|_| io::Error::other("verified inventory poisoned"))? - .extend(ids); + cache.register_checked_index(index)?; cache .pack_files .fetch_add(1, std::sync::atomic::Ordering::Relaxed); @@ -302,8 +260,16 @@ impl GitCache { if path.extension().and_then(|v| v.to_str()) != Some("idx") { continue; } - let ids = index_ids(&path, cache.object_format)?; - if !ids.is_subset(&verified) { + let index = + crate::git_format::pack_index::PackIndex::open(&path, cache.object_format)?; + let mut approved = true; + for oid in index.ids() { + if !verified.contains(&oid?) { + approved = false; + break; + } + } + if !approved { continue; } let pack = path.with_extension("pack"); @@ -329,18 +295,19 @@ impl GitCache { .map_err(|error| error.error)?; } } - retained += ids.len(); + retained += index.len() as usize; cache .write_generation .fetch_add(1, std::sync::atomic::Ordering::SeqCst); cache .pack_files .fetch_add(1, std::sync::atomic::Ordering::Relaxed); - cache - .packed - .write() - .map_err(|_| io::Error::other("packed inventory poisoned"))? - .extend(ids); + cache.register_index( + &cache.git_dir().join("objects/pack").join( + path.file_name() + .ok_or_else(|| io::Error::other("missing index filename"))?, + ), + )?; } // Caller serializes ingestion/hydration while this operation runs. cache.reservation()?.resize(tree_bytes(cache.root())?)?; @@ -356,14 +323,21 @@ impl GitCache { root: PathBuf, budget: DiskBudget, ) -> Result, CacheError> { - let next = Self::create(root, budget, "refs/heads/main", self.object_format).await?; + let next = Self::create( + root, + budget, + "refs/heads/main", + self.object_format, + self.native.clone(), + ) + .await?; // Native writes bypass CacheWriter. Reserve conservative scratch room // before starting, then reconcile the completed generation. This is // admission, not a hard OS disk quota (the deployment owns that quota). let reserve = self .bytes()? .checked_add( - (self.packed_count() as u64) + self.indexed_entries() .checked_mul(96) .ok_or_else(|| io::Error::other("maintenance index estimate overflow"))?, ) @@ -378,6 +352,9 @@ impl GitCache { .read() .map_err(|_| io::Error::other("durable inventory poisoned"))? .clone(); + let maintenance = self + .native + .for_class(crate::native_resources::NativeClass::Maintenance); let run = async { let mut listing_command = crate::native_git::command(&self.git_dir())?; listing_command @@ -388,7 +365,11 @@ impl GitCache { ]) .stdout(Stdio::piped()) .stderr(Stdio::piped()); - let mut listing = GitProcess::spawn(listing_command, Arc::clone(self))?; + let mut listing = GitProcess::spawn( + listing_command, + Arc::clone(self), + maintenance.try_admit(crate::native_resources::NativeWork::Read)?, + )?; let mut pack_command = crate::native_git::command(&self.git_dir())?; pack_command .args([ @@ -401,8 +382,11 @@ impl GitCache { .stdin(Stdio::piped()) .stdout(Stdio::piped()) .stderr(Stdio::piped()); - let mut packing = - GitProcess::spawn(pack_command, (Arc::clone(self), Arc::clone(&next)))?; + let mut packing = GitProcess::spawn( + pack_command, + (Arc::clone(self), Arc::clone(&next)), + maintenance.try_admit(crate::native_resources::NativeWork::Pack)?, + )?; let mut input = packing .child .stdin @@ -447,6 +431,9 @@ impl GitCache { )?; finish(&mut listing, listing_stderr).await?; finish(&mut packing, packing_stderr).await?; + // Both native workers drained; return their claims before validation. + drop(listing); + drop(packing); let hash = std::str::from_utf8(&hash) .map_err(|_| GitHttpError::MalformedCgi)? .trim(); @@ -459,11 +446,15 @@ impl GitCache { .join(format!("objects/pack/pack-{hash}.pack")); let mut command = crate::native_git::command(&next.git_dir())?; command - .args(["index-pack", "--verify"]) + .args(["index-pack", "--threads=2", "--verify"]) .arg(&pack) .stdout(Stdio::piped()) .stderr(Stdio::piped()); - let mut verify = GitProcess::spawn(command, Arc::clone(&next))?; + let mut verify = GitProcess::spawn( + command, + Arc::clone(&next), + maintenance.try_admit(crate::native_resources::NativeWork::Pack)?, + )?; let (_, stderr) = tokio::try_join!( read_bounded( verify @@ -482,8 +473,7 @@ impl GitCache { 64 << 10 ) )?; - let status = verify.child.wait().await?; - verify.disarm(); + let status = verify.wait().await?; if !status.success() { return Err(GitHttpError::GitExit { status, @@ -492,11 +482,7 @@ impl GitCache { } let next_copy = Arc::clone(&next); tokio::task::spawn_blocking(move || { - let ids = index_ids(&pack.with_extension("idx"), next_copy.object_format)?; - *next_copy - .packed - .write() - .map_err(|_| io::Error::other("packed inventory poisoned"))? = ids; + next_copy.register_index(&pack.with_extension("idx"))?; Ok::<_, io::Error>(()) }) .await @@ -520,9 +506,11 @@ impl GitCache { } } -async fn finish(process: &mut GitProcess, stderr: Vec) -> Result<(), GitHttpError> { - let status = process.child.wait().await?; - process.disarm(); +async fn finish( + process: &mut GitProcess, + stderr: Vec, +) -> Result<(), GitHttpError> { + let status = process.wait().await?; if !status.success() { return Err(GitHttpError::GitExit { status, diff --git a/crates/canopy-server/src/git_cache/mod.rs b/crates/canopy-server/src/git_cache/mod.rs index 4d2aa1e..e723b76 100644 --- a/crates/canopy-server/src/git_cache/mod.rs +++ b/crates/canopy-server/src/git_cache/mod.rs @@ -21,6 +21,9 @@ use crate::{ pub(crate) const CACHE_PREFIX: &str = "canopy-git-"; +mod artifacts; +mod cleanup; + #[derive(Debug, thiserror::Error)] pub enum CacheError { #[error("Git cache HEAD must name a valid branch")] @@ -48,14 +51,15 @@ pub(crate) enum ReceiveHook { } pub(crate) struct GitCache { - object_format: crate::ObjectFormat, + pub(crate) object_format: crate::ObjectFormat, + pub(crate) native: crate::native_resources::NativeScope, directory: tempfile::TempDir, reservation: Option, objects: Option>, // Only durable hydration writes this cache. Stripe by OID so concurrent // fetches share a completed loose object without serializing all objects. object_writes: OnceLock<[Arc>; 64]>, - packed: RwLock>, + packed: RwLock>, durable_packs: RwLock>, pub(crate) selection: Mutex<()>, pub(crate) prepared: Mutex>, @@ -71,8 +75,9 @@ impl GitCache { budget: DiskBudget, head: &str, object_format: crate::ObjectFormat, + native: crate::native_resources::NativeScope, ) -> Result, CacheError> { - Self::create_with_objects(root, budget, head, object_format, None).await + Self::create_with_objects(root, budget, head, object_format, None, native).await } pub(crate) async fn create_with_objects( @@ -81,6 +86,7 @@ impl GitCache { head: &str, object_format: crate::ObjectFormat, objects: Option>, + native: crate::native_resources::NativeScope, ) -> Result, CacheError> { if !crate::default_branch::valid_default_branch(head) { return Err(CacheError::InvalidHead); @@ -89,13 +95,14 @@ impl GitCache { tokio::task::spawn_blocking(move || { let cache = Arc::new(Self { object_format, + native, // Native workers change cwd to this cache; their paths must stay // absolute even when the node's data directory is relative. directory: tempfile::Builder::new().prefix(CACHE_PREFIX).tempdir_in(fs::canonicalize(root)?)?, reservation: Some(budget.try_reserve(0)?), objects, object_writes: OnceLock::new(), - packed: RwLock::new(HashSet::new()), + packed: RwLock::new(Vec::new()), durable_packs: RwLock::new(HashSet::new()), selection: Mutex::new(()), prepared: Mutex::new(BTreeSet::new()), @@ -198,13 +205,15 @@ impl GitCache { } fn object_present(&self, oid: crate::ObjectId) -> io::Result { - if self + for index in self .packed .read() .map_err(|_| io::Error::other("packed inventory poisoned"))? - .contains(&oid) + .iter() { - return Ok(true); + if index.contains(oid)? { + return Ok(true); + } } match fs::symlink_metadata(self.object_path(oid)) { Ok(metadata) if metadata.is_file() => Ok(true), @@ -305,11 +314,6 @@ impl GitCache { temporary .persist_noclobber(destination) .map_err(|error| error.error)?; - cache - .packed - .write() - .map_err(|_| io::Error::other("verified inventory poisoned"))? - .insert(oid); cache .loose_objects .fetch_add(1, std::sync::atomic::Ordering::Relaxed); @@ -372,11 +376,6 @@ impl GitCache { temporary .persist_noclobber(destination) .map_err(|error| error.error)?; - cache - .packed - .write() - .map_err(|_| io::Error::other("verified inventory poisoned"))? - .insert(oid); cache .loose_objects .fetch_add(1, std::sync::atomic::Ordering::Relaxed); @@ -429,8 +428,44 @@ impl GitCache { .await? } - pub(crate) fn packed_count(&self) -> usize { - self.packed.read().expect("packed inventory poisoned").len() + /// Physical index entries, including duplicates across packs. This is an + /// admission/telemetry bound, never a proof of canonical object coverage. + pub(crate) fn indexed_entries(&self) -> u64 { + self.packed + .read() + .expect("packed inventory poisoned") + .iter() + .fold(0_u64, |count, index| { + count.saturating_add(u64::from(index.len())) + }) + .saturating_add( + self.loose_objects + .load(std::sync::atomic::Ordering::Relaxed), + ) + } + + fn register_index(&self, path: &Path) -> io::Result<()> { + self.register_checked_index(crate::git_format::pack_index::PackIndex::open( + path, + self.object_format, + )?) + } + + fn register_checked_index( + &self, + index: crate::git_format::pack_index::PackIndex, + ) -> io::Result<()> { + let mut indexes = self + .packed + .write() + .map_err(|_| io::Error::other("packed inventory poisoned"))?; + if !indexes + .iter() + .any(|present| present.pack_checksum() == index.pack_checksum()) + { + indexes.push(index); + } + Ok(()) } } @@ -443,6 +478,19 @@ impl Drop for GitCache { if let Err(error) = cleanup() && error.kind() != io::ErrorKind::NotFound { + if error.kind() == io::ErrorKind::WouldBlock { + // Another short-lived fork may still hold an inherited lock. + // Orphan descendants can retain it longer. Defer reclamation + // without releasing this generation's admission or alternates. + self.directory.disable_cleanup(true); + cleanup::Cleanup { + path: self.root().to_path_buf(), + reservation: self.reservation.take(), + objects: self.objects.take(), + } + .defer(); + return; + } tracing::error!(path = %self.root().display(), error = %error, "Git cache cleanup failed; disk admission retained until process restart"); // Releasing capacity while files remain would undercount disk use. // Quarantine this reservation for the remaining process lifetime. diff --git a/crates/canopy-server/src/git_cache/tests.rs b/crates/canopy-server/src/git_cache/tests.rs index fac0ce0..b25a8df 100644 --- a/crates/canopy-server/src/git_cache/tests.rs +++ b/crates/canopy-server/src/git_cache/tests.rs @@ -1,5 +1,84 @@ use super::*; +async fn wait_for_cleanup(budget: &DiskBudget) -> Result<(), Box> { + tokio::time::timeout(std::time::Duration::from_secs(5), async { + while budget.used() != 0 { + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + }) + .await?; + Ok(()) +} + +#[tokio::test] +async fn overlapping_indexes_are_not_unique_coverage_and_registration_is_idempotent() +-> Result<(), Box> { + for format in [crate::ObjectFormat::Sha1, crate::ObjectFormat::Sha256] { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(512 << 20); + let mut packs = Vec::new(); + let mut verified = HashSet::new(); + for unique in [b"first".as_slice(), b"second".as_slice()] { + let source = GitCache::create( + root.path().into(), + budget.clone(), + "refs/heads/main", + format, + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + for body in [b"overlapping object".as_slice(), unique] { + let oid = object_id(format, ObjectKind::Blob, body); + verified.insert(oid); + source + .store_object(oid, ObjectKind::Blob, body.to_vec()) + .await?; + } + packs.push(source.repacked(root.path().into(), budget.clone()).await?); + } + let target = GitCache::create( + root.path().into(), + budget.clone(), + "refs/heads/main", + format, + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + for pack in &packs { + assert_eq!( + target + .retain_verified_packs(Arc::clone(pack), verified.clone()) + .await?, + 2 + ); + } + assert_eq!(verified.len(), 3); + assert_eq!(target.indexed_entries(), 4); + assert_eq!( + target + .retain_verified_packs(Arc::clone(&packs[0]), verified.clone()) + .await?, + 2 + ); + assert_eq!(target.indexed_entries(), 4); + drop(packs); + // Registered handles belong to copied destination indexes, not to the + // disposable source cache whose files have now been removed. + assert!( + target + .missing_objects(verified.into_iter().collect()) + .await? + .is_empty() + ); + drop(target); + wait_for_cleanup(&budget).await?; + assert_eq!(budget.used(), 0); + } + Ok(()) +} + #[tokio::test] async fn repacking_rotates_a_complete_cache_without_invalidating_active_readers() -> Result<(), Box> { @@ -10,6 +89,8 @@ async fn repacking_rotates_a_complete_cache_without_invalidating_active_readers( budget.clone(), "refs/heads/main", crate::ObjectFormat::Sha1, + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), ) .await?; let mut ids = Vec::new(); @@ -25,12 +106,14 @@ async fn repacking_rotates_a_complete_cache_without_invalidating_active_readers( let packed = cache.repacked(root.path().into(), budget.clone()).await?; assert!(packed.missing_objects(ids.clone()).await?.is_empty()); assert!(!packed.object_path(ids[0]).exists()); - assert_eq!(packed.packed_count(), 64); + assert_eq!(packed.indexed_entries(), 64); let reused = GitCache::create( root.path().into(), budget.clone(), "refs/heads/main", crate::ObjectFormat::Sha1, + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), ) .await?; assert_eq!( @@ -59,6 +142,7 @@ async fn repacking_rotates_a_complete_cache_without_invalidating_active_readers( drop(packed); assert!(reader.object_path(ids[0]).is_file()); drop(reader); + wait_for_cleanup(&budget).await?; assert_eq!(budget.used(), 0); Ok(()) } @@ -73,6 +157,8 @@ async fn concurrent_hydration_publishes_each_object_once() -> Result<(), Box Result<(), Box Result<(), Box Result<(), Box Result<(), Box Result<(), Box Result<(), Box Result { + pub(super) async fn read( + request: &GitHttpRequest, + format: crate::ObjectFormat, + ) -> Result { // Native Git owns media-type errors and command-limit hook reports. if request.content_type.as_deref() != Some("application/x-git-receive-pack-request") { return Ok(Self::OtherMedia); } match request.body.packet_prefix(PREFIX_LIMIT).await { Ok(prefix) => { - let mut parsed = match commands(&prefix) { + let mut parsed = match commands_in_format(&prefix, Some(format)) { Ok(parsed) => parsed, // Count and byte limits share the native Git rejection path. Err(InputError::TooLarge) => return Ok(Self::Limited), @@ -278,7 +281,13 @@ fn quote(value: &str) -> String { format!("'{}'", value.replace('\'', "'\\''")) } -fn commands(mut bytes: &[u8]) -> Result { +fn commands(bytes: &[u8]) -> Result { + commands_in_format(bytes, None) +} +fn commands_in_format( + mut bytes: &[u8], + format: Option, +) -> Result { let mut report_status = false; let mut sideband = false; let mut options_requested = false; @@ -343,7 +352,12 @@ fn commands(mut bytes: &[u8]) -> Result { *state = Certificate::Signature; } Certificate::Updates if payload.ends_with(b"\n") => { - parse_update(&payload[..payload.len() - 1], &mut updates, &mut names)?; + parse_update( + &payload[..payload.len() - 1], + &mut updates, + &mut names, + format, + )?; } Certificate::Signature if payload == b"push-cert-end\n" => { *state = Certificate::Done; @@ -355,7 +369,10 @@ fn commands(mut bytes: &[u8]) -> Result { } let payload = payload.strip_suffix(b"\n").unwrap_or(payload); if let Some(shallow) = payload.strip_prefix(b"shallow ") { - if !updates.is_empty() || parse_oid(shallow).is_none() { + if !updates.is_empty() + || parse_oid(shallow) + .is_none_or(|oid| format.is_some_and(|format| oid.format() != format)) + { return Err(InputError::Commands); } continue; @@ -383,7 +400,7 @@ fn commands(mut bytes: &[u8]) -> Result { certificate_body = Some(Vec::new()); continue; } - parse_update(payload, &mut updates, &mut names)?; + parse_update(payload, &mut updates, &mut names, format)?; } } @@ -410,6 +427,7 @@ fn parse_update<'a>( payload: &'a [u8], updates: &mut Vec, names: &mut BTreeSet<&'a str>, + format: Option, ) -> Result<(), InputError> { if updates.len() == MAX_UPDATES { return Err(InputError::TooLarge); @@ -423,7 +441,7 @@ fn parse_update<'a>( .next() .and_then(parse_oid) .ok_or(InputError::Commands)?; - if old.format() != new.format() { + if old.format() != new.format() || format.is_some_and(|format| old.format() != format) { return Err(InputError::Commands); } let name = std::str::from_utf8(fields.next().ok_or(InputError::Commands)?) diff --git a/crates/canopy-server/src/git_gateway/candidates/mod.rs b/crates/canopy-server/src/git_gateway/candidates/mod.rs index 8eaa4f5..292e9b3 100644 --- a/crates/canopy-server/src/git_gateway/candidates/mod.rs +++ b/crates/canopy-server/src/git_gateway/candidates/mod.rs @@ -249,7 +249,14 @@ async fn run( .stdin(Stdio::piped()) .stdout(Stdio::piped()) .stderr(Stdio::piped()); - let mut process = GitProcess::spawn(command, Arc::clone(&backend.cache))?; + let mut process = GitProcess::spawn( + command, + Arc::clone(&backend.cache), + backend + .cache + .native + .try_admit(crate::native_resources::NativeWork::Pack)?, + )?; let mut stdin = process .child .stdin @@ -277,8 +284,7 @@ async fn run( read_bounded(stdout, 128 * 1024), read_bounded(stderr, 64 * 1024) )?; - let status = process.child.wait().await?; - process.disarm(); + let status = process.wait().await?; Ok(Output { status, stdout, diff --git a/crates/canopy-server/src/git_gateway/discovery.rs b/crates/canopy-server/src/git_gateway/discovery.rs index 4aa74a1..4d65fac 100644 --- a/crates/canopy-server/src/git_gateway/discovery.rs +++ b/crates/canopy-server/src/git_gateway/discovery.rs @@ -61,6 +61,7 @@ impl GitGateway { self.disk_budget.clone(), &snapshot.head, self.repository.object_format(), + self.native.clone(), ) .await? .with_nonce(self.certificate_nonce().await?); diff --git a/crates/canopy-server/src/git_gateway/fetch.rs b/crates/canopy-server/src/git_gateway/fetch.rs index 27184dd..5b5d4a6 100644 --- a/crates/canopy-server/src/git_gateway/fetch.rs +++ b/crates/canopy-server/src/git_gateway/fetch.rs @@ -175,53 +175,10 @@ impl GitGateway { .lock() .await .extend(roots.iter().map(|oid| (*oid, unfiltered))); - if unfiltered && through == 0 { - // Count a covering OID index, not the large body table. A cache - // inventory consists exclusively of verified durable IDs. Equal - // cardinality therefore proves the entire captured Cell is warm. - let result = self - .repository - .sql - .query( - None, - SqlBatch { - statements: vec![ - SqlStatement { - sql: "SELECT COUNT(oid) FROM objects".into(), - parameters: vec![], - }, - SqlStatement { - sql: "SELECT COALESCE(MAX(sequence), 0) FROM objects".into(), - parameters: vec![], - }, - ], - }, - ) - .await - .map_err(|error| GatewayError::Cell(Box::new(error)))?; - if let (Some([SqlValue::Integer(count)]), Some([SqlValue::Integer(high_water)])) = ( - result - .output - .first() - .and_then(|set| set.rows.first()) - .map(Vec::as_slice), - result - .output - .get(1) - .and_then(|set| set.rows.first()) - .map(Vec::as_slice), - ) && usize::try_from(*count).ok() == Some(shared.packed_count()) - { - // Never invert build_cache's lock order by waiting here. - if let Ok(mut objects) = self.objects.try_lock() - && let Some(objects) = objects - .as_mut() - .filter(|objects| Arc::ptr_eq(&objects.cache, &shared)) - { - objects.through = objects.through.max(*high_water); - } - } - } + // Per-root preparation above is a coverage certificate. Physical + // index counts include cross-pack duplicates and cannot certify a + // whole-repository watermark. Durable covering packs are handled by + // hydrate_selected using their committed covered_through metadata. return Ok(()); } // Use the same native filter as upload-pack. Structure is present, so @@ -231,6 +188,7 @@ impl GitGateway { &cached.backend.git_dir(), roots, request.filter.as_deref(), + &cached.backend.cache.native, )?; let mut stats = Hydration::default(); loop { diff --git a/crates/canopy-server/src/git_gateway/maintenance.rs b/crates/canopy-server/src/git_gateway/maintenance.rs index edc43ee..fd28c88 100644 --- a/crates/canopy-server/src/git_gateway/maintenance.rs +++ b/crates/canopy-server/src/git_gateway/maintenance.rs @@ -49,7 +49,7 @@ impl GitGateway { return Ok(()); } tracing::info!(repository = %hex::encode(self.repository.repository_id()), - objects = next.packed_count(), previous_bytes = old.bytes()?, packed_bytes = next.bytes()?, + index_entries = next.indexed_entries(), previous_bytes = old.bytes()?, packed_bytes = next.bytes()?, elapsed_seconds = started.elapsed().as_secs_f64(), "published background Git repack generation"); self.pack_reader.replace(&old, Arc::clone(&next)).await; objects.cache = next; diff --git a/crates/canopy-server/src/git_gateway/mod.rs b/crates/canopy-server/src/git_gateway/mod.rs index 98407da..ee39566 100644 --- a/crates/canopy-server/src/git_gateway/mod.rs +++ b/crates/canopy-server/src/git_gateway/mod.rs @@ -38,6 +38,7 @@ mod discovery; mod fetch; mod hydration; mod maintenance; +pub mod preflight; mod push; mod ssh; @@ -108,6 +109,7 @@ pub struct GitGateway { lfs: LfsService, scratch_root: PathBuf, disk_budget: DiskBudget, + native: crate::native_resources::NativeScope, cache: Mutex>>, objects: Mutex>, push: Mutex<()>, @@ -119,17 +121,28 @@ impl GitGateway { scratch_root: PathBuf, blob_store: Arc, disk_budget: DiskBudget, + native: crate::native_resources::NativeResources, ) -> Self { + let native = native.scope(crate::native_resources::NativeClass::Foreground); let large_blobs = LargeBlobStore::new(Arc::clone(&blob_store), repository.repository_id()); - let pack_reader = Arc::clone(repository.pack_reader.get_or_init(|| { - Arc::new(crate::pack_store::PackReader::new( - Arc::clone(&blob_store), - repository.repository_id(), - scratch_root.clone(), - disk_budget.clone(), - repository.object_format(), - )) - })); + // A reader belongs to this gateway's workspace and disk admission. + // Another gateway may use a different root/budget for the same Cell. + let pack_reader = Arc::new(crate::pack_store::PackReader::new( + Arc::clone(&blob_store), + repository.repository_id(), + scratch_root.clone(), + disk_budget.clone(), + repository.object_format(), + native.clone(), + )); + { + let mut readers = repository + .pack_readers + .lock() + .expect("packed reader registry poisoned"); + readers.retain(|reader| reader.strong_count() > 0); + readers.push(Arc::downgrade(&pack_reader)); + } let lfs = LfsService::new(Arc::clone(&repository), blob_store); Self { repository, @@ -140,6 +153,7 @@ impl GitGateway { lfs, scratch_root, disk_budget, + native, cache: Mutex::new(None), objects: Mutex::new(None), push: Mutex::new(()), @@ -204,22 +218,33 @@ impl GitGateway { } let request = self.receive(request, None, admission).await?; let id = push_id.unwrap_or_else(|| uuid::Uuid::new_v4().into_bytes()); - let digest = request_digest(&request).await?; + let encoded = preflight::EncodedPush::new( + request, + &self.repository.target, + self.repository.repository_id(), + self.repository.object_format(), + actor, + id, + ) + .await?; // Upload spooling uses a private, budgeted scratch file. Serialize // the push-ID check, decode, native Git work and publication, but // do not let one slow client block another client's upload. let _push = self.push.lock().await; - if self.repository.begin_push(id, actor, digest).await? { + if self + .repository + .begin_push(id, actor, encoded.identity().request_digest) + .await? + { return Ok(http_body(with_push_id( self.repository.completed_response(id).await?, id, ))); } - let request = self.decode(request, None).await?; - return self - .handle_push(request, actor, id, digest) - .await - .map(http_body); + let preflight = encoded + .decode(&self.scratch_root, &self.disk_budget, None) + .await?; + return self.handle_push(preflight).await.map(http_body); } let request = self .receive(request, Some(MAX_FETCH_REQUEST_BYTES), admission) @@ -243,6 +268,7 @@ impl GitGateway { self.disk_budget.clone(), &head.output.reference, self.repository.object_format(), + self.native.clone(), ) .await? .with_nonce(self.certificate_nonce().await?); @@ -383,6 +409,7 @@ impl GitGateway { &snapshot.head, self.repository.object_format(), Some(Arc::clone(&shared.cache)), + self.native.clone(), ) .await?, nonce_seed: self.certificate_nonce().await?, @@ -473,9 +500,14 @@ impl GitGateway { packed_ids = Some(ids); } let mut objects = if let Some(ids) = packed_ids { - GitObjects::packed(&backend.git_dir(), ids)? + GitObjects::packed(&backend.git_dir(), ids, &backend.cache.native)? } else { - GitObjects::start(&backend.git_dir(), included, excluded)? + GitObjects::start( + &backend.git_dir(), + included, + excluded, + &backend.cache.native, + )? }; let mut batch = ObjectBatch::default(); @@ -654,30 +686,6 @@ fn http_body(response: GitHttpResponse) -> GitHttpResponse { } } -async fn request_digest(request: &GitHttpRequest) -> Result<[u8; 32], InputError> { - let mut hash = blake3::Hasher::new(); - hash.update(b"canopy-git-push-v2"); - hash.update(&[ - u8::from(request.protocol_v2), - u8::from(request.content_type.is_some()), - u8::from(request.gzip), - ]); - for field in [ - request.method.as_bytes(), - request.path_info.as_bytes(), - request.query.as_bytes(), - request - .content_type - .as_deref() - .unwrap_or_default() - .as_bytes(), - ] { - hash.update(&(field.len() as u64).to_le_bytes()); - hash.update(field); - } - request.body.digest(hash).await -} - fn with_push_id(mut response: GitHttpResponse, id: [u8; 16]) -> GitHttpResponse { response.headers.push(( "X-Canopy-Push-Id".into(), @@ -686,25 +694,64 @@ fn with_push_id(mut response: GitHttpResponse, id: [u8; 16]) -> GitHttpResponse response } -async fn git_output(git_dir: &Path, args: &[&str]) -> Result, GatewayError> { - let output = crate::native_git::command(git_dir)? +async fn git_output( + git_dir: &Path, + args: &[&str], + native: &crate::native_resources::NativeScope, +) -> Result, GatewayError> { + use crate::git_http::{GitProcess, WORKER_DEADLINE, read_bounded}; + use tokio::io::AsyncReadExt; + let mut command = crate::native_git::command(git_dir)?; + command .arg("--git-dir") .arg(git_dir) .args(args) - .output() - .await?; - if !output.status.success() { - return Err(GatewayError::Git( - String::from_utf8_lossy(&output.stderr).into_owned(), - )); - } - Ok(output.stdout) + .stdin(std::process::Stdio::null()) + .stdout(std::process::Stdio::piped()) + .stderr(std::process::Stdio::piped()); + let mut process = GitProcess::spawn( + command, + (), + native.try_admit(crate::native_resources::NativeWork::Read)?, + )?; + let mut stdout = process + .child + .stdout + .take() + .ok_or(GatewayError::MalformedCache)?; + let stderr = process + .child + .stderr + .take() + .ok_or(GatewayError::MalformedCache)?; + let run = async { + let mut bytes = Vec::new(); + let read_stdout = async { + stdout.read_to_end(&mut bytes).await?; + Ok::<_, GitHttpError>(()) + }; + let ((), stderr) = tokio::try_join!(read_stdout, read_bounded(stderr, 64 << 10))?; + let status = process.wait().await?; + if !status.success() { + return Err(GatewayError::Git( + String::from_utf8_lossy(&stderr).into_owned(), + )); + } + Ok(bytes) + }; + tokio::time::timeout(WORKER_DEADLINE, run) + .await + .map_err(|_| GitHttpError::Timeout)? } -async fn git_refs(git_dir: &Path) -> Result, GatewayError> { +async fn git_refs( + git_dir: &Path, + native: &crate::native_resources::NativeScope, +) -> Result, GatewayError> { let listing = git_output( git_dir, &["for-each-ref", "--format=%(refname)%00%(objectname)"], + native, ) .await?; let mut refs = BTreeMap::new(); diff --git a/crates/canopy-server/src/git_gateway/preflight.rs b/crates/canopy-server/src/git_gateway/preflight.rs new file mode 100644 index 0000000..b2b3d51 --- /dev/null +++ b/crates/canopy-server/src/git_gateway/preflight.rs @@ -0,0 +1,136 @@ +//! An owned handoff from authenticated encoded input to parsed native input. +//! Wire commands remain intent, not ref versions or a signature witness. +use super::*; +use crate::packs::publication::{BeginRequest, DEFAULT_LEASE_MS}; +use cellule_runtime::CellTarget; + +/// Construct only after the gateway's account/access authentication. Ownership +/// keeps hashing, gzip expansion and packet parsing on one request spool. +#[must_use] +pub struct EncodedPush { + request: GitHttpRequest, + identity: BeginRequest, + format: crate::ObjectFormat, + target: CellTarget, + content_digest: [u8; 32], +} +#[must_use] +pub struct PushPreflight(PushParts); +pub(super) struct PushParts { + pub request: GitHttpRequest, + pub commands: branch_policy::PushCommands, + pub identity: BeginRequest, +} +impl EncodedPush { + pub async fn new( + request: GitHttpRequest, + target: &CellTarget, + repository: [u8; 16], + format: crate::ObjectFormat, + actor: &str, + operation: [u8; 16], + ) -> Result { + if !request.authenticated { + return Err(GatewayError::Unauthorized); + } + if request.method != "POST" + || request.path_info != "/repo.git/git-receive-pack" + || crate::directory::validate_component(actor).is_err() + || crate::repository_target(target.tenant(), target.application(), repository) + .map_err(|_| GatewayError::MalformedCache)? + != *target + { + return Err(GatewayError::MalformedCache); + } + // Bind the exact encoded request, including compressed representation, + // to its server context. Native Git will read expanded bytes later. + let mut hash = blake3::Hasher::new(); + hash.update(b"canopy.git.push-request.v3\0"); + for bytes in [ + target.tenant().as_bytes().as_slice(), + target.application().as_bytes().as_slice(), + target.namespace().as_bytes().as_slice(), + target.partition(), + repository.as_slice(), + operation.as_slice(), + actor.as_bytes(), + ] { + field(&mut hash, bytes); + } + hash.update(&[ + format.bytes() as u8, + u8::from(request.protocol_v2), + u8::from(request.content_type.is_some()), + u8::from(request.gzip), + ]); + for bytes in [ + request.method.as_bytes(), + request.path_info.as_bytes(), + request.query.as_bytes(), + request + .content_type + .as_deref() + .unwrap_or_default() + .as_bytes(), + ] { + field(&mut hash, bytes); + } + let (request_digest, content_digest) = request.body.digest(hash).await?; + Ok(Self { + request, + format, + target: target.clone(), + content_digest, + identity: BeginRequest { + repository, + operation, + request_digest, + actor: actor.into(), + lease_ms: DEFAULT_LEASE_MS, + }, + }) + } + pub fn identity(&self) -> &BeginRequest { + &self.identity + } + /// Completed-request replay uses identity before this step. Consume the + /// encoded owner once, then keep normalized input and parsed intent together. + pub async fn decode( + self, + root: &Path, + budget: &DiskBudget, + limit: Option, + ) -> Result { + let mut request = self.request; + if request.gzip { + request.body = request.body.decode_gzip(root, budget, limit).await?; + request.gzip = false; + } + let commands = branch_policy::PushCommands::read(&request, self.format).await?; + Ok(PushPreflight(PushParts { + request, + commands, + identity: self.identity, + })) + } +} +impl PushPreflight { + pub fn identity(&self) -> &BeginRequest { + &self.0.identity + } + pub fn into_native_request(self) -> GitHttpRequest { + self.0.request + } + pub(super) fn into_parts(self) -> PushParts { + self.0 + } +} +fn field(hash: &mut blake3::Hasher, bytes: &[u8]) { + hash.update(&(bytes.len() as u64).to_le_bytes()); + hash.update(bytes); +} + +mod retention; +#[cfg(test)] +mod tests; +pub use retention::{RequestRetentionError, SavedPushRequest}; diff --git a/crates/canopy-server/src/git_gateway/preflight/retention.rs b/crates/canopy-server/src/git_gateway/preflight/retention.rs new file mode 100644 index 0000000..35c1ad1 --- /dev/null +++ b/crates/canopy-server/src/git_gateway/preflight/retention.rs @@ -0,0 +1,239 @@ +use super::*; +use crate::packs::{ + publication::{ + InputCheckpointError, LeaseCheck, NativeInputCertificate, PreparationSession, + StagingContext, StagingError, + }, + wire_request::{WireRequest, WireRequestError, WireRequestRoot}, +}; +use canopy_object_storage::artifact::ArtifactStore; + +/// Only an owned authenticated preflight can create this local signing input. +pub struct SavedPushRequest { + root: WireRequestRoot, + target: CellTarget, + identity: BeginRequest, + format: crate::ObjectFormat, +} +impl SavedPushRequest { + pub(crate) fn scoped_root( + &self, + target: &CellTarget, + check: &LeaseCheck, + format: crate::ObjectFormat, + ) -> Result { + if self.target != *target + || self.format != format + || self.identity.repository != check.token.repository + || self.identity.operation != check.token.operation + || self.identity.request_digest != check.token.request_digest + || self.identity.actor != check.actor + || self.root.operation() != check.token.artifact_operation + { + return Err(StagingError::Context.into()); + } + Ok(self.root) + } +} +#[derive(Debug, thiserror::Error)] +pub enum RequestRetentionError { + #[error("request preflight failed")] + Gateway(#[from] GatewayError), + #[error("request spool failed")] + Input(#[from] InputError), + #[error("request root failed")] + Wire(#[from] WireRequestError), + #[error("request checkpoint failed")] + Checkpoint(#[from] InputCheckpointError), + #[error("request custody differs")] + Context, +} +impl EncodedPush { + pub async fn retain( + self, + context: &StagingContext, + store: &ArtifactStore, + ) -> Result<(Self, SavedPushRequest), RequestRetentionError> { + context + .check_push_identity(&self.target, &self.identity, self.format) + .await?; + let token = context.token().map_err(InputCheckpointError::from)?; + if store.repository() != token.repository { + return Err(RequestRetentionError::Context); + } + let Self { + request, + identity, + format, + target, + content_digest, + } = self; + let GitHttpRequest { + method, + path_info, + query, + content_type, + gzip, + protocol_v2, + authenticated, + body, + } = request; + let (body, artifact) = body + .retain(store, token.artifact_operation, content_digest) + .await?; + let root = WireRequestRoot::upload( + store, + WireRequest { + tenant: target.tenant(), + application: target.application(), + operation: token.artifact_operation, + identity: identity.clone(), + format, + request: GitHttpRequest { + method: method.clone(), + path_info: path_info.clone(), + query: query.clone(), + content_type: content_type.clone(), + gzip, + protocol_v2, + authenticated, + body: artifact, + }, + }, + ) + .await?; + context + .check_push_identity(&target, &identity, format) + .await?; + let saved = SavedPushRequest { + root, + target: target.clone(), + identity: identity.clone(), + format, + }; + Ok(( + Self { + request: GitHttpRequest { + method, + path_info, + query, + content_type, + gzip, + protocol_v2, + authenticated, + body, + }, + identity, + format, + target, + content_digest, + }, + saved, + )) + } +} +async fn reopen( + proof: &NativeInputCertificate, + custody: (&CellTarget, &LeaseCheck, crate::ObjectFormat), + store: &ArtifactStore, + directory: &Path, + budget: &DiskBudget, + limit: Option, + admission: Option>, +) -> Result { + let (target, check, format) = custody; + let root = proof + .wire_request() + .map_err(WireRequestError::from)? + .ok_or(RequestRetentionError::Context)?; + let record = root.read(store).await?; + if !record.matches(target, check, format) { + return Err(RequestRetentionError::Context); + } + if limit.is_some_and(|limit| record.request.body.size > limit) { + return Err(InputError::TooLarge.into()); + } + let key = record.body_key(); + let artifact = store + .read(key, record.request.body) + .await + .map_err(WireRequestError::from)?; + let body = GitInput::reopen(artifact, directory, budget, limit, admission).await?; + let request = record.request; + let encoded = EncodedPush::new( + GitHttpRequest { + method: request.method, + path_info: request.path_info, + query: request.query, + content_type: request.content_type, + gzip: request.gzip, + protocol_v2: request.protocol_v2, + authenticated: request.authenticated, + body, + }, + target, + record.identity.repository, + record.format, + &record.identity.actor, + record.identity.operation, + ) + .await?; + if encoded.identity != record.identity { + return Err(RequestRetentionError::Context); + } + Ok(encoded) +} +impl StagingContext { + pub async fn reopen_push_request( + &self, + store: &ArtifactStore, + directory: &Path, + budget: &DiskBudget, + limit: Option, + admission: Option>, + ) -> Result { + let (proof, target, check, format) = self.push_checkpoint().await?; + let encoded = reopen( + &proof, + (&target, &check, format), + store, + directory, + budget, + limit, + admission, + ) + .await?; + let (current, _, _, _) = self.push_checkpoint().await?; + if current != proof { + return Err(RequestRetentionError::Context); + } + Ok(encoded) + } +} +impl PreparationSession { + pub async fn reopen_push_request( + &self, + store: &ArtifactStore, + directory: &Path, + budget: &DiskBudget, + limit: Option, + admission: Option>, + ) -> Result { + let (proof, target, check, format) = self.push_checkpoint().await?; + let encoded = reopen( + &proof, + (&target, &check, format), + store, + directory, + budget, + limit, + admission, + ) + .await?; + let (current, _, _, _) = self.push_checkpoint().await?; + if current != proof { + return Err(RequestRetentionError::Context); + } + Ok(encoded) + } +} diff --git a/crates/canopy-server/src/git_gateway/preflight/tests.rs b/crates/canopy-server/src/git_gateway/preflight/tests.rs new file mode 100644 index 0000000..e8a56c5 --- /dev/null +++ b/crates/canopy-server/src/git_gateway/preflight/tests.rs @@ -0,0 +1,354 @@ +use super::*; +use cellule_runtime::{ApplicationId, TenantId}; +use std::io::Write; + +type Result = std::result::Result>; + +struct Fixture { + root: tempfile::TempDir, + disk: DiskBudget, + repository: [u8; 16], + target: CellTarget, +} +impl Fixture { + fn new() -> Result { + let repository = *uuid::Uuid::parse_str("12345678-1234-4234-8234-123456789abc")?.as_bytes(); + Ok(Self { + root: tempfile::TempDir::new()?, + disk: DiskBudget::new(1 << 20), + repository, + target: crate::repository_target( + TenantId::from_bytes([12; 16]), + ApplicationId::from_bytes([13; 16]), + repository, + )?, + }) + } + async fn request(&self, bytes: Vec) -> Result { + Ok(GitHttpRequest { + method: "POST".into(), + path_info: "/repo.git/git-receive-pack".into(), + query: String::new(), + content_type: Some("application/x-git-receive-pack-request".into()), + gzip: false, + protocol_v2: false, + authenticated: true, + body: GitInput::receive(Body::from(bytes), self.root.path(), &self.disk, None, None) + .await?, + }) + } + async fn encoded( + &self, + request: GitHttpRequest, + format: crate::ObjectFormat, + ) -> Result { + Ok(EncodedPush::new( + request, + &self.target, + self.repository, + format, + "owner", + [2; 16], + ) + .await?) + } +} +fn packet(bytes: &[u8]) -> Vec { + [ + format!("{:04x}", bytes.len() + 4).into_bytes(), + bytes.to_vec(), + ] + .concat() +} +fn commands(format: crate::ObjectFormat, options: Option<&str>, signed: bool) -> Vec { + let line = format!( + "{} {} refs/heads/main", + "00".repeat(format.bytes()), + "12".repeat(format.bytes()) + ); + let capabilities = if options.is_some() { + "report-status side-band-64k push-options" + } else { + "report-status" + }; + let mut bytes = if signed { + let mut bytes = packet(format!("push-cert\0{capabilities}\n").as_bytes()); + bytes.extend(packet(b"certificate version 0.1\n")); + if let Some(option) = options { + bytes.extend(packet(format!("push-option {option}\n").as_bytes())); + } + for line in [ + "\n".to_owned(), + format!("{line}\n"), + "-----BEGIN SSH SIGNATURE-----\n".to_owned(), + "unverified-signature\n".to_owned(), + "-----END SSH SIGNATURE-----\n".to_owned(), + "push-cert-end\n".to_owned(), + ] { + bytes.extend(packet(line.as_bytes())); + } + bytes + } else { + packet(format!("{line}\0{capabilities}\n").as_bytes()) + }; + bytes.extend(b"0000"); + if let Some(option) = options { + bytes.extend(packet(option.as_bytes())); + bytes.extend(b"0000"); + } + bytes.extend(b"PACKnative-only"); + bytes +} +fn gzip(bytes: &[u8]) -> Result> { + let mut encoder = flate2::write::GzEncoder::new(Vec::new(), flate2::Compression::default()); + encoder.write_all(bytes)?; + Ok(encoder.finish()?) +} + +#[tokio::test] +async fn identity_binds_scope_actor_operation_format_metadata_and_encoded_bytes() -> Result { + let fixture = Fixture::new()?; + let bytes = commands(crate::ObjectFormat::Sha1, None, false); + let original = fixture + .encoded( + fixture.request(bytes.clone()).await?, + crate::ObjectFormat::Sha1, + ) + .await?; + let digest = original.identity().request_digest; + let repeated = fixture + .encoded( + fixture.request(bytes.clone()).await?, + crate::ObjectFormat::Sha1, + ) + .await?; + assert_eq!(digest, repeated.identity().request_digest); + drop(repeated); + for mutation in 0..10 { + let mut request = fixture.request(bytes.clone()).await?; + let mut target = fixture.target.clone(); + let mut repository = fixture.repository; + let mut operation = [2; 16]; + let mut actor = "owner"; + let mut format = crate::ObjectFormat::Sha1; + match mutation { + 0 => { + target = crate::repository_target( + TenantId::from_bytes([14; 16]), + target.application(), + repository, + )? + } + 1 => { + target = crate::repository_target( + target.tenant(), + ApplicationId::from_bytes([14; 16]), + repository, + )? + } + 2 => { + repository[15] ^= 1; + target = + crate::repository_target(target.tenant(), target.application(), repository)?; + } + 3 => operation[15] ^= 1, + 4 => actor = "another", + 5 => format = crate::ObjectFormat::Sha256, + 6 => request.query = "service=git-receive-pack".into(), + 7 => request.protocol_v2 = true, + 8 => request.content_type = None, + 9 => request.gzip = true, + _ => unreachable!(), + } + let changed = + EncodedPush::new(request, &target, repository, format, actor, operation).await?; + assert_ne!( + digest, + changed.identity().request_digest, + "mutation {mutation}" + ); + } + let changed = fixture + .encoded( + fixture + .request([bytes, b"different pack".to_vec()].concat()) + .await?, + crate::ObjectFormat::Sha1, + ) + .await?; + assert_ne!(digest, changed.identity().request_digest); + drop(changed); + drop(original); + assert_eq!(fixture.disk.used(), 0); + Ok(()) +} + +#[tokio::test] +async fn encoded_gzip_identity_survives_normalization_and_native_spool_rewind() -> Result { + for format in [crate::ObjectFormat::Sha1, crate::ObjectFormat::Sha256] { + let fixture = Fixture::new()?; + let bytes = commands(format, Some("canopy.note=normalized"), false); + let plain = fixture + .encoded(fixture.request(bytes.clone()).await?, format) + .await?; + let plain_digest = plain.identity().request_digest; + drop(plain); + let wire = gzip(&bytes)?; + let mut request = fixture.request(wire.clone()).await?; + request.gzip = true; + let encoded = fixture.encoded(request, format).await?; + let digest = encoded.identity().request_digest; + assert_ne!(digest, plain_digest); + assert_eq!(fixture.disk.used(), wire.len() as u64); + let parts = encoded + .decode(fixture.root.path(), &fixture.disk, Some(bytes.len() as u64)) + .await? + .into_parts(); + assert_eq!(parts.identity.request_digest, digest); + assert_eq!(parts.identity.operation, [2; 16]); + assert_eq!(parts.identity.actor, "owner"); + assert!(!parts.request.gzip); + assert_eq!(fixture.disk.used(), bytes.len() as u64); + assert_eq!(parts.request.body.prefix(bytes.len()).await?, bytes); + assert_eq!(parts.commands.options(), ["canopy.note=normalized"]); + assert_eq!(parts.commands.option_error(), None); + let branch_policy::PushCommands::Parsed { updates, .. } = &parts.commands else { + panic!("parsed commands") + }; + assert_eq!(updates.len(), 1); + assert_eq!(updates[0].new_oid.unwrap().format(), format); + assert!(updates[0].expected.is_none()); + drop(parts); + assert_eq!(fixture.disk.used(), 0); + } + Ok(()) +} + +#[tokio::test] +async fn repository_format_applies_to_zero_ids_signed_commands_and_shallow_ids() -> Result { + for expected in [crate::ObjectFormat::Sha1, crate::ObjectFormat::Sha256] { + let fixture = Fixture::new()?; + let wrong = if expected == crate::ObjectFormat::Sha1 { + crate::ObjectFormat::Sha256 + } else { + crate::ObjectFormat::Sha1 + }; + for signed in [false, true] { + let bytes = commands(wrong, None, signed); + let all_zero = String::from_utf8(bytes.clone())? + .replace(&"12".repeat(wrong.bytes()), &"00".repeat(wrong.bytes())) + .into_bytes(); + for bytes in [bytes, all_zero] { + let encoded = fixture + .encoded(fixture.request(bytes).await?, expected) + .await?; + assert!(matches!( + encoded + .decode(fixture.root.path(), &fixture.disk, None) + .await, + Err(GatewayError::Input(InputError::Commands)) + )); + } + } + let mut bytes = packet(format!("shallow {}\n", "12".repeat(wrong.bytes())).as_bytes()); + bytes.extend(commands(expected, None, false)); + let encoded = fixture + .encoded(fixture.request(bytes).await?, expected) + .await?; + assert!(matches!( + encoded + .decode(fixture.root.path(), &fixture.disk, None) + .await, + Err(GatewayError::Input(InputError::Commands)) + )); + assert_eq!(fixture.disk.used(), 0); + } + Ok(()) +} + +#[tokio::test] +async fn signed_intent_preserves_certificate_and_checks_separate_option_group() -> Result { + let fixture = Fixture::new()?; + for mismatch in [false, true] { + let mut bytes = commands( + crate::ObjectFormat::Sha256, + Some("canopy.note=signed"), + true, + ); + if mismatch { + let position = bytes + .windows(b"canopy.note=signed".len()) + .rposition(|part| part == b"canopy.note=signed") + .unwrap(); + bytes[position + b"canopy.note=".len()] = b'x'; + } + let encoded = fixture + .encoded( + fixture.request(bytes.clone()).await?, + crate::ObjectFormat::Sha256, + ) + .await?; + let parts = encoded + .decode(fixture.root.path(), &fixture.disk, None) + .await? + .into_parts(); + let certificate = parts.commands.certificate().expect("certificate intent"); + assert!( + certificate + .windows(b"unverified-signature".len()) + .any(|part| part == b"unverified-signature") + ); + assert_eq!(parts.commands.option_error().is_some(), mismatch); + assert_eq!(parts.request.body.prefix(bytes.len()).await?, bytes); + drop(parts); + assert_eq!(fixture.disk.used(), 0); + } + Ok(()) +} + +#[tokio::test] +async fn invalid_scope_authentication_and_transport_never_return_a_preflight() -> Result { + let fixture = Fixture::new()?; + for mutation in 0..4 { + let mut request = fixture.request(b"invalid gzip".to_vec()).await?; + let mut actor = "owner"; + let mut repository = fixture.repository; + match mutation { + 0 => request.authenticated = false, + 1 => request.method = "GET".into(), + 2 => actor = "bad/actor", + 3 => repository[15] ^= 1, + _ => unreachable!(), + } + let result = EncodedPush::new( + request, + &fixture.target, + repository, + crate::ObjectFormat::Sha1, + actor, + [2; 16], + ) + .await; + assert!(if mutation == 0 { + matches!(result, Err(GatewayError::Unauthorized)) + } else { + matches!(result, Err(GatewayError::MalformedCache)) + }); + assert_eq!(fixture.disk.used(), 0); + } + for wire in [b"invalid gzip".to_vec(), gzip(&vec![b'x'; 100_000])?] { + let mut request = fixture.request(wire).await?; + request.gzip = true; + let encoded = fixture.encoded(request, crate::ObjectFormat::Sha1).await?; + assert!(matches!( + encoded + .decode(fixture.root.path(), &fixture.disk, Some(100)) + .await, + Err(GatewayError::Input( + InputError::Gzip(_) | InputError::TooLarge + )) + )); + assert_eq!(fixture.disk.used(), 0); + } + Ok(()) +} diff --git a/crates/canopy-server/src/git_gateway/push.rs b/crates/canopy-server/src/git_gateway/push.rs index 81a7203..9781609 100644 --- a/crates/canopy-server/src/git_gateway/push.rs +++ b/crates/canopy-server/src/git_gateway/push.rs @@ -3,12 +3,16 @@ use super::*; impl GitGateway { pub(super) async fn handle_push( &self, - request: GitHttpRequest, - actor: &str, - id: [u8; 16], - digest: [u8; 32], + preflight: preflight::PushPreflight, ) -> Result { - let commands = branch_policy::PushCommands::read(&request).await?; + let preflight::PushParts { + request, + commands, + identity, + } = preflight.into_parts(); + let actor = identity.actor.as_str(); + let id = identity.operation; + let digest = identity.request_digest; let option_error = commands.option_error().or_else(|| { (commands.certificate().is_some() && self.signer_directory.is_none()) .then_some("Canopy signed pushes are unavailable on this gateway") @@ -29,11 +33,11 @@ impl GitGateway { |path| cached.backend.with_signers(path), ); let mut response = backend.run(request).await?; - let certificate = self.verified_certificate(&cached, &commands, actor).await?; + let certificate = self.verified_certificate(&cached, &commands, actor, digest).await?; // Git may accept some refs and reject others unless atomic was requested. // Publish its actual changes before returning any successful per-ref report. let plan = if response.status == 200 { - let after = git_refs(&cached.backend.git_dir()).await?; + let after = git_refs(&cached.backend.git_dir(), &cached.backend.cache.native).await?; let plan = diff_refs(&before, &after, actor); if plan.updates.is_empty() { None @@ -119,6 +123,7 @@ impl GitGateway { cached: &CachedRepository, commands: &branch_policy::PushCommands, actor: &str, + request_digest: [u8; 32], ) -> Result, GatewayError> { let Some(body) = commands.certificate() else { return Ok(None); @@ -165,6 +170,8 @@ impl GitGateway { )); } Ok(Some(crate::push::VerifiedPushCertificate { + target: self.repository.target.clone(), + request_digest, body: body.to_vec(), signer: signer.into(), key: key.into(), diff --git a/crates/canopy-server/src/git_gateway/ssh.rs b/crates/canopy-server/src/git_gateway/ssh.rs index 52e6042..e716045 100644 --- a/crates/canopy-server/src/git_gateway/ssh.rs +++ b/crates/canopy-server/src/git_gateway/ssh.rs @@ -34,7 +34,15 @@ impl GitGateway { if protocol_v2 { command.env("GIT_PROTOCOL", "version=2"); } - let mut process = GitProcess::spawn(command, (Arc::clone(&cached), admission))?; + let mut process = GitProcess::spawn( + command, + (Arc::clone(&cached), admission), + cached + .backend + .cache + .native + .try_admit(crate::native_resources::NativeWork::Pack)?, + )?; let mut stdin = process .child .stdin @@ -101,8 +109,7 @@ impl GitGateway { result = &mut output => result?, result = &mut input => { result?; output.await? } }; - let status = process.child.wait().await?; - process.disarm(); + let status = process.wait().await?; if !status.success() { return Err(GitHttpError::GitExit { status, diff --git a/crates/canopy-server/src/git_http/capture.rs b/crates/canopy-server/src/git_http/capture.rs new file mode 100644 index 0000000..2842306 --- /dev/null +++ b/crates/canopy-server/src/git_http/capture.rs @@ -0,0 +1,366 @@ +//! Capture request-private native inputs in the admitted creating namespace. +use super::*; +use crate::packs::{ + directory::index::IndexError, + metadata::{ + MetadataError, + transport::{PinnedFile, upload_file}, + }, + publication::{StagingContext, StagingError}, + sources::NativePackDescriptor, + verification::PhysicalLimits, +}; +use canopy_object_storage::{ + artifact::{ArtifactKind, ArtifactStore}, + external::MAX_ARTIFACT_BYTES, +}; +use std::fs::File; + +const MAX_CAPTURE_PACKS: usize = 32; +#[derive(Debug, thiserror::Error)] +pub enum NativeCaptureError { + #[error("native input custody failed")] + Staging(#[from] StagingError), + #[error("native input context or inventory is invalid")] + Context, + #[error("native input exceeds its admitted limits")] + Limit, + #[error("native input binding failed")] + Binding(#[from] IndexError), + #[error("native input upload failed")] + Upload(#[from] MetadataError), + #[error("native input cache failed")] + Cache(#[from] CacheError), + #[error("native input I/O failed")] + Io(#[from] std::io::Error), + #[error("native input worker failed")] + Task(#[from] tokio::task::JoinError), +} +struct CapturePin { + // Native commands require a shared lock; this exclusive fence keeps every + // captured file immutable while hash/upload background jobs retain it. + _fence: File, + cache: Arc, +} +impl Drop for CapturePin { + fn drop(&mut self) { + // CLOEXEC only closes accidental copies when unrelated children exec. + // Closing this parent's descriptor alone can leave the exclusive lock + // alive in such a child. All capture readers have drained when the last + // Arc drops, so explicitly release their lock before cache ownership. + // Native workers' shared locks still follow their descendants instead. + if let Err(error) = self._fence.unlock() { + // Failure stays conservative: native admission still probes the + // lock, and cache cleanup defers while any inherited lock survives. + tracing::error!(error = %error, "completed native capture fence unlock failed"); + } + } +} +struct InputFile { + path: PathBuf, + _pin: Arc, +} +impl PinnedFile for InputFile { + fn open(&self) -> std::io::Result { + File::open(&self.path) + } +} +struct Pair { + native: NativePackDescriptor, + pack: Arc, + index: Arc, +} +impl GitHttpBackend { + /// Run inside a StagingTicket producer after native receive completes. The + /// returned inputs establish authenticated bytes, not physical decoding, + /// closure, ref authorization or a durable completed network response. + /// No object bodies, OID inventory or legacy Git rows are constructed here. + pub async fn stage_native_packs( + &self, + context: &StagingContext, + store: &ArtifactStore, + limits: PhysicalLimits, + ) -> Result, NativeCaptureError> { + let token = context.token()?; + if token.repository != store.repository() || context.format() != self.cache.object_format { + return Err(NativeCaptureError::Context); + } + if limits.max_pack_bytes > MAX_ARTIFACT_BYTES || limits.max_index_bytes > MAX_ARTIFACT_BYTES + { + return Err(NativeCaptureError::Limit); + } + let _selection = self.cache.selection.lock().await; + self.cache.reconcile().await?; + let cache = Arc::clone(&self.cache); + let format = context.format(); + let claim = cache + .native + .try_admit(crate::native_resources::NativeWork::Read)?; + let pairs = tokio::task::spawn_blocking(move || { + let _claim = claim; + let fence = crate::native_git::lock_file( + &cache.git_dir().join(crate::native_git::WORKER_LOCK), + )?; + fence.try_lock().map_err(std::io::Error::from)?; + let pin = Arc::new(CapturePin { + cache, + _fence: fence, + }); + for entry in std::fs::read_dir(pin.cache.git_dir().join("objects"))? { + let entry = entry?; + if !entry.file_type()?.is_dir() + || !matches!(entry.file_name().to_str(), Some("pack" | "info")) + { + return Err(NativeCaptureError::Context); + } + } + let root = pin.cache.git_dir().join("objects/pack"); + let mut paths = Vec::new(); + for entry in std::fs::read_dir(&root)? { + let entry = entry?; + let name = entry.file_name(); + let name = name.to_str().ok_or(NativeCaptureError::Context)?; + if !name.ends_with(".pack") { + continue; + } + if paths.len() == MAX_CAPTURE_PACKS || !entry.file_type()?.is_file() { + return Err(NativeCaptureError::Limit); + } + paths.push(entry.path()); + } + paths.sort(); + let mut pairs = Vec::with_capacity(paths.len()); + for path in paths { + let index_path = path.with_extension("idx"); + if !std::fs::symlink_metadata(&index_path)?.is_file() { + return Err(NativeCaptureError::Context); + } + if std::fs::metadata(&path)?.len() > limits.max_pack_bytes + || std::fs::metadata(&index_path)?.len() > limits.max_index_bytes + { + return Err(NativeCaptureError::Limit); + } + let native = NativePackDescriptor::inspect_files( + token.repository, + token.artifact_operation, + format, + &path, + &index_path, + )?; + if path.file_name().and_then(|name| name.to_str()) + != Some(format!("pack-{}.pack", hex::encode(native.git_checksum)).as_str()) + { + return Err(NativeCaptureError::Context); + } + pairs.push(Pair { + native, + pack: Arc::new(InputFile { + path, + _pin: Arc::clone(&pin), + }), + index: Arc::new(InputFile { + path: index_path, + _pin: Arc::clone(&pin), + }), + }); + } + Ok::<_, NativeCaptureError>(pairs) + }) + .await??; + context.ensure_live()?; + let mut inputs = Vec::with_capacity(pairs.len()); + for Pair { + mut native, + pack, + index, + } in pairs + { + context.ensure_live()?; + native.pack = upload_file( + pack, + store, + native.key(ArtifactKind::Pack)?, + native.pack.size, + native.pack.digest, + ) + .await?; + context.ensure_live()?; + native.index = upload_file( + index, + store, + native.key(ArtifactKind::Index)?, + native.index.size, + native.index.digest, + ) + .await?; + context.ensure_live()?; + inputs.push(native); + } + context.ensure_live()?; + Ok(inputs) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::future::Future; + #[cfg(unix)] + #[test] + fn completed_capture_releases_fence_despite_unrelated_pre_exec_inheritance() + -> Result<(), Box> { + use std::io::{Read, Write}; + use std::os::{ + fd::AsRawFd, + unix::{net::UnixStream, process::CommandExt}, + }; + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build()?; + let root = tempfile::TempDir::new()?; + let backend = runtime.block_on(GitHttpBackend::initialize( + root.path().into(), + DiskBudget::new(1 << 20), + "refs/heads/main", + crate::ObjectFormat::Sha1, + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ))?; + let fence = + crate::native_git::lock_file(&backend.git_dir().join(crate::native_git::WORKER_LOCK))?; + fence.try_lock().map_err(std::io::Error::from)?; + let captured = Arc::new(InputFile { + path: backend.git_dir().join("config"), + _pin: Arc::new(CapturePin { + _fence: fence, + cache: backend.cache.clone(), + }), + }); + let retained = captured.clone(); + let (mut ready, child_ready) = UnixStream::pair()?; + let (mut release, child_release) = UnixStream::pair()?; + ready.set_read_timeout(Some(Duration::from_secs(5)))?; + let child = std::thread::spawn(move || { + let mut command = std::process::Command::new("true"); + // SAFETY: only async-signal-safe read/write run after fork. Socket + // owners are captured until spawn returns; the parent controls EOF. + unsafe { + command.pre_exec(move || { + let mut byte = 1u8; + if libc::write(child_ready.as_raw_fd(), (&byte as *const u8).cast(), 1) != 1 + || libc::read(child_release.as_raw_fd(), (&mut byte as *mut u8).cast(), 1) + != 1 + { + return Err(std::io::Error::last_os_error()); + } + Ok(()) + }); + } + command.status() + }); + let ready_result = ready.read_exact(&mut [0]); + drop(captured); + // Another input/upload owner still protects the pair. + let retained_blocks = crate::native_git::command(&backend.git_dir()) + .is_err_and(|e| e.kind() == std::io::ErrorKind::WouldBlock); + drop(retained); + // Capture is fully complete. An unrelated child cannot mutate this + // cache; its accidentally inherited descriptor must not retain custody. + let admission = crate::native_git::command(&backend.git_dir()); + // Release the task-owned child before asserting on any failure. + let released = release.write_all(&[1]); + drop(release); + let status = child.join().map_err(|_| "foreign child thread panicked")?; + ready_result?; + released?; + assert!(status?.success()); + assert!(retained_blocks); + assert!( + admission.is_ok(), + "finished capture retained by foreign fork: {:?}", + admission.err() + ); + Ok(()) + } + + #[test] + fn canceled_queued_capture_upload_retains_cache_fence_and_disk() + -> Result<(), Box> { + let runtime = tokio::runtime::Builder::new_multi_thread() + .worker_threads(1) + .max_blocking_threads(1) + .enable_all() + .build()?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(1 << 20); + let backend = runtime.block_on(GitHttpBackend::initialize( + root.path().into(), + budget.clone(), + "refs/heads/main", + crate::ObjectFormat::Sha256, + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ))?; + let path = backend.git_dir().join("config"); + let bytes = std::fs::read(&path)?; + let fence = + crate::native_git::lock_file(&backend.git_dir().join(crate::native_git::WORKER_LOCK))?; + fence.try_lock().map_err(std::io::Error::from)?; + let weak = Arc::downgrade(&backend.cache); + let captured = Arc::new(InputFile { + path, + _pin: Arc::new(CapturePin { + _fence: fence, + cache: Arc::clone(&backend.cache), + }), + }); + let (ready_tx, ready_rx) = std::sync::mpsc::channel(); + let (release_tx, release_rx) = std::sync::mpsc::channel(); + let blocker = runtime.spawn_blocking(move || { + ready_tx.send(()).unwrap(); + release_rx.recv().unwrap(); + }); + ready_rx.recv_timeout(Duration::from_secs(5))?; + let store = ArtifactStore::new(Arc::new(object_store::memory::InMemory::new()), [1; 16]); + let mut upload = Box::pin(upload_file( + captured, + &store, + canopy_object_storage::artifact::ArtifactKey { + operation: [2; 16], + binding_digest: [3; 32], + kind: ArtifactKind::Pack, + }, + bytes.len() as u64, + *blake3::hash(&bytes).as_bytes(), + )); + let pending = { + let _entered = runtime.enter(); + let mut cx = Context::from_waker(std::task::Waker::noop()); + upload.as_mut().poll(&mut cx).is_pending() + }; + drop(upload); + let git_dir = backend.git_dir(); + drop(backend); + let retained = weak.upgrade().is_some(); + let charged = budget.used(); + let locked = crate::native_git::command(&git_dir) + .is_err_and(|error| error.kind() == std::io::ErrorKind::WouldBlock); + // Release before assertions so a failure cannot strand runtime shutdown. + release_tx.send(())?; + runtime.block_on(async { + blocker.await?; + tokio::time::timeout(Duration::from_secs(5), async { + while budget.used() != 0 { + tokio::time::sleep(Duration::from_millis(5)).await; + } + }) + .await?; + Ok::<_, Box>(()) + })?; + assert!(pending && retained && locked); + assert!(charged > 0); + assert!(weak.upgrade().is_none()); + assert!(!git_dir.exists()); + Ok(()) + } +} diff --git a/crates/canopy-server/src/git_http/mod.rs b/crates/canopy-server/src/git_http/mod.rs index e2ed75c..43f2ddc 100644 --- a/crates/canopy-server/src/git_http/mod.rs +++ b/crates/canopy-server/src/git_http/mod.rs @@ -11,18 +11,20 @@ use std::{ }; pub use crate::git_cache::CacheError; +mod capture; use crate::{ git_cache::GitCache, git_input::{GitInput, MAX_FETCH_REQUEST_BYTES}, }; use bytes::Bytes; +pub use capture::NativeCaptureError; use cellule_ltx::DiskBudget; use futures_core::Stream; use tokio_util::task::AbortOnDropHandle; use tokio::{ io::{AsyncRead, AsyncReadExt, BufReader}, - process::{Child, Command}, + process::Command, sync::{mpsc, oneshot}, }; @@ -72,6 +74,7 @@ pub struct GitHttpRequest { } /// CGI response with a streamed or collected body. +#[derive(Clone, Debug, PartialEq, Eq)] pub struct GitHttpResponse> { pub status: u16, pub headers: Vec<(String, String)>, @@ -97,9 +100,10 @@ impl GitHttpBackend { budget: DiskBudget, head: &str, object_format: crate::ObjectFormat, + native: crate::native_resources::NativeScope, ) -> Result { Ok(Self { - cache: GitCache::create(scratch_root, budget, head, object_format).await?, + cache: GitCache::create(scratch_root, budget, head, object_format, native).await?, nonce_seed: None, signers: None, }) @@ -121,6 +125,32 @@ impl GitHttpBackend { /// Runs Git on decoded input and collects a bounded reply for durable push publication. pub async fn run(&self, request: GitHttpRequest) -> Result { let response = self.stream(request, ()).await?; + self.collect(response).await + } + + /// Receives native pack/index inputs for staged catalog preparation. The + /// caller must authenticate and admit the decoded request before invoking + /// this API, then capture and verify inputs before durable publication. + pub async fn run_native_receive( + &self, + request: GitHttpRequest, + ) -> Result { + if request.method != "POST" + || request.path_info != "/repo.git/git-receive-pack" + || !request.query.is_empty() + { + return Err(GitHttpError::InvalidPath); + } + let mut command = self.transport_command()?; + command.args(["-c", "receive.unpackLimit=0"]); + let response = self.stream_command(request, (), command).await?; + self.collect(response).await + } + + async fn collect( + &self, + response: GitHttpResponse, + ) -> Result { let GitHttpResponse { status, headers, @@ -203,6 +233,16 @@ impl GitHttpBackend { &self, request: GitHttpRequest, keep_alive: T, + ) -> Result, GitHttpError> { + self.stream_command(request, keep_alive, self.transport_command()?) + .await + } + + async fn stream_command( + &self, + request: GitHttpRequest, + keep_alive: T, + mut process: Command, ) -> Result, GitHttpError> { if request.gzip { return Err(GitHttpError::EncodedInput); @@ -213,7 +253,6 @@ impl GitHttpBackend { { return Err(GitHttpError::InvalidPath); } - let mut process = self.transport_command()?; process .arg("http-backend") .env("GIT_PROJECT_ROOT", self.cache.root()) @@ -247,6 +286,9 @@ impl GitHttpBackend { process, (keep_alive, Arc::clone(&self.cache), request.body), WORKER_DEADLINE, + self.cache + .native + .try_admit(crate::native_resources::NativeWork::Pack)?, ) .await } @@ -290,8 +332,9 @@ async fn start_stream( command: Command, keep_alive: T, deadline: Duration, + native: crate::native_resources::NativePermit, ) -> Result, GitHttpError> { - let mut process = GitProcess::spawn(command, keep_alive)?; + let mut process = GitProcess::spawn(command, keep_alive, native)?; let stdout = process .child .stdout @@ -339,8 +382,7 @@ async fn start_stream( tokio::try_join!(read_stdout, read_bounded(stderr, MAX_CGI_STDERR_BYTES),)?; // Keep the group leader unreaped while descendants still own pipes; // cancellation can then signal its group without PID reuse ambiguity. - let status = process.child.wait().await?; - process.disarm(); + let status = process.wait().await?; if !status.success() { return Err(GitHttpError::GitExit { status, @@ -384,57 +426,7 @@ async fn start_stream( }) } -pub(crate) struct GitProcess { - pub(crate) child: Child, - #[cfg(unix)] - group: Option, - // Drop signals the process group before fields release cache and input owners. - _keep_alive: T, -} - -impl GitProcess { - pub(crate) fn spawn(mut command: Command, keep_alive: T) -> Result { - #[cfg(unix)] - command.process_group(0); - let child = command.kill_on_drop(true).spawn(); - // The pre-exec closure owns a parent copy of the worker fence. Release - // it before cleanup can drop the cache, including on spawn failure; - // only the running child and its descendants should retain that lock. - drop(command); - let child = child?; - #[cfg(unix)] - let group = Some( - i32::try_from(child.id().ok_or(GitHttpError::Interrupted)?) - .map_err(|_| GitHttpError::Interrupted)?, - ); - Ok(Self { - child, - #[cfg(unix)] - group, - _keep_alive: keep_alive, - }) - } - - pub(crate) fn disarm(&mut self) { - #[cfg(unix)] - { - self.group = None; - } - } -} - -impl Drop for GitProcess { - fn drop(&mut self) { - #[cfg(unix)] - if let Some(group) = self.group { - // SAFETY: spawn created a separate process group with this positive PID. - // Its leader is not reaped until pipes close, preventing PID reuse here. - unsafe { - libc::kill(-group, libc::SIGKILL); - } - } - } -} +pub(crate) use crate::native_git::process::GitProcess; pub(crate) async fn read_bounded( reader: R, diff --git a/crates/canopy-server/src/git_http/stream_tests.rs b/crates/canopy-server/src/git_http/stream_tests.rs index c0619b4..111e395 100644 --- a/crates/canopy-server/src/git_http/stream_tests.rs +++ b/crates/canopy-server/src/git_http/stream_tests.rs @@ -23,7 +23,15 @@ async fn streaming_backpressure_bounds_queued_output_above_old_pack_limit() "printf 'Content-Type: application/octet-stream\r\n\r\n'; dd if=/dev/zero bs=65536 count=1088 2>/dev/null; touch completed", ); command.current_dir(files.path()); - let mut response = start_stream(command, files, Duration::from_secs(30)).await?; + let mut response = start_stream( + command, + files, + Duration::from_secs(30), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground) + .try_admit(crate::native_resources::NativeWork::Read)?, + ) + .await?; tokio::time::timeout(Duration::from_secs(5), async { while response.body.receiver.len() < 4 { tokio::task::yield_now().await; @@ -48,8 +56,7 @@ async fn streaming_backpressure_bounds_queued_output_above_old_pack_limit() async fn exit_failure_after_headers_is_a_body_error() -> Result<(), Box> { let mut response = start_stream( shell("printf 'Content-Type: application/octet-stream\r\n\r\npartial'; echo failed >&2; exit 7"), - (), Duration::from_secs(5), - ).await?; + (), Duration::from_secs(5), crate::native_resources::NativeResources::default().scope(crate::native_resources::NativeClass::Foreground).try_admit(crate::native_resources::NativeWork::Read)?).await?; let mut bytes = Vec::new(); let error = loop { match chunk(&mut response.body).await { @@ -72,6 +79,9 @@ async fn deadline_after_headers_is_a_body_error() -> Result<(), Box Result<(), Box>, + // Release disk credit last: zero spool usage must not become visible + // before this owner's transfer admission has finished releasing. + reservation: DiskReservation, } /// An immutable request spool, deleted when its last file handle closes. @@ -247,12 +254,28 @@ impl GitInput { Ok(Stdio::from(self.spool.file.try_clone()?)) } - pub(crate) async fn digest(&self, mut hash: blake3::Hasher) -> Result<[u8; 32], InputError> { + pub(crate) async fn digest( + &self, + mut hash: blake3::Hasher, + ) -> Result<([u8; 32], [u8; 32]), InputError> { // Preserve the canonical length-prefixed HTTP digest without keeping the // body in memory. Only this pre-execution pass shares the file cursor. hash.update(&self.size.to_le_bytes()); + let (content, scoped) = self.hashes(Some(hash)).await?; + Ok((scoped.expect("scoped digest requested"), content)) + } + pub(crate) async fn content_digest(&self) -> Result<[u8; 32], InputError> { + Ok(self.hashes(None).await?.0) + } + async fn hashes( + &self, + mut hash: Option, + ) -> Result<([u8; 32], Option<[u8; 32]>), InputError> { let spool = Arc::clone(&self.spool); Ok(tokio::task::spawn_blocking(move || { + // The raw artifact digest shares this scan with the scoped request + // digest. Upload still independently verifies the immutable bytes. + let mut content = blake3::Hasher::new(); let mut file = &spool.file; file.rewind()?; let mut buffer = [0; CHUNK_BYTES]; @@ -261,13 +284,125 @@ impl GitInput { if count == 0 { break; } - hash.update(&buffer[..count]); + if let Some(hash) = &mut hash { + hash.update(&buffer[..count]); + } + content.update(&buffer[..count]); } file.rewind()?; - Ok::<_, std::io::Error>(*hash.finalize().as_bytes()) + Ok::<_, std::io::Error>(( + *content.finalize().as_bytes(), + hash.map(|hash| *hash.finalize().as_bytes()), + )) }) .await??) } + /// Consume the spool so cancellation cannot race its queued file cursor. + pub(crate) async fn read_owned( + self, + read: impl FnOnce(&mut File) -> io::Result + Send + 'static, + ) -> Result { + Ok(tokio::task::spawn_blocking(move || { + let mut file = self.spool.file.try_clone()?; + file.rewind()?; + let result = read(&mut file); + drop(file); + drop(self); + result + }) + .await??) + } + + /// Consume the observer while uploading: canceled blocking reads keep the + /// spool/account pin, and no remaining caller can race its shared cursor. + pub(crate) async fn retain( + self, + store: &ArtifactStore, + operation: [u8; 16], + digest: [u8; 32], + ) -> Result<(Self, ArtifactDescriptor), InputError> { + let descriptor = crate::packs::metadata::transport::upload_file( + Arc::new(SpoolPin(Arc::clone(&self.spool))), + store, + ArtifactKey { + operation, + binding_digest: digest, + kind: ArtifactKind::InputBody, + }, + self.size, + digest, + ) + .await?; + (&self.spool.file).rewind()?; + Ok((self, descriptor)) + } + pub(crate) async fn reopen( + reader: ArtifactRead, + root: &Path, + budget: &DiskBudget, + limit: Option, + admission: Option>, + ) -> Result { + Self::receive( + Body::from_stream(ArtifactStream { + reader: Some(reader), + job: None, + }), + root, + budget, + limit, + admission, + ) + .await + } +} + +struct SpoolPin(Arc); +impl crate::packs::metadata::transport::PinnedFile for SpoolPin { + fn open(&self) -> io::Result { + self.0.file.try_clone() + } +} +type ArtifactJob = Pin< + Box< + dyn std::future::Future< + Output = (ArtifactRead, Result, ArtifactError>), + > + Send, + >, +>; +struct ArtifactStream { + reader: Option, + job: Option, +} +impl Stream for ArtifactStream { + type Item = Result; + fn poll_next( + mut self: Pin<&mut Self>, + cx: &mut std::task::Context<'_>, + ) -> std::task::Poll> { + if self.job.is_none() { + let Some(mut reader) = self.reader.take() else { + return std::task::Poll::Ready(None); + }; + self.job = Some(Box::pin(async move { + let result = reader.next().await; + (reader, result) + })); + } + let std::task::Poll::Ready((reader, result)) = self.job.as_mut().unwrap().as_mut().poll(cx) + else { + return std::task::Poll::Pending; + }; + self.job = None; + match result { + Ok(Some(bytes)) => { + self.reader = Some(reader); + std::task::Poll::Ready(Some(Ok(bytes))) + } + Ok(None) => std::task::Poll::Ready(None), + Err(error) => std::task::Poll::Ready(Some(Err(error))), + } + } } struct DecodeReader<'a> { diff --git a/crates/canopy-server/src/git_input/tests.rs b/crates/canopy-server/src/git_input/tests.rs index 62da75c..1e5e396 100644 --- a/crates/canopy-server/src/git_input/tests.rs +++ b/crates/canopy-server/src/git_input/tests.rs @@ -90,7 +90,13 @@ async fn chunked_input_preserves_digest_and_rewinds_for_git() -> Result<()> { let mut expected = prefix.clone(); expected.update(&6u64.to_le_bytes()); expected.update(b"abcdef"); - assert_eq!(input.digest(prefix).await?, *expected.finalize().as_bytes()); + assert_eq!( + input.digest(prefix).await?, + ( + *expected.finalize().as_bytes(), + *blake3::hash(b"abcdef").as_bytes() + ) + ); let mut bytes = Vec::new(); (&input.spool.file).read_to_end(&mut bytes)?; assert_eq!(bytes, b"abcdef"); @@ -223,7 +229,7 @@ async fn gzip_members_preserve_wire_digest_and_release_encoded_admission() -> Re ) .await?; assert_eq!( - input.digest(blake3::Hasher::new()).await?, + input.digest(blake3::Hasher::new()).await?.0, *expected.finalize().as_bytes() ); let decoded = input @@ -431,3 +437,87 @@ async fn progressing_upload_outlives_the_idle_deadline() -> Result<()> { assert_eq!(upload.await??.size(), 3); Ok(()) } + +#[test] +fn cancelling_queued_request_hash_retention_or_read_keeps_spool_and_admission() -> Result<()> { + use std::future::Future; + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .max_blocking_threads(1) + .build()?; + runtime.block_on(async { + for operation in 0..3 { + let directory = tempfile::TempDir::new()?; + let budget = DiskBudget::new(1024); + let transfers = crate::admission::AccountAdmission::new(2, "total", "account"); + let permit = Arc::new(transfers.acquire(crate::ReadIdentity::Anonymous).await?); + let raw = b"anonymous retained request"; + let input = GitInput::receive( + Body::from(raw.as_slice()), + directory.path(), + &budget, + None, + Some(permit), + ) + .await?; + let weak = Arc::downgrade(&input.spool); + let (entered, ready) = tokio::sync::oneshot::channel(); + let (release, held) = std::sync::mpsc::channel(); + let blocker = tokio::task::spawn_blocking(move || { + let _ = entered.send(()); + held.recv() + }); + ready.await?; + let store = + ArtifactStore::new(Arc::new(object_store::memory::InMemory::new()), [1; 16]); + let mut work = Box::pin(async move { + if operation == 0 { + input.digest(blake3::Hasher::new()).await?; + } else if operation == 1 { + input + .retain(&store, [2; 16], *blake3::hash(raw).as_bytes()) + .await?; + } else { + input + .read_owned(|file| { + let mut body = Vec::new(); + std::io::Read::read_to_end(file, &mut body)?; + Ok(body) + }) + .await?; + } + Ok::<_, InputError>(()) + }); + let pending = poll_fn(|cx| Poll::Ready(work.as_mut().poll(cx).is_pending())).await; + drop(work); + let charged = budget.used(); + let retained = weak.upgrade().is_some(); + let available = transfers + .acquire(crate::ReadIdentity::Anonymous) + .await + .is_ok(); + // Release before assertions so a failed assertion cannot strand the + // only blocking thread during runtime shutdown. + release.send(())?; + blocker.await??; + assert!(pending); + assert_eq!(charged, raw.len() as u64); + assert!(retained); + assert!(!available); + tokio::time::timeout(Duration::from_secs(5), async { + while budget.used() != 0 || weak.upgrade().is_some() { + tokio::task::yield_now().await; + } + }) + .await?; + assert!( + transfers + .acquire(crate::ReadIdentity::Anonymous) + .await + .is_ok() + ); + assert_eq!(std::fs::read_dir(directory.path())?.count(), 0); + } + Ok(()) + }) +} diff --git a/crates/canopy-server/src/git_objects/mod.rs b/crates/canopy-server/src/git_objects/mod.rs index 96e0c45..7e95295 100644 --- a/crates/canopy-server/src/git_objects/mod.rs +++ b/crates/canopy-server/src/git_objects/mod.rs @@ -4,7 +4,7 @@ use std::{io, path::Path, process::Stdio, time::Duration}; use tokio::{ io::{AsyncRead, AsyncReadExt, AsyncWriteExt, BufReader}, - process::{Child, ChildStdin, ChildStdout}, + process::{ChildStdin, ChildStdout}, time::timeout, }; use tokio_util::task::AbortOnDropHandle; @@ -34,25 +34,41 @@ pub enum ObjectReadError { } struct Process { - child: Child, + worker: crate::native_git::process::GitProcess<()>, output: BufReader, stderr: AbortOnDropHandle, io::Error>>, } impl Process { - fn start(git_dir: &Path, args: &[&str]) -> Result<(Self, ChildStdin), ObjectReadError> { - let mut child = crate::native_git::command(git_dir)? + fn start( + git_dir: &Path, + args: &[&str], + native: &crate::native_resources::NativeScope, + ) -> Result<(Self, ChildStdin), ObjectReadError> { + let mut command = crate::native_git::command(git_dir)?; + command .arg("--git-dir") .arg(git_dir) .args(args) .stdin(Stdio::piped()) .stdout(Stdio::piped()) - .stderr(Stdio::piped()) - .kill_on_drop(true) - .spawn()?; - let input = child.stdin.take().ok_or(ObjectReadError::Malformed)?; - let output = child.stdout.take().ok_or(ObjectReadError::Malformed)?; - let mut stderr = child.stderr.take().ok_or(ObjectReadError::Malformed)?; + .stderr(Stdio::piped()); + let mut child = crate::native_git::process::GitProcess::spawn( + command, + (), + native.try_admit(crate::native_resources::NativeWork::Read)?, + )?; + let input = child.child.stdin.take().ok_or(ObjectReadError::Malformed)?; + let output = child + .child + .stdout + .take() + .ok_or(ObjectReadError::Malformed)?; + let mut stderr = child + .child + .stderr + .take() + .ok_or(ObjectReadError::Malformed)?; let stderr = AbortOnDropHandle::new(tokio::spawn(async move { let mut retained = Vec::new(); let mut chunk = [0; 8192]; @@ -68,7 +84,7 @@ impl Process { })); Ok(( Self { - child, + worker: child, output: BufReader::new(output), stderr, }, @@ -77,7 +93,7 @@ impl Process { } async fn finish(mut self) -> Result<(), ObjectReadError> { - let status = self.child.wait().await?; + let status = self.worker.wait().await?; let stderr = self.stderr.await??; if !status.success() { return Err(ObjectReadError::Git(format!( @@ -100,8 +116,9 @@ impl GitObjectWalk { git_dir: &Path, included: Vec, filter: Option<&str>, + native: &crate::native_resources::NativeScope, ) -> Result { - Self::start(git_dir, included, Vec::new(), true, filter) + Self::start(git_dir, included, Vec::new(), true, filter, native) } fn start( @@ -110,6 +127,7 @@ impl GitObjectWalk { excluded: Vec, missing_only: bool, filter: Option<&str>, + native: &crate::native_resources::NativeScope, ) -> Result { let filter = filter.map(|value| format!("--filter={value}")); let mut args = vec!["rev-list", "--objects", "--no-object-names", "--stdin"]; @@ -119,7 +137,7 @@ impl GitObjectWalk { if let Some(filter) = &filter { args.push(filter); } - let (process, mut input) = Process::start(git_dir, &args)?; + let (process, mut input) = Process::start(git_dir, &args, native)?; // Ref lists can exceed argv limits; feed stdin concurrently with stdout consumption. let revisions = AbortOnDropHandle::new(tokio::spawn(async move { for (prefix, roots) in [("", included), ("^", excluded)] { @@ -171,6 +189,17 @@ pub(crate) struct GitObjects { inventory: Option>, batch: Process, requests: ChildStdin, + // A canceled/failed streamed inspection leaves a partial native frame. + // Never reuse that process for another object or a successful finish. + inspection_failed: bool, +} + +pub trait EdgeSink: Send { + fn append( + &mut self, + parent: crate::ObjectId, + edges: &[crate::packs::metadata::TypedEdge], + ) -> impl std::future::Future> + Send; } impl GitObjects { @@ -178,31 +207,74 @@ impl GitObjects { git_dir: &Path, included: Vec, excluded: Vec, + native: &crate::native_resources::NativeScope, ) -> Result { - let walk = GitObjectWalk::start(git_dir, included, excluded, false, None)?; - let (batch, requests) = Process::start(git_dir, &["cat-file", "--batch"])?; + let walk = GitObjectWalk::start(git_dir, included, excluded, false, None, native)?; + let (batch, requests) = Process::start(git_dir, &["cat-file", "--batch"], native)?; Ok(Self { walk: Some(walk), inventory: None, batch, requests, + inspection_failed: false, }) } pub(crate) fn packed( git_dir: &Path, ids: Vec, + native: &crate::native_resources::NativeScope, ) -> Result { - let (batch, requests) = Process::start(git_dir, &["cat-file", "--batch"])?; + let mut objects = Self::batch(git_dir, native)?; + objects.inventory = Some(ids.into_iter()); + Ok(objects) + } + + /// Persistent native reader with caller-owned bounded index iteration. + /// Verification uses an isolated admitted object directory without alternates. + pub(crate) fn batch( + git_dir: &Path, + native: &crate::native_resources::NativeScope, + ) -> Result { + let (batch, requests) = Process::start(git_dir, &["cat-file", "--batch"], native)?; Ok(Self { walk: None, - inventory: Some(ids.into_iter()), + inventory: None, batch, requests, + inspection_failed: false, }) } + /// Streams canonical hashing and typed structural extraction. Sink writes + /// are private preparation; discard them if this returns an error or is + /// canceled. Pack binding and graph closure remain verifier obligations. + pub(crate) async fn inspect_graph( + &mut self, + oid: crate::ObjectId, + sink: &mut impl EdgeSink, + ) -> Result { + if self.inspection_failed { + return Err(ObjectReadError::Malformed); + } + self.inspection_failed = true; + let object = timeout(IO_TIMEOUT, async { + self.requests + .write_all(format!("{}\n", hex::encode(oid)).as_bytes()) + .await?; + open_object(&mut self.batch.output, oid).await + }) + .await + .map_err(|_| ObjectReadError::Timeout)??; + let canonical = object.inspect_graph(sink).await?; + self.inspection_failed = false; + Ok(canonical) + } + pub(crate) async fn next(&mut self) -> Result, ObjectReadError> { + if self.inspection_failed { + return Err(ObjectReadError::Malformed); + } if let Some(inventory) = &mut self.inventory { return Ok(inventory.next()); } @@ -217,6 +289,9 @@ impl GitObjects { &mut self, oid: crate::ObjectId, ) -> Result>, ObjectReadError> { + if self.inspection_failed { + return Err(ObjectReadError::Malformed); + } timeout(IO_TIMEOUT, async { self.requests .write_all(format!("{}\n", hex::encode(oid)).as_bytes()) @@ -228,6 +303,9 @@ impl GitObjects { } pub(crate) async fn finish(self) -> Result<(), ObjectReadError> { + if self.inspection_failed { + return Err(ObjectReadError::Malformed); + } timeout(IO_TIMEOUT, async move { if let Some(walk) = self.walk { walk.finish().await?; @@ -271,6 +349,58 @@ pub(crate) struct GitObject<'a, R> { } impl GitObject<'_, R> { + async fn inspect_graph( + mut self, + sink: &mut impl EdgeSink, + ) -> Result { + use crate::{ + graph::stream::{CHUNK_BYTES, EdgeParser}, + packs::metadata::{CanonicalObject, PAGE_OBJECTS}, + }; + let mut canonical = + crate::git_format::ObjectHasher::new(self.oid.format(), self.kind, self.size); + let mut hash = blake3::Hasher::new(); + let mut parser = EdgeParser::new(self.oid.format(), self.kind); + let mut buffer = vec![0; CHUNK_BYTES]; + // A bounded input chunk plus one crossing record bounds occurrences. + // Repository size, wide trees and repeated parents do not grow this Vec. + let max_edges = CHUNK_BYTES / (self.oid.len() + 4) + 1; + let mut edges = Vec::with_capacity(max_edges); + loop { + let count = timeout(IO_TIMEOUT, self.reader.read(&mut buffer)) + .await + .map_err(|_| ObjectReadError::Timeout)??; + if count == 0 { + break; + } + canonical.update(&buffer[..count]); + hash.update(&buffer[..count]); + edges.clear(); + parser + .feed(&buffer[..count], |edge| edges.push(edge)) + .map_err(|_| ObjectReadError::Malformed)?; + if edges.len() > max_edges { + return Err(ObjectReadError::Malformed); + } + for batch in edges.chunks(PAGE_OBJECTS) { + timeout(IO_TIMEOUT, sink.append(self.oid, batch)) + .await + .map_err(|_| ObjectReadError::Timeout)??; + } + } + parser.finish().map_err(|_| ObjectReadError::Malformed)?; + let result = CanonicalObject { + oid: self.oid, + kind: self.kind, + size: self.size, + digest: *hash.finalize().as_bytes(), + }; + self.finish().await?; + if canonical.finalize() != result.oid { + return Err(ObjectReadError::Malformed); + } + Ok(result) + } /// Verify a packed body with constant memory, including oversized blobs. pub(crate) async fn fingerprint(mut self) -> Result<[u8; 32], ObjectReadError> { let expected = self.oid; diff --git a/crates/canopy-server/src/git_objects/tests.rs b/crates/canopy-server/src/git_objects/tests.rs index 689b011..b872638 100644 --- a/crates/canopy-server/src/git_objects/tests.rs +++ b/crates/canopy-server/src/git_objects/tests.rs @@ -42,7 +42,13 @@ async fn collect( included: Vec, excluded: Vec, ) -> TestResult)>> { - let mut objects = GitObjects::start(&path.join(".git"), included, excluded)?; + let mut objects = GitObjects::start( + &path.join(".git"), + included, + excluded, + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )?; let mut result = BTreeMap::new(); while let Some(oid) = objects.next().await? { let object = objects.read(oid).await?.body().await?; @@ -119,6 +125,8 @@ async fn missing_walk_root_cannot_finish_successfully() -> TestResult { &directory.path().join(".git"), vec![crate::ObjectId::Sha1([42; 20])], vec![], + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), )?; assert_eq!(objects.next().await?, None); assert!(matches!( @@ -165,10 +173,24 @@ async fn malformed_and_oversized_batches_fail_before_publication() { async fn dropping_reader_kills_both_children() -> TestResult { let directory = fixture().await?; let tip = oid(directory.path(), "HEAD").await?; - let objects = GitObjects::start(&directory.path().join(".git"), vec![tip], vec![])?; + let objects = GitObjects::start( + &directory.path().join(".git"), + vec![tip], + vec![], + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )?; let pids = [ - objects.walk.as_ref().unwrap().process.child.id().unwrap(), - objects.batch.child.id().unwrap(), + objects + .walk + .as_ref() + .unwrap() + .process + .worker + .child + .id() + .unwrap(), + objects.batch.worker.child.id().unwrap(), ]; drop(objects); timeout(Duration::from_secs(5), async { @@ -194,7 +216,13 @@ async fn batch_reads_large_blob_across_pipe_buffers() -> TestResult { tokio::fs::write(directory.path().join("large"), &body).await?; let output = git(directory.path(), &["hash-object", "-w", "large"]).await?; let oid = parse_oid(output.trim_ascii())?; - let mut objects = GitObjects::start(&directory.path().join(".git"), vec![oid], vec![])?; + let mut objects = GitObjects::start( + &directory.path().join(".git"), + vec![oid], + vec![], + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )?; assert_eq!(objects.next().await?, Some(oid)); let mut object = objects.read(oid).await?; let store = crate::blob::LargeBlobStore::new( @@ -212,6 +240,27 @@ async fn batch_reads_large_blob_across_pipe_buffers() -> TestResult { offset += bytes.len(); } assert_eq!(offset, body.len()); + struct BlobSink; + impl EdgeSink for BlobSink { + async fn append( + &mut self, + _: crate::ObjectId, + _: &[crate::packs::metadata::TypedEdge], + ) -> Result<(), ObjectReadError> { + Err(ObjectReadError::Malformed) + } + } + let mut verifier = crate::packs::verification::CanonicalVerifier::new( + &directory.path().join(".git"), + oid.format(), + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )?; + let canonical = verifier.inspect(oid, &mut BlobSink).await?; + assert_eq!(canonical.size, body.len() as u64); + assert_eq!(canonical.kind, ObjectKind::Blob); + assert_eq!(canonical.digest, *blake3::hash(&body).as_bytes()); + verifier.finish().await?; Ok(()) } @@ -234,7 +283,13 @@ async fn missing_walk_streams_requested_history_without_unrelated_blobs() -> Tes let hex = hex::encode(oid); tokio::fs::remove_file(path.join(".git/objects").join(&hex[..2]).join(&hex[2..])).await?; } - let mut walk = GitObjectWalk::missing(&path.join(".git"), vec![root], None)?; + let mut walk = GitObjectWalk::missing( + &path.join(".git"), + vec![root], + None, + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )?; let mut found = std::collections::BTreeSet::new(); while let Some(oid) = walk.next().await? { assert!(found.insert(oid), "duplicate missing object"); @@ -243,3 +298,47 @@ async fn missing_walk_streams_requested_history_without_unrelated_blobs() -> Tes assert_eq!(found, expected); Ok(()) } + +#[tokio::test] +async fn streamed_inspection_rejects_hash_mismatch_partial_bodies_bad_separators_and_invalid_graphs() +-> TestResult { + struct NoEdges; + impl EdgeSink for NoEdges { + async fn append( + &mut self, + _: crate::ObjectId, + _: &[crate::packs::metadata::TypedEdge], + ) -> Result<(), ObjectReadError> { + Ok(()) + } + } + for format in [crate::ObjectFormat::Sha1, crate::ObjectFormat::Sha256] { + let body = b"canonical bytes"; + let oid = crate::object_id(format, ObjectKind::Blob, body); + for (bytes, size, separator) in [ + (&b"incorrect bytes"[..], body.len(), b'\n'), + (&body[..2], body.len(), b'\n'), + (&body[..], body.len(), b'X'), + (&body[..], 1, b'\n'), + ] { + let mut frame = format!("{} blob {size}\n", hex::encode(oid)).into_bytes(); + frame.extend_from_slice(bytes); + frame.push(separator); + let mut input = frame.as_slice(); + let object = open_object(&mut input, oid).await?; + assert!(object.inspect_graph(&mut NoEdges).await.is_err()); + } + let body = b"100644 unfinished"; + let oid = crate::object_id(format, ObjectKind::Tree, body); + let mut frame = format!("{} tree {}\n", hex::encode(oid), body.len()).into_bytes(); + frame.extend_from_slice(body); + frame.push(b'\n'); + let mut input = frame.as_slice(); + let object = open_object(&mut input, oid).await?; + assert!(matches!( + object.inspect_graph(&mut NoEdges).await, + Err(ObjectReadError::Malformed) + )); + } + Ok(()) +} diff --git a/crates/canopy-server/src/graph/mod.rs b/crates/canopy-server/src/graph/mod.rs index 652f41c..be80433 100644 --- a/crates/canopy-server/src/graph/mod.rs +++ b/crates/canopy-server/src/graph/mod.rs @@ -11,6 +11,7 @@ use cellule_runtime::{ }; mod preparation; +pub(crate) mod stream; type Oid = crate::ObjectId; type Edge = (Oid, Option); @@ -273,7 +274,11 @@ fn object_edges( Ok(Some(edges)) } -fn edges(format: crate::ObjectFormat, kind: ObjectKind, mut body: &[u8]) -> Option> { +pub(crate) fn edges( + format: crate::ObjectFormat, + kind: ObjectKind, + mut body: &[u8], +) -> Option> { let mut edges = match kind { ObjectKind::Blob => Some(Vec::new()), ObjectKind::Tree => tree_edges(format, body), @@ -363,7 +368,7 @@ fn hex_oid(text: &[u8]) -> Option { (!oid.is_zero()).then_some(oid) } -fn parse_kind(kind: &[u8]) -> Option { +pub(crate) fn parse_kind(kind: &[u8]) -> Option { match kind { b"blob" => Some(ObjectKind::Blob), b"tree" => Some(ObjectKind::Tree), diff --git a/crates/canopy-server/src/graph/stream.rs b/crates/canopy-server/src/graph/stream.rs new file mode 100644 index 0000000..f20e75c --- /dev/null +++ b/crates/canopy-server/src/graph/stream.rs @@ -0,0 +1,235 @@ +//! Constant-space structural extraction. Names, messages and signatures are +//! never retained. Native body bytes preserve commit-parent order; this stream +//! emits dependency occurrences for disk-backed typed deduplication. + +use crate::{ObjectFormat, ObjectId, ObjectKind, packs::metadata::TypedEdge}; + +pub(crate) const CHUNK_BYTES: usize = 64 << 10; +#[derive(Debug, thiserror::Error)] +#[error("malformed structural Git object stream")] +pub(crate) struct GraphStreamError; + +#[derive(Clone, Copy)] +enum Role { + Tree, + Parent, + TagObject, + TagType, + TagName, +} +impl Role { + fn prefix(self) -> &'static [u8] { + match self { + Self::Tree => b"tree ", + Self::Parent => b"parent ", + Self::TagObject => b"object ", + Self::TagType => b"type ", + Self::TagName => b"tag ", + } + } +} +enum State { + Ignore, + Prefix(Role, usize), + Field(Role, [u8; 64], usize), + TagName, + TreeMode(u32, bool), + TreeName(Option, u64, bool), + TreeOid(Option, [u8; 32], usize), +} +pub(crate) struct EdgeParser { + format: ObjectFormat, + state: State, + tag_target: Option, + poisoned: bool, +} +impl EdgeParser { + pub(crate) fn new(format: ObjectFormat, kind: ObjectKind) -> Self { + let state = match kind { + ObjectKind::Blob => State::Ignore, + ObjectKind::Tree => State::TreeMode(0, false), + ObjectKind::Commit => State::Prefix(Role::Tree, 0), + ObjectKind::Tag => State::Prefix(Role::TagObject, 0), + }; + Self { + format, + state, + tag_target: None, + poisoned: false, + } + } + fn oid(&self, bytes: &[u8], hexadecimal: bool) -> Result { + let oid = if hexadecimal { + ObjectId::from_hex(bytes) + } else { + ObjectId::try_from(bytes) + } + .map_err(|_| GraphStreamError)?; + if oid.format() != self.format || oid.is_zero() { + return Err(GraphStreamError); + } + Ok(oid) + } + pub(crate) fn feed( + &mut self, + chunk: &[u8], + mut emit: impl FnMut(TypedEdge), + ) -> Result<(), GraphStreamError> { + if self.poisoned || chunk.len() > CHUNK_BYTES { + self.poisoned = true; + return Err(GraphStreamError); + } + self.poisoned = true; + if matches!(self.state, State::Ignore) { + self.poisoned = false; + return Ok(()); + } + for &byte in chunk { + let state = std::mem::replace(&mut self.state, State::Ignore); + self.state = match state { + State::Ignore => State::Ignore, + State::Prefix(role, at) => { + if byte != role.prefix()[at] { + // Git stops recognizing parents at the first other + // header; later signatures/messages have no graph edges. + if matches!(role, Role::Parent) { + State::Ignore + } else { + return Err(GraphStreamError); + } + } else if at + 1 == role.prefix().len() { + if matches!(role, Role::TagName) { + State::TagName + } else { + State::Field(role, [0; 64], 0) + } + } else { + State::Prefix(role, at + 1) + } + } + State::Field(role, mut bytes, length) => { + if byte == b'\n' { + match role { + Role::Tree | Role::Parent => { + let child = self.oid(&bytes[..length], true)?; + emit(TypedEdge { + child, + expected_kind: if matches!(role, Role::Tree) { + ObjectKind::Tree + } else { + ObjectKind::Commit + }, + }); + State::Prefix(Role::Parent, 0) + } + Role::TagObject => { + self.tag_target = Some(self.oid(&bytes[..length], true)?); + State::Prefix(Role::TagType, 0) + } + Role::TagType => { + let expected_kind = + super::parse_kind(&bytes[..length]).ok_or(GraphStreamError)?; + emit(TypedEdge { + child: self.tag_target.ok_or(GraphStreamError)?, + expected_kind, + }); + State::Prefix(Role::TagName, 0) + } + Role::TagName => return Err(GraphStreamError), + } + } else { + let limit = if matches!(role, Role::TagType) { + 6 + } else { + self.format.bytes() * 2 + }; + if length >= limit { + return Err(GraphStreamError); + } + bytes[length] = byte; + State::Field(role, bytes, length + 1) + } + } + State::TagName => { + if byte == b'\n' { + State::Ignore + } else { + State::TagName + } + } + State::TreeMode(mode, digits) => { + if byte == b' ' { + if !digits { + return Err(GraphStreamError); + } + let kind = match mode & 0o170000 { + 0o040000 => Some(ObjectKind::Tree), + 0o100000 | 0o120000 => Some(ObjectKind::Blob), + 0o160000 => None, + _ => return Err(GraphStreamError), + }; + State::TreeName(kind, 0, true) + } else { + if !(b'0'..=b'7').contains(&byte) { + return Err(GraphStreamError); + } + let mode = mode + .checked_mul(8) + .and_then(|m| m.checked_add(u32::from(byte - b'0'))) + .filter(|m| *m <= 0o177777) + .ok_or(GraphStreamError)?; + State::TreeMode(mode, true) + } + } + State::TreeName(kind, length, dots) => { + if byte == 0 { + if length == 0 || (dots && length <= 2) { + return Err(GraphStreamError); + } + State::TreeOid(kind, [0; 32], 0) + } else { + if byte == b'/' { + return Err(GraphStreamError); + } + State::TreeName( + kind, + length.checked_add(1).ok_or(GraphStreamError)?, + dots && byte == b'.', + ) + } + } + State::TreeOid(kind, mut bytes, at) => { + bytes[at] = byte; + if at + 1 == self.format.bytes() { + let child = self.oid(&bytes[..at + 1], false)?; + if let Some(expected_kind) = kind { + emit(TypedEdge { + child, + expected_kind, + }); + } + State::TreeMode(0, false) + } else { + State::TreeOid(kind, bytes, at + 1) + } + } + }; + } + self.poisoned = false; + Ok(()) + } + pub(crate) fn finish(self) -> Result<(), GraphStreamError> { + if self.poisoned + || !matches!( + self.state, + State::Ignore | State::TreeMode(0, false) | State::Prefix(Role::Parent, _) + ) + { + return Err(GraphStreamError); + } + Ok(()) + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/canopy-server/src/graph/stream/tests.rs b/crates/canopy-server/src/graph/stream/tests.rs new file mode 100644 index 0000000..92efdde --- /dev/null +++ b/crates/canopy-server/src/graph/stream/tests.rs @@ -0,0 +1,157 @@ +use super::*; + +fn inspect( + format: ObjectFormat, + kind: ObjectKind, + body: &[u8], + chunk: usize, +) -> Option)>> { + let mut parser = EdgeParser::new(format, kind); + let mut edges = Vec::new(); + for bytes in body.chunks(chunk) { + parser + .feed(bytes, |edge| { + edges.push((edge.child, Some(edge.expected_kind))) + }) + .ok()?; + } + parser.finish().ok()?; + edges.sort_unstable(); + edges.dedup(); + Some(edges) +} +#[test] +fn streamed_graph_matches_existing_semantics_across_every_field_boundary() { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let bytes = vec![1; format.bytes()]; + let oid = hex::encode(&bytes); + let mut tree = Vec::new(); + for (mode, name) in [ + ("100644", &b"a"[..]), + ("100755", &b"b"[..]), + ("040000", &b"tree"[..]), + ("120000", &b"link"[..]), + ("160000", &b"gitlink"[..]), + ("100644", &b"\xff\n"[..]), + ] { + tree.extend_from_slice(mode.as_bytes()); + tree.push(b' '); + tree.extend_from_slice(name); + tree.push(0); + tree.extend_from_slice(&bytes); + } + let commit = format!("tree {oid}\nparent {oid}\nparent {oid}\nauthor Long Person\ngpgsig signature\n parent fake\n\nparent ignored\n").into_bytes(); + let tag = format!( + "object {oid}\ntype commit\ntag {}\n\nmessage", + "x".repeat(1024) + ) + .into_bytes(); + for (kind, body) in [ + (ObjectKind::Blob, b"arbitrary\0bytes\xff".to_vec()), + (ObjectKind::Tree, tree), + (ObjectKind::Commit, commit), + (ObjectKind::Tag, tag), + ] { + let expected = crate::graph::edges(format, kind, &body); + for chunk in [1, 2, 7, 19, 20, 31, 32, 64, CHUNK_BYTES] { + assert_eq!( + inspect(format, kind, &body, chunk), + expected, + "format={format:?} kind={kind:?} chunk={chunk}" + ); + } + } + for suffix in ["", "par", "parent", "author", "\nmessage"] { + let body = format!("tree {oid}\n{suffix}"); + assert_eq!( + inspect(format, ObjectKind::Commit, body.as_bytes(), 1), + crate::graph::edges(format, ObjectKind::Commit, body.as_bytes()) + ); + } + } +} +#[test] +fn malformed_structures_and_failed_parsers_cannot_finish_or_resume() { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let raw = vec![1; format.bytes()]; + let oid = hex::encode(&raw); + let mut cases = Vec::new(); + for (mode, name) in [ + ("", "name"), + ("8", "name"), + ("1000000", "name"), + ("000000", "name"), + ("100644", ""), + ("100644", "."), + ("100644", ".."), + ("100644", "a/b"), + ] { + let mut body = format!("{mode} {name}\0").into_bytes(); + body.extend_from_slice(&raw); + cases.push((ObjectKind::Tree, body)); + } + let mut zero = b"160000 gitlink\0".to_vec(); + zero.extend(vec![0; format.bytes()]); + cases.push((ObjectKind::Tree, zero)); + cases.push((ObjectKind::Tree, b"100644 unfinished".to_vec())); + cases.push(( + ObjectKind::Commit, + format!("tree {oid}\nparent ").into_bytes(), + )); + cases.push((ObjectKind::Commit, format!("tree {oid}").into_bytes())); + cases.push((ObjectKind::Commit, format!("tree {oid}0\n").into_bytes())); + cases.push(( + ObjectKind::Tag, + format!("object {oid}\ntype unknown\ntag x\n").into_bytes(), + )); + cases.push(( + ObjectKind::Tag, + format!("object {oid}\ntype commit\ntag x").into_bytes(), + )); + for (kind, body) in cases { + assert!( + inspect(format, kind, &body, 1).is_none(), + "accepted {kind:?} {body:?}" + ); + assert!(crate::graph::edges(format, kind, &body).is_none()); + } + let mut parser = EdgeParser::new(format, ObjectKind::Tree); + assert!(parser.feed(b"8", |_| {}).is_err()); + assert!(parser.feed(b"100644 a\0", |_| {}).is_err()); + assert!(parser.finish().is_err()); + let mut parser = EdgeParser::new(format, ObjectKind::Blob); + assert!(parser.feed(&vec![0; CHUNK_BYTES + 1], |_| {}).is_err()); + assert!(parser.finish().is_err()); + } +} +#[test] +fn very_long_names_messages_and_tags_keep_constant_parser_state() -> Result<(), GraphStreamError> { + assert!(std::mem::size_of::() <= 256); + let mut tree = EdgeParser::new(ObjectFormat::Sha256, ObjectKind::Tree); + tree.feed(b"100644 ", |_| {})?; + let bytes = vec![b'x'; CHUNK_BYTES]; + for _ in 0..128 { + tree.feed(&bytes, |_| panic!("early edge"))?; + } + let mut edges = 0; + tree.feed(&[0], |_| {})?; + tree.feed(&[1; 32], |_| edges += 1)?; + tree.finish()?; + assert_eq!(edges, 1); + for kind in [ObjectKind::Commit, ObjectKind::Tag] { + let mut parser = EdgeParser::new(ObjectFormat::Sha256, kind); + let oid = hex::encode([1; 32]); + let prefix = if kind == ObjectKind::Commit { + format!("tree {oid}\nauthor ") + } else { + format!("object {oid}\ntype commit\ntag ") + }; + parser.feed(prefix.as_bytes(), |_| {})?; + for _ in 0..128 { + parser.feed(&bytes, |_| panic!("unexpected edge"))?; + } + parser.feed(b"\n\nmessage", |_| {})?; + parser.finish()?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/http/mod.rs b/crates/canopy-server/src/http/mod.rs index 911ba4f..62a40b6 100644 --- a/crates/canopy-server/src/http/mod.rs +++ b/crates/canopy-server/src/http/mod.rs @@ -704,6 +704,17 @@ async fn git_request(State(api): State>, request: Request) fn git_failure(error: GatewayError) -> Response { match error { + GatewayError::Io(error) + | GatewayError::Http(crate::git_http::GitHttpError::Io(error)) + | GatewayError::Cache(crate::git_cache::CacheError::Io(error)) + | GatewayError::Objects(crate::git_objects::ObjectReadError::Io(error)) + if crate::native_resources::is_exhausted(&error) => + { + plain( + StatusCode::SERVICE_UNAVAILABLE, + "Native Git capacity exhausted; retry the request", + ) + } GatewayError::Cell(error) | GatewayError::Push(crate::PushError::Cell(error)) if cell_unavailable(error.as_ref()) => { @@ -853,4 +864,30 @@ mod tests { StatusCode::INTERNAL_SERVER_ERROR ); } + #[test] + fn native_capacity_returns_503_only_for_typed_admission_exhaustion() { + use crate::native_resources::{NativeClass, NativeResources, NativeWork}; + let pool = NativeResources::default(); + let scope = pool.scope(NativeClass::Foreground); + let mut permits = Vec::new(); + while let Ok(permit) = scope.try_admit(NativeWork::Read) { + permits.push(permit); + } + for wrap in [ + GatewayError::Io, + |error| GatewayError::Http(crate::git_http::GitHttpError::Io(error)), + |error| GatewayError::Cache(crate::git_cache::CacheError::Io(error)), + |error| GatewayError::Objects(crate::git_objects::ObjectReadError::Io(error)), + ] { + let exhausted = scope.try_admit(NativeWork::Read).err().unwrap(); + assert_eq!( + git_failure(wrap(exhausted)).status(), + StatusCode::SERVICE_UNAVAILABLE + ); + assert_eq!( + git_failure(wrap(std::io::Error::from(std::io::ErrorKind::WouldBlock))).status(), + StatusCode::INTERNAL_SERVER_ERROR + ); + } + } } diff --git a/crates/canopy-server/src/lfs/mod.rs b/crates/canopy-server/src/lfs/mod.rs index cf1a807..3810e2c 100644 --- a/crates/canopy-server/src/lfs/mod.rs +++ b/crates/canopy-server/src/lfs/mod.rs @@ -45,7 +45,7 @@ pub enum LfsError { #[error("LFS Cell operation failed")] Cell(#[source] Box), #[error("LFS object store failed")] - Store(#[from] object_store::Error), + Store(#[source] object_store::Error), #[error("LFS request body failed")] Body(#[from] axum::Error), #[error("LFS transfer timed out")] @@ -67,6 +67,16 @@ pub enum LfsError { } /// Repository-scoped LFS transfer service. +impl From for LfsError { + fn from(error: object_store::Error) -> Self { + if crate::external::is_corruption(&error) { + Self::Corrupt + } else { + Self::Store(error) + } + } +} + pub struct LfsService { repository: Arc, store: Arc, diff --git a/crates/canopy-server/src/lfs/read.rs b/crates/canopy-server/src/lfs/read.rs index bb218c8..5038563 100644 --- a/crates/canopy-server/src/lfs/read.rs +++ b/crates/canopy-server/src/lfs/read.rs @@ -38,9 +38,13 @@ impl LfsRead { admission: Option>, ) -> Result { let path = lfs_path(repository_id, &expected.sha256); - let manifest = - crate::external::open_lfs(store.as_ref(), &path, expected.size, expected.parts_digest) - .await?; + let manifest = crate::external::open_hashed( + store.as_ref(), + &path, + expected.size, + expected.parts_digest, + ) + .await?; let state = ReadState { store, path, diff --git a/crates/canopy-server/src/lfs/upload.rs b/crates/canopy-server/src/lfs/upload.rs index 6a7117b..020667a 100644 --- a/crates/canopy-server/src/lfs/upload.rs +++ b/crates/canopy-server/src/lfs/upload.rs @@ -23,7 +23,7 @@ pub(super) async fn receive( let result = async { let body = parts(&mut upload, oid, body, declared, admission.clone()).await?; let parts_digest = upload - .publish_lfs(&lfs_path(repository_id, &oid), body.size, &body.digests) + .publish_hashed(&lfs_path(repository_id, &oid), body.size, &body.digests) .await?; let object = LfsObject { sha256: oid, diff --git a/crates/canopy-server/src/lib.rs b/crates/canopy-server/src/lib.rs index ce96fef..619df0f 100644 --- a/crates/canopy-server/src/lib.rs +++ b/crates/canopy-server/src/lib.rs @@ -44,10 +44,12 @@ pub mod blob { } pub mod lfs; mod native_git; +pub mod native_resources; mod object_batch; mod object_chunks; mod object_reads; mod pack_store; +pub mod packs; pub mod pulls; mod push; mod refs; @@ -61,7 +63,7 @@ pub use access::{COLLABORATOR_PAGE_SIZE, Collaborator, ReadIdentity}; pub use default_branch::DefaultBranch; pub use object_batch::ObjectBatch; pub use object_chunks::ObjectStageError; -pub use push::{PushCertificateReceipt, PushError, PushReceipt}; +pub use push::{PushCertificateReceipt, PushError, PushReceipt, VerifiedPushCertificate}; pub use refs::{FinalizePush, PushPlan, RefExpectation, RefPage, RefReadError, RefUpdate}; pub const REPOSITORIES: NamespaceId = NamespaceId::from_bytes([71; 16]); @@ -183,6 +185,9 @@ impl CellModule for RepositoryModule { let mut source = blake3::Hasher::new(); source.update(include_bytes!("lib.rs")); source.update(include_bytes!("../../canopy-git-format/src/lib.rs")); + source.update(include_bytes!( + "../../canopy-git-format/src/pack_index/mod.rs" + )); source.update(include_bytes!("refs.rs")); source.update(include_bytes!("default_branch.rs")); source.update(include_bytes!("graph/mod.rs")); @@ -195,7 +200,67 @@ impl CellModule for RepositoryModule { source.update(include_bytes!("object_reads/mod.rs")); source.update(include_bytes!("pack_store.rs")); source.update(include_bytes!("git_objects/mod.rs")); + source.update(include_bytes!("native_resources.rs")); + source.update(include_bytes!("native_git.rs")); + source.update(include_bytes!("native_git/process.rs")); + source.update(include_bytes!("native_git/process/fence.rs")); source.update(include_bytes!("git_gateway/mod.rs")); + source.update(include_bytes!("git_gateway/preflight.rs")); + source.update(include_bytes!("git_gateway/preflight/retention.rs")); + source.update(include_bytes!("git_gateway/branch_policy.rs")); + source.update(include_bytes!("git_gateway/push.rs")); + source.update(include_bytes!("git_input/mod.rs")); + source.update(include_bytes!("git_http/capture.rs")); + source.update(include_bytes!("packs/wire_request.rs")); + source.update(include_bytes!("packs/input_artifact.rs")); + source.update(include_bytes!("packs/directory/index/mod.rs")); + source.update(include_bytes!("packs/directory/index/record.rs")); + source.update(include_bytes!("packs/directory/index/codec.rs")); + source.update(include_bytes!("packs/directory/index/cursor.rs")); + source.update(include_bytes!("packs/directory/index/update.rs")); + source.update(include_bytes!("packs/directory/index/bulk.rs")); + source.update(include_bytes!("packs/directory/index/rewrite.rs")); + source.update(include_bytes!("packs/sources/codec.rs")); + source.update(include_bytes!("packs/sources/inputs.rs")); + source.update(include_bytes!("packs/ref_state/mod.rs")); + source.update(include_bytes!("packs/ref_state/record.rs")); + source.update(include_bytes!("packs/ref_state/transition.rs")); + source.update(include_bytes!("packs/ref_state/snapshot.rs")); + source.update(include_bytes!("packs/publication/native_result.rs")); + source.update(include_bytes!("packs/publication/native_result/codec.rs")); + source.update(include_bytes!("packs/publication/native_result/plan.rs")); + source.update(include_bytes!("packs/publication/root_completion/mod.rs")); + source.update(include_bytes!("packs/publication/root_completion/codec.rs")); + source.update(include_bytes!( + "packs/publication/root_completion/prepare.rs" + )); + source.update(include_bytes!( + "packs/publication/root_completion/outcome.rs" + )); + source.update(include_bytes!("packs/publication/completion.rs")); + source.update(include_bytes!("packs/publication/ref_proof.rs")); + source.update(include_bytes!("packs/publication/ref_snapshot.rs")); + source.update(include_bytes!("packs/publication/initialization.rs")); + source.update(include_bytes!("packs/publication/ref_policy/mod.rs")); + source.update(include_bytes!("packs/publication/ref_policy/codec.rs")); + source.update(include_bytes!("packs/publication/ref_policy/commands.rs")); + source.update(include_bytes!("packs/publication/ref_policy/prepare.rs")); + source.update(include_bytes!("packs/publication/ref_policy/schema.sql")); + source.update(include_bytes!( + "packs/publication/initialization/publish.rs" + )); + source.update(include_bytes!("packs/publication/mod.rs")); + source.update(include_bytes!("packs/publication/codec.rs")); + source.update(include_bytes!("packs/publication/sql.rs")); + source.update(include_bytes!("packs/publication/schema.sql")); + source.update(include_bytes!("packs/publication/certificate.rs")); + source.update(include_bytes!("packs/publication/publish.rs")); + source.update(include_bytes!("packs/publication/compaction/publish.rs")); + source.update(include_bytes!("packs/metadata/transport.rs")); + source.update(include_bytes!("packs/publication/inputs.rs")); + source.update(include_bytes!( + "../../canopy-object-storage/src/artifact.rs" + )); source.update(include_bytes!( "../../canopy-object-storage/src/blob/mod.rs" )); @@ -293,7 +358,9 @@ pub struct RepositoryCell { sql: SqlCell, application: ApplicationHandle, target: CellTarget, - pack_reader: OnceLock>, + // Gateways own the cache lifetime. Sharing the reader through a weak + // reference must not retain its original disk budget after gateway eviction. + pack_readers: std::sync::Mutex>>, } impl RepositoryCell { @@ -320,7 +387,7 @@ impl RepositoryCell { sql: application.sql::(target.clone())?, application: application.clone(), target, - pack_reader: OnceLock::new(), + pack_readers: std::sync::Mutex::new(Vec::new()), }) } @@ -330,11 +397,15 @@ impl RepositoryCell { identity: cellule_runtime::MutationIdentity, plan: PushPlan, ) -> std::result::Result, cellule_runtime::InvocationError> { - self.prepare_graph(&plan).await?; - self.prepare_branch_proofs(&plan).await?; - self.application - .command::(&self.target, identity, plan) - .await + // Keep each transport-heavy phase in its own allocation. Embedding all + // three futures multiplies stack copies when debug callers poll a push. + Box::pin(self.prepare_graph(&plan)).await?; + Box::pin(self.prepare_branch_proofs(&plan)).await?; + Box::pin( + self.application + .command::(&self.target, identity, plan), + ) + .await } pub async fn object( @@ -405,12 +476,19 @@ impl RepositoryCell { )); } let record = self.pack_record(pack).await?; - let reader = - self.pack_reader - .get() - .ok_or(cellule_runtime::InvocationError::NotStarted( - Error::Command("packed reader unavailable"), - ))?; + let reader = self + .pack_readers + .lock() + .map_err(|_| { + cellule_runtime::InvocationError::NotStarted(Error::Command( + "packed reader registry poisoned", + )) + })? + .iter() + .find_map(std::sync::Weak::upgrade) + .ok_or(cellule_runtime::InvocationError::NotStarted( + Error::Command("packed reader unavailable"), + ))?; let body = reader .read_blob( record, diff --git a/crates/canopy-server/src/main.rs b/crates/canopy-server/src/main.rs index 1d20c90..ecb4ca8 100644 --- a/crates/canopy-server/src/main.rs +++ b/crates/canopy-server/src/main.rs @@ -33,6 +33,7 @@ struct FileConfig { ssh: Option, data_dir: PathBuf, local_disk_limit_bytes: u64, + native_limits: canopy_server::native_resources::NativeLimits, max_active_repositories: usize, } @@ -195,6 +196,7 @@ async fn run() -> Result<(), StartupError> { data_dir: file.data_dir, store_prefix: provider.prefix().clone(), local_disk_limit_bytes: file.local_disk_limit_bytes, + native_limits: file.native_limits, max_active_repositories: file.max_active_repositories, }; let server = CanopyServer::start(config, provider.store_arc()).await?; @@ -391,4 +393,22 @@ mod tests { assert!(output.contains("logging workload")); Ok(()) } + #[test] + fn native_configuration_is_required_and_rejects_unknown_profile_fields() + -> Result<(), Box> { + let example: serde_json::Value = + serde_json::from_str(include_str!("../../../config.example.json"))?; + let config: FileConfig = serde_json::from_value(example.clone())?; + canopy_server::native_resources::NativeResources::new(config.native_limits)?; + let mut old = example.clone(); + old.as_object_mut().unwrap().remove("native_limits"); + assert!(serde_json::from_value::(old).is_err()); + let mut unknown = example.clone(); + unknown["native_limits"]["read"]["unexpected"] = serde_json::json!(1); + assert!(serde_json::from_value::(unknown).is_err()); + let mut negative = example; + negative["native_limits"]["total"]["memory_bytes"] = serde_json::json!(-1); + assert!(serde_json::from_value::(negative).is_err()); + Ok(()) + } } diff --git a/crates/canopy-server/src/native_git.rs b/crates/canopy-server/src/native_git.rs index 00ffd9e..616aaf9 100644 --- a/crates/canopy-server/src/native_git.rs +++ b/crates/canopy-server/src/native_git.rs @@ -4,6 +4,8 @@ use std::{fs::File, io, path::Path}; use tokio::process::Command; +pub(crate) mod process; + pub(crate) const WORKER_LOCK: &str = ".canopy-native.lock"; pub(crate) fn lock_file(path: &Path) -> io::Result { @@ -155,6 +157,8 @@ mod tests { cellule_ltx::DiskBudget::new(1 << 20), "refs/heads/main", crate::ObjectFormat::Sha1, + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), ) .await?; let git_dir = cache.git_dir(); diff --git a/crates/canopy-server/src/native_git/process.rs b/crates/canopy-server/src/native_git/process.rs new file mode 100644 index 0000000..945cc58 --- /dev/null +++ b/crates/canopy-server/src/native_git/process.rs @@ -0,0 +1,216 @@ +//! Native process ownership survives cancellation until leader reaping and +//! inherited completion descriptors establish that descendant work drained. +use std::{ + io, + process::ExitStatus, + sync::{Arc, OnceLock}, +}; +use tokio::{ + process::{Child, ChildStderr, ChildStdin, ChildStdout, Command}, + sync::{OwnedSemaphorePermit, Semaphore}, +}; + +#[cfg(unix)] +mod fence; + +/// Expose standard streams, but keep wait/try_wait private to the guard. +/// Reaping the leader early would invalidate its safe group-signal identity. +pub(crate) struct ChildSlot { + native: Option, + pub(crate) stdin: Option, + pub(crate) stdout: Option, + pub(crate) stderr: Option, +} +impl ChildSlot { + fn new(mut child: Child) -> Self { + Self { + stdin: child.stdin.take(), + stdout: child.stdout.take(), + stderr: child.stderr.take(), + native: Some(child), + } + } + pub(crate) fn id(&self) -> Option { + self.native.as_ref().and_then(Child::id) + } +} + +pub(crate) struct GitProcess { + pub(crate) child: ChildSlot, + #[cfg(unix)] + group: Option, + #[cfg(unix)] + fence: Option, + owner: Option>, +} +struct ProcessOwner { + _owner: T, + _native: crate::native_resources::NativePermit, +} +impl GitProcess { + pub(crate) fn spawn( + mut command: Command, + owner: T, + native: crate::native_resources::NativePermit, + ) -> io::Result { + let owner = ProcessOwner { + _owner: owner, + _native: native, + }; + #[cfg(unix)] + let fence = { + command.process_group(0); + match fence::CompletionFence::install(&mut command) { + Ok(fence) => fence, + Err(error) => { + drop(command); + return Err(error); + } + } + }; + let child = command.kill_on_drop(true).spawn(); + // Release parent copies of both inherited descriptors before failed + // spawn can drop owners. Only the child/descendants retain those ends. + drop(command); + let child = child?; + #[cfg(unix)] + let group = Some( + i32::try_from( + child + .id() + .ok_or_else(|| io::Error::other("native child has no PID"))?, + ) + .map_err(io::Error::other)?, + ); + Ok(Self { + child: ChildSlot::new(child), + #[cfg(unix)] + group, + #[cfg(unix)] + fence: Some(fence), + owner: Some(owner), + }) + } + + /// Do not reap the group leader while a descendant retains completion + /// ownership. Its unreaped PID makes cancellation's group signal safe. + pub(crate) async fn wait(&mut self) -> io::Result { + #[cfg(unix)] + self.fence + .as_mut() + .ok_or_else(|| io::Error::other("native fence is absent"))? + .drain() + .await?; + let status = self + .child + .native + .as_mut() + .ok_or_else(|| io::Error::other("native child is absent"))? + .wait() + .await?; + #[cfg(unix)] + { + self.group = None; + } + Ok(status) + } +} +impl Drop for GitProcess { + fn drop(&mut self) { + #[cfg(unix)] + if let Some(group) = self.group + && self.child.id() == u32::try_from(group).ok() + { + // SAFETY: spawn created this private process group; its leader + // remains unreaped. Never signal after a caller reaped it. + unsafe { + libc::kill(-group, libc::SIGKILL); + } + } + let Some(child) = self.child.native.take() else { + return; + }; + let cleanup = Cleanup { + child: Some(child), + #[cfg(unix)] + fence: self.fence.take(), + owner: self.owner.take(), + }; + cleanup.defer(); + } +} + +struct Cleanup { + child: Option, + #[cfg(unix)] + fence: Option, + owner: Option, +} +impl Cleanup { + fn defer(mut self) { + // Completed wait needs no task or queue credit. Otherwise signaling is + // not evidence of drain; keep the owner in a bounded supervised reaper. + if self + .child + .as_ref() + .is_some_and(|child| child.id().is_none()) + { + #[cfg(unix)] + let drained = self + .fence + .as_mut() + .is_some_and(|fence| matches!(fence.try_drained(), Ok(true))); + #[cfg(not(unix))] + let drained = true; + if drained { + self.owner.take(); + return; + } + } + static SLOTS: OnceLock> = OnceLock::new(); + let slots = SLOTS.get_or_init(|| Arc::new(Semaphore::new(512))); + let (Ok(runtime), Ok(slot)) = ( + tokio::runtime::Handle::try_current(), + Arc::clone(slots).try_acquire_owned(), + ) else { + tracing::error!("native reaper unavailable; ownership quarantined until restart"); + return; + }; + runtime.spawn(self.reap(slot)); + } + async fn reap(mut self, _slot: OwnedSemaphorePermit) { + let Some(child) = self.child.as_mut() else { + return; + }; + #[cfg(unix)] + let result = if let Some(fence) = self.fence.as_mut() { + // The original group signal preceded this task. Reap the leader and + // drain descendants independently without issuing later PID signals. + tokio::try_join!(child.wait(), fence.drain()).map(|_| ()) + } else { + Err(io::Error::other("native cleanup fence is absent")) + }; + #[cfg(not(unix))] + let result = child.wait().await.map(|_| ()); + match result { + Ok(()) => { + self.owner.take(); + } + Err(error) => { + tracing::error!(%error, "native drain failed; ownership quarantined until restart"); + } + } + } +} +impl Drop for Cleanup { + fn drop(&mut self) { + if let Some(owner) = self.owner.take() { + // Shutdown, saturation or uncertain drain cannot release account/ + // process/cache admission while an inherited worker may still run. + std::mem::forget(owner); + } + } +} + +#[cfg(all(test, unix))] +mod tests; diff --git a/crates/canopy-server/src/native_git/process/fence.rs b/crates/canopy-server/src/native_git/process/fence.rs new file mode 100644 index 0000000..3094925 --- /dev/null +++ b/crates/canopy-server/src/native_git/process/fence.rs @@ -0,0 +1,53 @@ +use std::{io, os::unix::net::UnixStream as StdStream}; +use tokio::{net::UnixStream, process::Command}; + +pub(super) struct CompletionFence(UnixStream); +impl CompletionFence { + pub(super) fn install(command: &mut Command) -> io::Result { + use std::os::fd::{AsRawFd, FromRawFd}; + let (reader, inherited) = StdStream::pair()?; + // Standard streams may be closed in a daemon. Keep the inherited end + // outside 0/1/2, which spawning replaces with the child's standard I/O. + // SAFETY: duplicate this live socket into a new owned descriptor. + let fd = unsafe { libc::fcntl(inherited.as_raw_fd(), libc::F_DUPFD_CLOEXEC, 3) }; + if fd == -1 { + return Err(io::Error::last_os_error()); + } + // SAFETY: successful duplication returned an unowned socket descriptor. + let inherited = unsafe { StdStream::from_raw_fd(fd) }; + reader.set_nonblocking(true)?; + let reader = UnixStream::from_std(reader)?; + // SAFETY: only async-signal-safe fcntl runs after fork. The command + // retains its end until spawn; exec/fork descendants inherit that end. + unsafe { + command.pre_exec(move || { + let fd = inherited.as_raw_fd(); + let flags = libc::fcntl(fd, libc::F_GETFD); + if flags == -1 || libc::fcntl(fd, libc::F_SETFD, flags & !libc::FD_CLOEXEC) == -1 { + return Err(io::Error::last_os_error()); + } + Ok(()) + }); + } + Ok(Self(reader)) + } + pub(super) fn try_drained(&mut self) -> io::Result { + match self.0.try_read(&mut [0]) { + Ok(0) => Ok(true), + Ok(_) => Err(io::Error::new( + io::ErrorKind::InvalidData, + "native completion descriptor carried data", + )), + Err(error) if error.kind() == io::ErrorKind::WouldBlock => Ok(false), + Err(error) => Err(error), + } + } + pub(super) async fn drain(&mut self) -> io::Result<()> { + loop { + if self.try_drained()? { + return Ok(()); + } + self.0.readable().await?; + } + } +} diff --git a/crates/canopy-server/src/native_git/process/tests.rs b/crates/canopy-server/src/native_git/process/tests.rs new file mode 100644 index 0000000..16e271b --- /dev/null +++ b/crates/canopy-server/src/native_git/process/tests.rs @@ -0,0 +1,242 @@ +use super::*; +use std::{future::Future, process::Stdio, time::Duration}; +use tokio::{io::AsyncReadExt, sync::oneshot}; +type Result = std::result::Result>; + +fn shell(root: &std::path::Path, script: &str) -> Command { + let mut command = Command::new("sh"); + command + .current_dir(root) + .args(["-c", script]) + .stdin(Stdio::null()) + .stdout(Stdio::piped()) + .stderr(Stdio::null()); + command +} +struct Owner(Option>); +impl Drop for Owner { + fn drop(&mut self) { + if let Some(done) = self.0.take() { + let _ = done.send(()); + } + } +} +async fn ready(path: &std::path::Path) -> Result { + tokio::time::timeout(Duration::from_secs(5), async { + while !path.exists() { + tokio::time::sleep(Duration::from_millis(5)).await; + } + }) + .await?; + Ok(()) +} + +#[tokio::test] +async fn closed_stdio_does_not_complete_wait_before_descendant_drain() -> Result { + let root = tempfile::TempDir::new()?; + let permits = Arc::new(Semaphore::new(1)); + let permit = Arc::clone(&permits).try_acquire_owned()?; + let (done, wait_done) = oneshot::channel(); + // The helper closes all standard streams. The leader exits immediately; + // only the inherited completion descriptor exposes the remaining work. + let command = shell( + root.path(), + "(touch ready; while [ ! -f release ]; do sleep 0.01; done) /dev/null 2>&1 & printf '%s\n' \"$!\"; exit 0", + ); + let resources = crate::native_resources::NativeResources::default(); + let mut process = GitProcess::spawn( + command, + (permit, Owner(Some(done))), + resources + .scope(crate::native_resources::NativeClass::Foreground) + .try_admit(crate::native_resources::NativeWork::Read)?, + )?; + let mut output = process.child.stdout.take().ok_or("stdout")?; + let mut pid = Vec::new(); + tokio::time::timeout(Duration::from_secs(5), output.read_to_end(&mut pid)).await??; + assert!(!pid.is_empty()); + ready(&root.path().join("ready")).await?; + { + let mut waiting = std::pin::pin!(process.wait()); + std::future::poll_fn(|cx| { + assert!(waiting.as_mut().poll(cx).is_pending()); + std::task::Poll::Ready(()) + }) + .await; + } + assert_eq!( + resources.usage()?.foreground, + crate::native_resources::NativeWork::Read.claim() + ); + assert_eq!(permits.available_permits(), 0); + let mut drain = std::pin::pin!(resources.drain()); + std::future::poll_fn(|cx| { + assert!(drain.as_mut().poll(cx).is_pending()); + std::task::Poll::Ready(()) + }) + .await; + std::fs::write(root.path().join("release"), b"drain")?; + assert!( + tokio::time::timeout(Duration::from_secs(5), process.wait()) + .await?? + .success() + ); + assert_eq!(permits.available_permits(), 0); + drop(process); + tokio::time::timeout(Duration::from_secs(5), wait_done).await??; + assert_eq!(permits.available_permits(), 1); + tokio::time::timeout(Duration::from_secs(5), drain).await?; + assert_eq!( + resources.usage()?, + crate::native_resources::NativeUsage::default() + ); + Ok(()) +} + +#[tokio::test] +async fn canceled_owner_remains_charged_when_a_descendant_escapes_the_group() -> Result { + const HELPER_ROOT: &str = "CANOPY_TEST_NATIVE_ESCAPED_ROOT"; + if let Some(root) = std::env::var_os(HELPER_ROOT) { + let root = std::path::PathBuf::from(root); + // SAFETY: this branch runs only in the separately spawned fixture + // process. Its background PID is not the original group leader. + assert!(unsafe { libc::setsid() } > 0); + std::fs::write(root.join("ready"), b"escaped")?; + let deadline = std::time::Instant::now() + Duration::from_secs(5); + while !root.join("release").exists() && std::time::Instant::now() < deadline { + tokio::time::sleep(Duration::from_millis(10)).await; + } + return Ok(()); + } + let root = tempfile::TempDir::new()?; + let permits = Arc::new(Semaphore::new(1)); + let permit = Arc::clone(&permits).try_acquire_owned()?; + let (done, mut wait_done) = oneshot::channel(); + // An adversarial helper creates a separate session and closes standard + // streams. Group SIGKILL cannot stop it; its inherited completion end must + // keep ownership charged until it exits on release or its fixture deadline. + let binary = std::env::current_exe()?; + let binary = binary + .to_str() + .ok_or("test binary path")? + .replace('\'', "'\\''"); + let script = format!( + "'{binary}' --exact native_git::process::tests::canceled_owner_remains_charged_when_a_descendant_escapes_the_group /dev/null 2>&1 & printf '%s\\n' \"$!\"; exit 0" + ); + let mut command = shell(root.path(), &script); + command.env(HELPER_ROOT, root.path()); + let resources = crate::native_resources::NativeResources::default(); + let mut process = GitProcess::spawn( + command, + (permit, Owner(Some(done))), + resources + .scope(crate::native_resources::NativeClass::Foreground) + .try_admit(crate::native_resources::NativeWork::Read)?, + )?; + let mut output = process.child.stdout.take().ok_or("stdout")?; + let mut pid = Vec::new(); + tokio::time::timeout(Duration::from_secs(5), output.read_to_end(&mut pid)).await??; + let pid: i32 = std::str::from_utf8(&pid)?.trim().parse()?; + ready(&root.path().join("ready")).await?; + // SAFETY: getpgid is a read-only query for this fixture's announced child. + assert_eq!(unsafe { libc::getpgid(pid) }, pid); + drop(process); + assert!( + tokio::time::timeout(Duration::from_millis(50), &mut wait_done) + .await + .is_err() + ); + assert_eq!( + resources.usage()?.foreground, + crate::native_resources::NativeWork::Read.claim() + ); + assert_eq!(permits.available_permits(), 0); + let mut drain = std::pin::pin!(resources.drain()); + std::future::poll_fn(|cx| { + assert!(drain.as_mut().poll(cx).is_pending()); + std::task::Poll::Ready(()) + }) + .await; + std::fs::write(root.path().join("release"), b"drain")?; + tokio::time::timeout(Duration::from_secs(5), wait_done).await??; + assert_eq!(permits.available_permits(), 1); + tokio::time::timeout(Duration::from_secs(5), drain).await?; + assert_eq!( + resources.usage()?, + crate::native_resources::NativeUsage::default() + ); + Ok(()) +} + +#[tokio::test] +async fn closed_daemon_standard_descriptors_cannot_replace_the_completion_end() -> Result { + const CHILD_ROOT: &str = "CANOPY_TEST_NATIVE_CLOSED_STANDARD_ROOT"; + if let Some(root) = std::env::var_os(CHILD_ROOT) { + use std::os::fd::{AsRawFd, FromRawFd}; + struct Restore { + input: std::fs::File, + output: std::fs::File, + } + impl Drop for Restore { + fn drop(&mut self) { + // SAFETY: these are the duplicated original descriptors in + // this isolated test process; restore its stdin/stdout only. + unsafe { + libc::dup2(self.input.as_raw_fd(), 0); + libc::dup2(self.output.as_raw_fd(), 1); + } + } + } + fn duplicate(fd: i32) -> std::io::Result { + // SAFETY: duplicate a live standard descriptor into a fresh slot. + let copy = unsafe { libc::fcntl(fd, libc::F_DUPFD_CLOEXEC, 3) }; + if copy == -1 { + return Err(std::io::Error::last_os_error()); + } + // SAFETY: the successful duplication returned an unowned file. + Ok(unsafe { std::fs::File::from_raw_fd(copy) }) + } + let restore = Restore { + input: duplicate(0)?, + output: duplicate(1)?, + }; + // SAFETY: only this separately spawned fixture closes its own standard + // descriptors. The parent and concurrent tests retain their originals. + unsafe { + libc::close(0); + libc::close(1); + } + let root = std::path::PathBuf::from(root); + let result = async { + let command = shell(&root, "(touch ready; n=0; while [ ! -f release ] && [ \"$n\" -lt 500 ]; do n=$((n+1)); sleep 0.01; done) /dev/null 2>&1 & exit 0"); + let resources = crate::native_resources::NativeResources::default(); + let mut process = GitProcess::spawn(command, (), resources.scope(crate::native_resources::NativeClass::Foreground).try_admit(crate::native_resources::NativeWork::Read)?)?; + let mut output = process.child.stdout.take().ok_or("stdout")?; + let mut bytes = Vec::new(); + tokio::time::timeout(Duration::from_secs(5), output.read_to_end(&mut bytes)).await??; + ready(&root.join("ready")).await?; + { + let mut pending = std::pin::pin!(process.wait()); + std::future::poll_fn(|cx| { assert!(pending.as_mut().poll(cx).is_pending()); std::task::Poll::Ready(()) }).await; + } + std::fs::write(root.join("release"), b"drain")?; + assert!(tokio::time::timeout(Duration::from_secs(5), process.wait()).await??.success()); + drop(output); + drop(process); + Ok::<_, Box>(()) + }.await; + drop(restore); + return result; + } + let root = tempfile::TempDir::new()?; + let output = Command::new(std::env::current_exe()?) + .args(["--exact", "native_git::process::tests::closed_daemon_standard_descriptors_cannot_replace_the_completion_end"]) + .env(CHILD_ROOT, root.path()).stdin(Stdio::null()).kill_on_drop(true) + .output().await?; + assert!( + output.status.success(), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + Ok(()) +} diff --git a/crates/canopy-server/src/native_resources.rs b/crates/canopy-server/src/native_resources.rs new file mode 100644 index 0000000..e8d10ed --- /dev/null +++ b/crates/canopy-server/src/native_resources.rs @@ -0,0 +1,311 @@ +//! Node-wide claims for native Git work. Claims are admission, not OS limits. +use serde::Deserialize; +use std::{ + io, + sync::{Arc, Mutex}, +}; +use tokio::sync::Notify; + +/// Capacity charged atomically as one vector; no partially held reservations. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct NativeCapacity { + pub processes: u32, + pub cpu_units: u32, + pub memory_bytes: u64, + pub descriptors: u32, +} +impl NativeCapacity { + fn add(self, other: Self) -> Option { + Some(Self { + processes: self.processes.checked_add(other.processes)?, + cpu_units: self.cpu_units.checked_add(other.cpu_units)?, + memory_bytes: self.memory_bytes.checked_add(other.memory_bytes)?, + descriptors: self.descriptors.checked_add(other.descriptors)?, + }) + } + fn sub(self, other: Self) -> Option { + Some(Self { + processes: self.processes.checked_sub(other.processes)?, + cpu_units: self.cpu_units.checked_sub(other.cpu_units)?, + memory_bytes: self.memory_bytes.checked_sub(other.memory_bytes)?, + descriptors: self.descriptors.checked_sub(other.descriptors)?, + }) + } + fn fits(self, limit: Self) -> bool { + self.processes <= limit.processes + && self.cpu_units <= limit.cpu_units + && self.memory_bytes <= limit.memory_bytes + && self.descriptors <= limit.descriptors + } +} + +/// Maintenance has a disjoint reserved share. Foreground cannot consume it. +#[derive(Clone, Copy, Debug, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct NativeLimits { + pub total: NativeCapacity, + pub maintenance_reserved: NativeCapacity, + pub read: NativeCapacity, + pub pack: NativeCapacity, +} +impl Default for NativeLimits { + fn default() -> Self { + Self { + total: NativeCapacity { + processes: 32, + cpu_units: 48, + memory_bytes: 16 << 30, + descriptors: 2048, + }, + maintenance_reserved: NativeCapacity { + processes: 2, + cpu_units: 8, + memory_bytes: 2 << 30, + descriptors: 128, + }, + read: NativeWork::Read.claim(), + pack: NativeWork::Pack.claim(), + } + } +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum NativeClass { + Foreground, + Maintenance, +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum NativeWork { + Read, + Pack, +} +impl NativeWork { + /// Default admission estimates for the sanitized two-thread Git command policy. + /// Parent/helpers, heaps and mappings require empirical/OS qualification. + pub fn claim(self) -> NativeCapacity { + match self { + Self::Read => NativeCapacity { + processes: 1, + cpu_units: 1, + memory_bytes: 256 << 20, + descriptors: 32, + }, + Self::Pack => NativeCapacity { + processes: 1, + cpu_units: 4, + memory_bytes: 1 << 30, + descriptors: 64, + }, + } + } +} +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub struct NativeUsage { + pub foreground: NativeCapacity, + pub maintenance: NativeCapacity, +} +struct Pool { + limits: NativeLimits, + foreground: NativeCapacity, + state: Mutex, + changed: Notify, +} +#[derive(Debug, Default)] +struct PoolState { + used: NativeUsage, + closed: bool, + faulted: bool, +} + +/// Create once per node and clone into all gateways and preparation services. +#[derive(Clone)] +pub struct NativeResources(Arc); +impl NativeResources { + pub fn new(limits: NativeLimits) -> io::Result { + let read_minimum = NativeCapacity { + processes: 1, + cpu_units: 1, + memory_bytes: 128 << 20, + descriptors: 32, + }; + let pack_minimum = NativeCapacity { + processes: 1, + cpu_units: 4, + memory_bytes: 512 << 20, + descriptors: 64, + }; + if limits.read.processes != 1 + || limits.pack.processes != 1 + || !read_minimum.fits(limits.read) + || !pack_minimum.fits(limits.pack) + { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + "native profiles are below their minimum claims", + )); + } + let minimum = limits.read.add(limits.pack).ok_or_else(|| { + io::Error::new(io::ErrorKind::InvalidInput, "native profile claim overflow") + })?; + let foreground = limits + .total + .sub(limits.maintenance_reserved) + .filter(|foreground| { + minimum.fits(*foreground) && minimum.fits(limits.maintenance_reserved) + }) + .ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidInput, + "native shares must each fit a read/pack pipeline", + ) + })?; + Ok(Self(Arc::new(Pool { + limits, + foreground, + state: Mutex::new(PoolState::default()), + changed: Notify::new(), + }))) + } + pub fn scope(&self, class: NativeClass) -> NativeScope { + NativeScope { + resources: self.clone(), + class, + } + } + pub fn usage(&self) -> io::Result { + self.0 + .state + .lock() + .map(|state| state.used) + .map_err(|_| io::Error::other("native admission poisoned")) + } + + /// Permanently reject launches through every cloned scope. Serialized with + /// admission: an already admitted claim remains owned until actual drain. + pub fn close(&self) { + match self.0.state.lock() { + Ok(mut state) => state.closed = true, + Err(poisoned) => { + poisoned.into_inner().closed = true; + tracing::error!("native admission poisoned; shutdown ownership retained"); + } + } + self.0.changed.notify_waiters(); + } + + /// Close admission and wait for all foreground and maintenance owners. + /// Cancellation only abandons this observer, never claims or closure. + /// Poison, underflow or quarantined owners cannot prove drain; callers + /// must retain workspace and lease ownership while this remains pending. + pub async fn drain(&self) { + self.close(); + loop { + let changed = self.0.changed.notified(); + tokio::pin!(changed); + // Register before checking state so a final release cannot be lost, + // including when several independent observers wait for drain. + changed.as_mut().enable(); + if self.0.state.lock().is_ok_and(|state| { + state.closed && !state.faulted && state.used == NativeUsage::default() + }) { + return; + } + changed.await; + } + } +} +impl Default for NativeResources { + fn default() -> Self { + Self::new(NativeLimits::default()).expect("valid native defaults") + } +} + +/// Scope retains the node pool, not an active-work reservation. +#[derive(Clone)] +pub struct NativeScope { + resources: NativeResources, + class: NativeClass, +} +impl NativeScope { + pub fn for_class(&self, class: NativeClass) -> Self { + self.resources.scope(class) + } + pub fn try_admit(&self, work: NativeWork) -> io::Result { + let claim = match work { + NativeWork::Read => self.resources.0.limits.read, + NativeWork::Pack => self.resources.0.limits.pack, + }; + let mut state = self + .resources + .0 + .state + .lock() + .map_err(|_| io::Error::other("native admission poisoned"))?; + if state.closed { + return Err(io::Error::new(io::ErrorKind::WouldBlock, NativeClosed)); + } + if state.faulted { + return Err(io::Error::other("native admission faulted")); + } + let (current, limit) = match self.class { + NativeClass::Foreground => (&mut state.used.foreground, self.resources.0.foreground), + NativeClass::Maintenance => ( + &mut state.used.maintenance, + self.resources.0.limits.maintenance_reserved, + ), + }; + let next = current + .add(claim) + .filter(|next| next.fits(limit)) + .ok_or_else(|| io::Error::new(io::ErrorKind::WouldBlock, NativeExhausted))?; + *current = next; + Ok(NativePermit { + scope: self.clone(), + claim, + }) + } +} + +/// Unforgeable single-owner claim. Native process drain owns its release. +pub struct NativePermit { + scope: NativeScope, + claim: NativeCapacity, +} +impl Drop for NativePermit { + fn drop(&mut self) { + let Ok(mut state) = self.scope.resources.0.state.lock() else { + tracing::error!("native admission poisoned; claim quarantined"); + return; + }; + let current = match self.scope.class { + NativeClass::Foreground => &mut state.used.foreground, + NativeClass::Maintenance => &mut state.used.maintenance, + }; + if let Some(next) = current.sub(self.claim) { + *current = next; + } else { + state.faulted = true; + tracing::error!("native admission underflow; claim quarantined"); + } + drop(state); + self.scope.resources.0.changed.notify_waiters(); + } +} + +#[cfg(test)] +mod tests; + +#[derive(Debug, thiserror::Error)] +#[error("native resource admission exhausted")] +struct NativeExhausted; + +#[derive(Debug, thiserror::Error)] +#[error("native resource admission closed")] +struct NativeClosed; + +pub(crate) fn is_exhausted(error: &io::Error) -> bool { + error + .get_ref() + .is_some_and(|source| source.is::() || source.is::()) +} diff --git a/crates/canopy-server/src/native_resources/tests.rs b/crates/canopy-server/src/native_resources/tests.rs new file mode 100644 index 0000000..09a04ba --- /dev/null +++ b/crates/canopy-server/src/native_resources/tests.rs @@ -0,0 +1,297 @@ +use super::*; +use std::{future::Future, task::Poll, time::Duration}; + +async fn pending(future: std::pin::Pin<&mut impl Future>) { + let mut future = future; + std::future::poll_fn(|cx| { + assert!(future.as_mut().poll(cx).is_pending()); + Poll::Ready(()) + }) + .await; +} + +#[tokio::test] +async fn idle_drain_permanently_closes_every_scope() -> io::Result<()> { + let pool = NativeResources::default(); + let foreground = pool.scope(NativeClass::Foreground); + let maintenance = foreground.for_class(NativeClass::Maintenance); + pool.drain().await; + pool.close(); + pool.drain().await; + for scope in [foreground, maintenance, pool.scope(NativeClass::Foreground)] { + for work in [NativeWork::Read, NativeWork::Pack] { + let denied = scope.try_admit(work).err().unwrap(); + assert!(is_exhausted(&denied)); + assert!(denied.get_ref().unwrap().is::()); + } + } + assert_eq!(pool.usage()?, NativeUsage::default()); + Ok(()) +} + +#[tokio::test] +async fn all_drain_observers_wait_for_both_classes_and_survive_cancellation() -> io::Result<()> { + let pool = NativeResources::default(); + let read = pool + .scope(NativeClass::Foreground) + .try_admit(NativeWork::Read)?; + let pack = pool + .scope(NativeClass::Maintenance) + .try_admit(NativeWork::Pack)?; + { + let mut canceled = std::pin::pin!(pool.drain()); + pending(canceled.as_mut()).await; + } + assert!( + pool.scope(NativeClass::Foreground) + .try_admit(NativeWork::Read) + .is_err() + ); + let mut first = std::pin::pin!(pool.drain()); + let mut second = std::pin::pin!(pool.drain()); + pending(first.as_mut()).await; + pending(second.as_mut()).await; + drop(read); + pending(first.as_mut()).await; + pending(second.as_mut()).await; + assert_eq!(pool.usage()?.maintenance, NativeWork::Pack.claim()); + drop(pack); + tokio::time::timeout(Duration::from_secs(1), async { + tokio::join!(first, second); + }) + .await + .map_err(io::Error::other)?; + // A late observer needs no notification retained from the final release. + pool.drain().await; + Ok(()) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn final_release_racing_observer_registration_cannot_strand_drain() -> io::Result<()> { + for _ in 0..100 { + let pool = NativeResources::default(); + let claim = pool + .scope(NativeClass::Foreground) + .try_admit(NativeWork::Read)?; + let observer = pool.clone(); + let drain = tokio::spawn(async move { observer.drain().await }); + drop(claim); + tokio::time::timeout(Duration::from_secs(1), drain) + .await + .map_err(io::Error::other)? + .map_err(io::Error::other)?; + } + Ok(()) +} + +#[test] +fn close_racing_shared_admission_is_terminal() -> io::Result<()> { + let pool = NativeResources::default(); + let race = Arc::new(std::sync::Barrier::new(9)); + let mut threads = Vec::new(); + for _ in 0..8 { + let pool = pool.clone(); + let race = Arc::clone(&race); + threads.push(std::thread::spawn(move || { + race.wait(); + let before = pool + .scope(NativeClass::Foreground) + .try_admit(NativeWork::Read) + .ok(); + pool.close(); + let denied = pool + .scope(NativeClass::Maintenance) + .try_admit(NativeWork::Read) + .err() + .unwrap(); + assert!(denied.get_ref().unwrap().is::()); + before + })); + } + race.wait(); + pool.close(); + let claims: Vec<_> = threads + .into_iter() + .filter_map(|thread| thread.join().unwrap()) + .collect(); + assert_eq!(pool.usage()?.foreground.processes as usize, claims.len()); + drop(claims); + assert_eq!(pool.usage()?, NativeUsage::default()); + Ok(()) +} + +#[tokio::test] +async fn poisoned_or_underflowed_accounting_cannot_prove_zero_owner_drain() -> io::Result<()> { + for poison in [false, true] { + let pool = NativeResources::default(); + let claim = pool + .scope(NativeClass::Foreground) + .try_admit(NativeWork::Read)?; + let mut drain = std::pin::pin!(pool.drain()); + pending(drain.as_mut()).await; + if poison { + let _panic = std::panic::catch_unwind(|| { + let _lock = pool.0.state.lock().unwrap(); + panic!("fault injection"); + }); + } else { + // Deliberate counter corruption: a failed release at zero must + // poison the proof rather than report successful shutdown. + pool.0.state.lock().unwrap().used = NativeUsage::default(); + } + drop(claim); + pending(drain.as_mut()).await; + let mut another = std::pin::pin!(pool.drain()); + pending(another.as_mut()).await; + assert!( + pool.scope(NativeClass::Foreground) + .try_admit(NativeWork::Read) + .is_err() + ); + } + Ok(()) +} +#[test] +fn vector_denials_do_not_leak_other_dimensions() -> io::Result<()> { + let pool = NativeResources::default(); + let scope = pool.scope(NativeClass::Foreground); + let mut permits = Vec::new(); + while let Ok(permit) = scope.try_admit(NativeWork::Pack) { + permits.push(permit); + } + assert_eq!(permits.len(), 10); // CPU, before process/memory/descriptor limits. + let prior = pool.usage()?; + for _ in 0..100 { + assert_eq!( + scope.try_admit(NativeWork::Read).err().unwrap().kind(), + io::ErrorKind::WouldBlock + ); + assert_eq!(pool.usage()?, prior); + } + drop(permits); + assert_eq!(pool.usage()?, NativeUsage::default()); + Ok(()) +} +#[test] +fn maintenance_and_foreground_shares_survive_each_others_saturation() -> io::Result<()> { + let pool = NativeResources::default(); + let foreground = pool.scope(NativeClass::Foreground); + let maintenance = foreground.for_class(NativeClass::Maintenance); + let mut active = Vec::new(); + while let Ok(permit) = foreground.try_admit(NativeWork::Pack) { + active.push(permit); + } + let listing = maintenance.try_admit(NativeWork::Read)?; + let pack = maintenance.try_admit(NativeWork::Pack)?; + assert!(maintenance.try_admit(NativeWork::Read).is_err()); + drop(active); + let foreground_work = foreground.try_admit(NativeWork::Pack)?; + drop((listing, pack, foreground_work)); + assert_eq!(pool.usage()?, NativeUsage::default()); + Ok(()) +} +#[test] +fn invalid_and_overflowing_shares_reject_before_pool_creation() { + let mut limits = NativeLimits::default(); + limits.maintenance_reserved.memory_bytes = u64::MAX; + assert!(NativeResources::new(limits).is_err()); + limits = NativeLimits::default(); + limits.total.cpu_units = limits.maintenance_reserved.cpu_units; + assert!(NativeResources::new(limits).is_err()); + limits = NativeLimits::default(); + limits.maintenance_reserved.processes = 1; + assert!(NativeResources::new(limits).is_err()); + limits = NativeLimits::default(); + limits.pack.memory_bytes = u64::MAX; + assert!(NativeResources::new(limits).is_err()); + limits = NativeLimits::default(); + limits.pack.cpu_units = u32::MAX; + assert!(NativeResources::new(limits).is_err()); +} + +#[test] +fn every_dimension_can_be_the_limiting_resource() -> io::Result<()> { + for dimension in 0..4 { + let mut limits = NativeLimits::default(); + let mut foreground = NativeCapacity { + processes: 100, + cpu_units: 100, + memory_bytes: 100 << 30, + descriptors: 10000, + }; + let exact = limits.read.add(limits.pack).unwrap(); + match dimension { + 0 => foreground.processes = exact.processes, + 1 => foreground.cpu_units = exact.cpu_units, + 2 => foreground.memory_bytes = exact.memory_bytes, + _ => foreground.descriptors = exact.descriptors, + } + limits.total = foreground.add(limits.maintenance_reserved).unwrap(); + let pool = NativeResources::new(limits)?; + let scope = pool.scope(NativeClass::Foreground); + let pack = scope.try_admit(NativeWork::Pack)?; + let read = scope.try_admit(NativeWork::Read)?; + let used = pool.usage()?; + let denied = scope.try_admit(NativeWork::Read).err().unwrap(); + assert!(is_exhausted(&denied)); + assert_eq!(pool.usage()?, used); + drop((pack, read)); + assert_eq!(pool.usage()?, NativeUsage::default()); + } + Ok(()) +} + +#[test] +fn simultaneous_admission_is_atomic_across_shared_scopes() -> io::Result<()> { + let mut limits = NativeLimits::default(); + limits.total.processes = limits.maintenance_reserved.processes + 4; + let pool = NativeResources::new(limits)?; + let admitted = Arc::new(std::sync::Barrier::new(9)); + let release = Arc::new(std::sync::Barrier::new(9)); + let mut threads = Vec::new(); + for _ in 0..8 { + let scope = pool.scope(NativeClass::Foreground); + let admitted = Arc::clone(&admitted); + let release = Arc::clone(&release); + threads.push(std::thread::spawn(move || { + let permit = scope.try_admit(NativeWork::Read); + admitted.wait(); + release.wait(); + permit.is_ok() + })); + } + admitted.wait(); + assert_eq!(pool.usage()?.foreground.processes, 4); + assert_eq!( + pool.usage()?.foreground.memory_bytes, + 4 * limits.read.memory_bytes + ); + release.wait(); + assert_eq!( + threads + .into_iter() + .map(|thread| thread.join().unwrap()) + .filter(|admitted| *admitted) + .count(), + 4 + ); + assert_eq!(pool.usage()?, NativeUsage::default()); + Ok(()) +} + +#[test] +fn poisoned_admission_never_returns_a_live_claim() -> io::Result<()> { + let pool = NativeResources::default(); + let scope = pool.scope(NativeClass::Foreground); + let permit = scope.try_admit(NativeWork::Read)?; + let prior = pool.usage()?; + let _panic = std::panic::catch_unwind(|| { + let _lock = pool.0.state.lock().unwrap(); + panic!("fault injection"); + }); + assert!(scope.try_admit(NativeWork::Read).is_err()); + drop(permit); + assert!(pool.usage().is_err()); + assert_eq!(pool.0.state.lock().err().unwrap().into_inner().used, prior); + Ok(()) +} diff --git a/crates/canopy-server/src/pack_store.rs b/crates/canopy-server/src/pack_store.rs index 042929f..6968a69 100644 --- a/crates/canopy-server/src/pack_store.rs +++ b/crates/canopy-server/src/pack_store.rs @@ -29,6 +29,7 @@ pub(crate) struct PackReader { root: PathBuf, budget: DiskBudget, format: crate::ObjectFormat, + native: crate::native_resources::NativeScope, cache: Mutex>>, installation: Mutex<()>, private: Mutex)>>, @@ -40,12 +41,14 @@ impl PackReader { root: PathBuf, budget: DiskBudget, format: crate::ObjectFormat, + native: crate::native_resources::NativeScope, ) -> Self { Self { store: LargeBlobStore::new(store, repository), root, budget, format, + native, cache: Mutex::new(None), installation: Mutex::new(()), private: Mutex::new(None), @@ -60,6 +63,7 @@ impl PackReader { self.budget.clone(), "refs/heads/main", self.format, + self.native.clone(), ) .await?, ); @@ -119,6 +123,7 @@ impl PackReader { self.budget.clone(), "refs/heads/main", self.format, + self.native.clone(), ) .await?, )); @@ -134,7 +139,13 @@ impl PackReader { .args(["cat-file", "blob", &hex::encode(oid)]) .stdout(std::process::Stdio::piped()) .stderr(std::process::Stdio::null()); - let mut process = GitProcess::spawn(command, Arc::clone(&cache))?; + let mut process = GitProcess::spawn( + command, + Arc::clone(&cache), + cache + .native + .try_admit(crate::native_resources::NativeWork::Read)?, + )?; let output = process .child .stdout @@ -253,13 +264,10 @@ impl NativePackedRead { { return Err(GatewayError::MalformedCache); } - let status = tokio::time::timeout( - std::time::Duration::from_secs(120), - self.process.child.wait(), - ) - .await - .map_err(|_| crate::git_http::GitHttpError::Timeout)??; - self.process.disarm(); + let status = + tokio::time::timeout(std::time::Duration::from_secs(120), self.process.wait()) + .await + .map_err(|_| crate::git_http::GitHttpError::Timeout)??; if !status.success() { return Err(GatewayError::MalformedCache); } diff --git a/crates/canopy-server/src/packs/catalog/codec.rs b/crates/canopy-server/src/packs/catalog/codec.rs new file mode 100644 index 0000000..624015b --- /dev/null +++ b/crates/canopy-server/src/packs/catalog/codec.rs @@ -0,0 +1,94 @@ +use super::super::directory::index::codec::{ + artifact, fixed, read_artifact, read_reference, reference, +}; +use super::*; +use cellule_runtime::codec::{BoundedDecoder, BoundedEncoder}; + +impl CatalogSnapshot { + pub(super) fn encode(self, operation: [u8; 16]) -> Result, IndexError> { + self.validate()?; + let mut encoder = BoundedEncoder::new(CATALOG_BYTES)?; + encoder.write_bytes(b"canopy.catalog-root.v1\0")?; + encoder.write_bytes(&self.directory.repository)?; + encoder.write_bytes(&operation)?; + encoder.write_u8(self.directory.format.bytes() as u8)?; + encoder.write_bytes(&self.directory.operation)?; + artifact(&mut encoder, self.directory.artifact)?; + encoder.write_bool(self.sources.is_some())?; + if let Some(root) = self.sources { + reference(&mut encoder, root)?; + } + Ok(encoder.finish()) + } + pub(super) fn decode(bytes: &[u8]) -> Result<(Self, [u8; 16]), IndexError> { + let mut decoder = BoundedDecoder::new(bytes, CATALOG_BYTES)?; + if decoder.read_bytes()? != b"canopy.catalog-root.v1\0" { + return Err(IndexError::Integrity); + } + let repository = fixed(&mut decoder)?; + let operation = fixed(&mut decoder)?; + let format = match decoder.read_u8()? { + 20 => ObjectFormat::Sha1, + 32 => ObjectFormat::Sha256, + _ => return Err(IndexError::Integrity), + }; + let directory = StoredSnapshot { + repository, + format, + operation: fixed(&mut decoder)?, + artifact: read_artifact(&mut decoder)?, + }; + let sources = if decoder.read_bool()? { + Some(read_reference(&mut decoder, format)?) + } else { + None + }; + decoder.finish()?; + let snapshot = Self { directory, sources }; + snapshot.validate()?; + Ok((snapshot, operation)) + } +} + +// Reuse the same descriptor and artifact encoding for authoritative Cell facts. +// This record is distinct from the catalog artifact's two-root payload. +impl cellule_runtime::codec::WireValue for StoredCatalog { + fn encode( + &self, + encoder: &mut BoundedEncoder, + ) -> Result<(), cellule_runtime::codec::CodecError> { + self.validate().map_err(|_| { + cellule_runtime::codec::CodecError::Invalid("invalid catalog reference") + })?; + encoder.write_bytes(b"canopy.catalog-ref.v1\0")?; + encoder.write_bytes(&self.repository)?; + encoder.write_bytes(&self.operation)?; + encoder.write_u8(self.format.bytes() as u8)?; + artifact(encoder, self.artifact) + } + fn decode( + decoder: &mut BoundedDecoder<'_>, + ) -> Result { + use cellule_runtime::codec::CodecError; + if decoder.read_bytes()? != b"canopy.catalog-ref.v1\0" { + return Err(CodecError::Invalid("invalid catalog reference domain")); + } + let repository = fixed(decoder)?; + let operation = fixed(decoder)?; + let format = match decoder.read_u8()? { + 20 => ObjectFormat::Sha1, + 32 => ObjectFormat::Sha256, + _ => return Err(CodecError::Invalid("invalid catalog reference format")), + }; + let value = Self { + repository, + operation, + format, + artifact: read_artifact(decoder)?, + }; + value + .validate() + .map_err(|_| CodecError::Invalid("invalid catalog reference"))?; + Ok(value) + } +} diff --git a/crates/canopy-server/src/packs/catalog/files.rs b/crates/canopy-server/src/packs/catalog/files.rs new file mode 100644 index 0000000..5be6fc8 --- /dev/null +++ b/crates/canopy-server/src/packs/catalog/files.rs @@ -0,0 +1,289 @@ +//! Service-owned authenticated SQLite files. Bounds apply to cached files, +//! borrowed readers and canceled blocking jobs together, not just cache entries. +use super::*; +use crate::packs::{ + directory::{DirectoryRun, StoredRun}, + metadata::{MetadataError, MetadataLimits, MetadataSegment, ReaderAdmission, StoredSegment}, +}; +use cellule_ltx::DiskBudget; +use std::{ + collections::VecDeque, + path::Path, + sync::{ + Mutex, + atomic::{AtomicU64, Ordering}, + }, +}; +use tokio::sync::{Mutex as AsyncMutex, Semaphore}; + +const LOAD_STRIPES: usize = 16; +pub const MAX_OPEN_CATALOG_FILES: u32 = 128; + +#[derive(Clone, Copy, Debug)] +pub struct CatalogFileLimits { + /// Includes borrowed/evicted files and outstanding canceled worker jobs. + pub open_files: u32, + pub cached_files: usize, + pub metadata: MetadataLimits, +} +impl Default for CatalogFileLimits { + fn default() -> Self { + Self { + open_files: 32, + cached_files: 16, + metadata: MetadataLimits::default(), + } + } +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct CatalogFileStats { + pub open_files: u32, + pub cached_files: usize, + pub cache_hits: u64, + pub downloaded_files: u64, +} + +/// Shared across catalog generations in one worker. Creating a loader does not +/// certify its catalogs, grant access, or pin remote generations against GC. +pub struct CatalogFiles { + root: Arc, + store: Arc, + format: ObjectFormat, + budget: DiskBudget, + limits: CatalogFileLimits, + slots: Arc, + cache: Mutex>, + // Coalesce equal misses without serializing all unrelated transfers. + // The fixed stripe array cannot grow with repository history. + loads: [AsyncMutex<()>; LOAD_STRIPES], + cache_hits: AtomicU64, + downloaded_files: AtomicU64, +} +#[derive(Clone, Copy, PartialEq, Eq)] +enum Key { + Run(StoredRun), + Metadata(StoredSegment), +} +impl Key { + fn identity(self) -> (u8, [u8; 16], [u8; 32]) { + match self { + Self::Run(stored) => (1, stored.run.operation, stored.artifact.digest), + Self::Metadata(stored) => { + (2, stored.segment.identity.operation, stored.artifact.digest) + } + } + } + fn stripe(self) -> usize { + let (kind, operation, digest) = self.identity(); + let mut hash = blake3::Hasher::new(); + hash.update(&[kind]); + hash.update(&operation); + hash.update(&digest); + usize::from(hash.finalize().as_bytes()[0]) % LOAD_STRIPES + } +} +#[derive(Clone)] +enum Value { + Run(Arc), + Metadata(Arc), +} +impl Value { + fn unborrowed(&self) -> bool { + match self { + Self::Run(run) => Arc::strong_count(run) == 1, + Self::Metadata(segment) => Arc::strong_count(segment) == 1, + } + } +} +struct Entry { + key: Key, + value: Value, +} +impl CatalogFiles { + pub fn new( + workspace: &Path, + budget: DiskBudget, + store: Arc, + format: ObjectFormat, + limits: CatalogFileLimits, + ) -> Result { + if limits.open_files == 0 + || limits.open_files > MAX_OPEN_CATALOG_FILES + || limits.cached_files > limits.open_files as usize + || limits.metadata.cache_kib == 0 + || limits.metadata.cache_kib > i32::MAX as u32 + || limits.metadata.max_file_bytes == 0 + || limits.metadata.max_file_bytes > canopy_object_storage::external::MAX_ARTIFACT_BYTES + { + return Err(MetadataError::Limit); + } + let root = tempfile::Builder::new() + .prefix("canopy-catalog-files-") + .tempdir_in(workspace)?; + Ok(Self { + root: Arc::new(root), + store, + format, + budget, + limits, + slots: Arc::new(Semaphore::new(limits.open_files as usize)), + cache: Mutex::new(VecDeque::new()), + loads: std::array::from_fn(|_| AsyncMutex::new(())), + cache_hits: AtomicU64::new(0), + downloaded_files: AtomicU64::new(0), + }) + } + pub fn stats(&self) -> Result { + Ok(CatalogFileStats { + open_files: self.limits.open_files - self.slots.available_permits() as u32, + cached_files: self + .cache + .lock() + .map_err(|_| MetadataError::Integrity)? + .len(), + cache_hits: self.cache_hits.load(Ordering::Relaxed), + downloaded_files: self.downloaded_files.load(Ordering::Relaxed), + }) + } + fn cached(&self, key: Key) -> Result, MetadataError> { + let mut cache = self.cache.lock().map_err(|_| MetadataError::Integrity)?; + let Some(at) = cache + .iter() + .position(|entry| entry.key.identity() == key.identity()) + else { + return Ok(None); + }; + // A digest cache key must never hide a different descriptor/context. + if cache[at].key != key { + return Err(MetadataError::Integrity); + } + let entry = cache.remove(at).ok_or(MetadataError::Integrity)?; + let value = entry.value.clone(); + cache.push_back(entry); + self.cache_hits.fetch_add(1, Ordering::Relaxed); + Ok(Some(value)) + } + async fn admission(&self) -> Result { + loop { + if let Ok(slot) = Arc::clone(&self.slots).try_acquire_owned() { + return Ok(ReaderAdmission { + _slot: slot, + _root: Arc::clone(&self.root), + }); + } + // Evict only an unborrowed file to make room. If all live slots + // belong to callers/jobs, fail admission instead of deadlocking. + let evicted = { + let mut cache = self.cache.lock().map_err(|_| MetadataError::Integrity)?; + cache + .iter() + .position(|entry| entry.value.unborrowed()) + .and_then(|at| cache.remove(at)) + }; + let Some(evicted) = evicted else { + return Err(MetadataError::Limit); + }; + tokio::task::spawn_blocking(move || drop(evicted)).await?; + } + } + async fn remember(&self, key: Key, value: Value) -> Result<(), MetadataError> { + let evicted = { + let mut cache = self.cache.lock().map_err(|_| MetadataError::Integrity)?; + if self.limits.cached_files == 0 { + return Ok(()); + } + cache.push_back(Entry { key, value }); + if cache.len() > self.limits.cached_files { + cache.pop_front() + } else { + None + } + }; + if let Some(evicted) = evicted { + // Queued cleanup keeps the file's slot and disk charge until its + // connection closes and private file is removed. + tokio::task::spawn_blocking(move || drop(evicted)).await?; + } + Ok(()) + } + async fn load_value(&self, key: Key) -> Result { + match key { + Key::Run(stored) => { + stored.validate()?; + if stored.run.repository != self.store.repository() + || stored.run.format != self.format + { + return Err(MetadataError::Integrity); + } + } + Key::Metadata(stored) => { + if stored.segment.identity.repository != self.store.repository() + || stored.segment.identity.format != self.format + || stored.segment.size != stored.artifact.size + || stored.segment.digest != stored.artifact.digest + { + return Err(MetadataError::Integrity); + } + } + } + if let Some(value) = self.cached(key)? { + return Ok(value); + } + let _loading = self.loads[key.stripe()].lock().await; + if let Some(value) = self.cached(key)? { + return Ok(value); + } + let reader = self.admission().await?; + let value = match key { + Key::Run(stored) => Value::Run( + DirectoryRun::download_for_reader( + self.root.path(), + self.budget.clone(), + &self.store, + stored, + self.limits.metadata, + Some(reader), + ) + .await?, + ), + Key::Metadata(stored) => Value::Metadata( + MetadataSegment::download_for_reader( + self.root.path(), + self.budget.clone(), + &self.store, + stored, + self.limits.metadata, + Some(reader), + ) + .await?, + ), + }; + self.downloaded_files.fetch_add(1, Ordering::Relaxed); + self.remember(key, value.clone()).await?; + Ok(value) + } +} +impl RunLoader for CatalogFiles { + async fn load(&self, run: StoredRun) -> Result, MetadataError> { + run.validate()?; + // Cache the complete authenticated file once across disjoint projections. + // Coverage is certified by the catalog and explicitly folded by writers; + // a different physical descriptor/manifest still rejects on a cache hit. + let physical = StoredRun { + coverage: run.run.coverage(), + ..run + }; + match self.load_value(Key::Run(physical)).await? { + Value::Run(run) => Ok(run), + Value::Metadata(_) => Err(MetadataError::Integrity), + } + } +} +impl SourceLoader for CatalogFiles { + async fn load(&self, segment: StoredSegment) -> Result, MetadataError> { + match self.load_value(Key::Metadata(segment)).await? { + Value::Metadata(segment) => Ok(segment), + Value::Run(_) => Err(MetadataError::Integrity), + } + } +} diff --git a/crates/canopy-server/src/packs/catalog/mod.rs b/crates/canopy-server/src/packs/catalog/mod.rs new file mode 100644 index 0000000..0ee499b --- /dev/null +++ b/crates/canopy-server/src/packs/catalog/mod.rs @@ -0,0 +1,118 @@ +//! One immutable catalog binding directory and source roots. Root upload is +//! preparation, never the ref commit point or a canonical/closure certificate. + +use super::{ + directory::{ + DirectoryEntry, + index::{IndexError, RangeIndex}, + snapshot::{DirectorySnapshot, RunLoader, StoredSnapshot}, + }, + sources::{ResolvedSource, SourceIndex, SourceLoader, SourceRoot}, +}; +use crate::{ObjectFormat, ObjectId}; +use canopy_object_storage::artifact::{ + ArtifactDescriptor, ArtifactKey, ArtifactKind, ArtifactStore, +}; +use std::sync::Arc; + +mod codec; +mod files; +pub use files::{CatalogFileLimits, CatalogFileStats, CatalogFiles, MAX_OPEN_CATALOG_FILES}; +mod reader; +pub use reader::{CatalogIndexes, CatalogReader, ResolvedObject}; + +/// Exactly two root descriptors; history is not a linked list of prior roots. +pub const CATALOG_BYTES: u32 = 1024; +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct CatalogSnapshot { + pub directory: StoredSnapshot, + pub sources: Option, +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct StoredCatalog { + pub repository: [u8; 16], + pub operation: [u8; 16], + pub format: ObjectFormat, + pub artifact: ArtifactDescriptor, +} +impl StoredCatalog { + pub fn validate(self) -> Result<(), IndexError> { + if self.artifact.size == 0 || self.artifact.size > u64::from(CATALOG_BYTES) { + return Err(IndexError::Integrity); + } + Ok(()) + } + fn key(self) -> ArtifactKey { + ArtifactKey { + operation: self.operation, + binding_digest: self.artifact.digest, + kind: ArtifactKind::CatalogNode, + } + } +} +impl CatalogSnapshot { + pub fn validate(self) -> Result<(), IndexError> { + if self.directory.artifact.size == 0 + || self.directory.artifact.size > u64::from(super::directory::index::NODE_BYTES) + { + return Err(IndexError::Integrity); + } + if let Some(root) = self.sources { + root.validate(self.directory.format)?; + } + Ok(()) + } + /// Persist prepared root bytes. The Cell later publishes this descriptor + /// together with verified facts, refs and exact durable outcomes under CAS. + pub async fn upload( + self, + store: &ArtifactStore, + operation: [u8; 16], + ) -> Result { + self.validate()?; + if self.directory.repository != store.repository() { + return Err(IndexError::Integrity); + } + let bytes = self.encode(operation)?; + let digest = *blake3::hash(&bytes).as_bytes(); + let key = ArtifactKey { + operation, + binding_digest: digest, + kind: ArtifactKind::CatalogNode, + }; + let artifact = store + .put(key, bytes.len() as u64, digest, &mut bytes.as_slice()) + .await?; + Ok(StoredCatalog { + repository: store.repository(), + operation, + format: self.directory.format, + artifact, + }) + } + pub async fn download( + store: &ArtifactStore, + stored: StoredCatalog, + ) -> Result { + stored.validate()?; + if stored.repository != store.repository() { + return Err(IndexError::Integrity); + } + let mut reader = store.read(stored.key(), stored.artifact).await?; + let bytes = reader.next().await?.ok_or(IndexError::Integrity)?; + if reader.next().await?.is_some() { + return Err(IndexError::Integrity); + } + let (snapshot, operation) = Self::decode(&bytes)?; + if snapshot.directory.repository != stored.repository + || snapshot.directory.format != stored.format + || operation != stored.operation + { + return Err(IndexError::Integrity); + } + Ok(snapshot) + } +} + +#[cfg(test)] +pub(in crate::packs) mod tests; diff --git a/crates/canopy-server/src/packs/catalog/reader.rs b/crates/canopy-server/src/packs/catalog/reader.rs new file mode 100644 index 0000000..1d7d640 --- /dev/null +++ b/crates/canopy-server/src/packs/catalog/reader.rs @@ -0,0 +1,167 @@ +use super::*; + +/// One repository/format's bounded node clients, retained by a read worker +/// across catalog publications. New generations reuse unchanged authenticated +/// nodes rather than allocating new caches per snapshot/request. +pub struct CatalogIndexes { + store: Arc, + ranges: RangeIndex, + sources: Arc, + inputs: super::super::sources::NativeInputIndex, +} +impl CatalogIndexes { + pub fn new(store: Arc, format: ObjectFormat) -> Self { + Self { + ranges: RangeIndex::new(Arc::clone(&store), format), + sources: Arc::new(SourceIndex::new(Arc::clone(&store), format)), + inputs: super::super::sources::NativeInputIndex::new(Arc::clone(&store), format), + store, + } + } + pub(in crate::packs) fn store(&self) -> Arc { + Arc::clone(&self.store) + } + pub(in crate::packs) fn sources(&self) -> Arc { + Arc::clone(&self.sources) + } + pub(in crate::packs) fn inputs(&self) -> &super::super::sources::NativeInputIndex { + &self.inputs + } + pub fn input_stats(&self) -> super::super::directory::index::ReadStats { + self.inputs.stats() + } + pub(in crate::packs) fn ranges(&self) -> &RangeIndex { + &self.ranges + } + pub fn stats( + &self, + ) -> ( + super::super::directory::index::ReadStats, + super::super::directory::index::ReadStats, + ) { + (self.ranges.stats(), self.sources.stats()) + } +} + +/// Immutable reader for one complete catalog binding. The service supplies +/// authorization, closure certificates, generation leases and loader admission. +/// Opening verifies root context; it does not certify every descendant. +pub struct CatalogReader { + stored: StoredCatalog, + directory: DirectorySnapshot, + indexes: Arc, + source_root: Option, +} +pub struct ResolvedObject { + pub entry: DirectoryEntry, + pub source: ResolvedSource, +} +impl CatalogReader { + pub async fn open( + indexes: Arc, + stored: StoredCatalog, + ) -> Result { + if stored.repository != indexes.store.repository() + || stored.format != indexes.sources.format() + { + return Err(IndexError::Integrity); + } + let snapshot = CatalogSnapshot::download(&indexes.store, stored).await?; + let directory = DirectorySnapshot::download(&indexes.store, snapshot.directory).await?; + let nonempty = + !directory.level_zero.is_empty() || directory.levels.iter().any(Option::is_some); + if nonempty && snapshot.sources.is_none() { + return Err(IndexError::Integrity); + } + // At most 48 directory roots and one source root, independent of + // the number of objects/artifacts. Node clients retain bounded caches. + for root in directory + .level_zero + .iter() + .chain(directory.levels.iter().flatten()) + { + indexes.ranges.validate_root(*root).await?; + } + if let Some(root) = snapshot.sources { + indexes.sources.validate_root(root).await?; + } + Ok(Self { + stored, + directory, + indexes, + source_root: snapshot.sources, + }) + } + pub fn stored(&self) -> StoredCatalog { + self.stored + } + pub(in crate::packs) fn directory(&self) -> DirectorySnapshot { + self.directory.clone() + } + pub(in crate::packs) fn source_root(&self) -> Option { + self.source_root + } + pub async fn lookup( + &self, + oid: ObjectId, + runs: &impl RunLoader, + metadata: &impl SourceLoader, + ) -> Result, IndexError> { + let Some(entry) = self + .directory + .lookup(&self.indexes.ranges, runs, oid) + .await? + else { + return Ok(None); + }; + let source = self + .indexes + .sources + .resolve(self.source_root, entry, metadata) + .await?; + Ok(Some(ResolvedObject { entry, source })) + } + /// Authenticated canonical headers in request order, at most 512. This + /// verifies preferred source bindings; it does not grant closure authority. + /// Each metadata file is released before the next group is opened. + pub async fn headers( + &self, + ids: &[ObjectId], + runs: &impl RunLoader, + metadata: &impl SourceLoader, + ) -> Result>, IndexError> { + let entries = self + .directory + .lookup_batch(&self.indexes.ranges, runs, ids) + .await?; + let mut groups = std::collections::BTreeMap::< + super::super::directory::SegmentKey, + Vec<(usize, DirectoryEntry)>, + >::new(); + for (at, entry) in entries.into_iter().enumerate() { + if let Some(entry) = entry { + groups.entry(entry.source).or_default().push((at, entry)); + } + } + let mut output = vec![None; ids.len()]; + for (key, group) in groups { + let record = self + .indexes + .sources + .find(self.source_root, key) + .await? + .ok_or(IndexError::Integrity)?; + let segment = metadata.load(record.metadata).await?; + let requested: Vec<_> = group.iter().map(|(_, entry)| *entry).collect(); + tokio::task::spawn_blocking(move || { + record.verify_directory_entries(&segment, &requested) + }) + .await + .map_err(super::super::metadata::MetadataError::from)??; + for (at, entry) in group { + output[at] = Some(entry.header); + } + } + Ok(output) + } +} diff --git a/crates/canopy-server/src/packs/catalog/tests.rs b/crates/canopy-server/src/packs/catalog/tests.rs new file mode 100644 index 0000000..20176a9 --- /dev/null +++ b/crates/canopy-server/src/packs/catalog/tests.rs @@ -0,0 +1,386 @@ +use super::super::{ + directory::{DirectoryBuilder, DirectoryRun}, + metadata::{ + MetadataError, MetadataSegment, StoredSegment, + tests::{Fixture, builder, fill, fixture, limits}, + }, + sources::{SourceRecord, tests::source}, +}; +use super::*; +use cellule_ltx::DiskBudget; +use object_store::{ObjectStore, ObjectStoreExt, memory::InMemory}; +use std::path::Path; + +type Result = std::result::Result>; +pub(in crate::packs) struct Prepared { + pub(in crate::packs) fixture: Fixture, + pub(in crate::packs) store: Arc, + provider: Arc, + snapshot: CatalogSnapshot, + pub(in crate::packs) stored: StoredCatalog, + pub(in crate::packs) indexes: Arc, +} +async fn prepared(format: ObjectFormat) -> Result { + prepared_for_repository(format, [1; 16]).await +} +pub(in crate::packs) async fn prepared_for_repository( + format: ObjectFormat, + repository: [u8; 16], +) -> Result { + let mut fixture = fixture(format, 4).await?; + fixture.identity.repository = repository; + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new(Arc::clone(&provider), repository)); + let mut writer = builder(&fixture, DiskBudget::new(128 << 20), fixture.identity)?; + fill( + &mut writer, + &fixture.objects.values().cloned().collect::>(), + )?; + let segment = Arc::new(writer.seal(&fixture.index)?); + let metadata = Arc::clone(&segment).upload(&store).await?; + let index_path = std::fs::read_dir(fixture.root.path().join("objects/pack"))? + .filter_map(|e| e.ok().map(|e| e.path())) + .find(|path| path.extension().is_some_and(|ext| ext == "idx")) + .ok_or("index")?; + let mut artifacts = Vec::new(); + for (kind, path) in [ + (ArtifactKind::Pack, index_path.with_extension("pack")), + (ArtifactKind::Index, index_path), + ] { + let bytes = std::fs::read(path)?; + artifacts.push( + store + .put( + ArtifactKey { + operation: fixture.identity.operation, + binding_digest: fixture.identity.pack_digest, + kind, + }, + bytes.len() as u64, + *blake3::hash(&bytes).as_bytes(), + &mut bytes.as_slice(), + ) + .await?, + ); + } + let source = SourceRecord { + metadata, + pack: artifacts[0], + index: artifacts[1], + pack_object_count: fixture.index.len(), + }; + let sources = SourceIndex::new(Arc::clone(&store), format) + .insert(None, [6; 16], source) + .await?; + let mut directory = DirectoryBuilder::new( + fixture.root.path(), + DiskBudget::new(128 << 20), + repository, + [5; 16], + format, + limits(), + )?; + directory.add_segment(&segment)?; + let run = Arc::new(directory.seal()?).upload(&store).await?; + let indexes = Arc::new(CatalogIndexes::new(Arc::clone(&store), format)); + let root = indexes.ranges().insert(None, [5; 16], run).await?; + let mut directory = DirectorySnapshot::empty(repository, format); + directory.append(indexes.ranges(), root).await?; + let snapshot = CatalogSnapshot { + directory: directory.upload(&store, [7; 16]).await?, + sources: Some(sources), + }; + let stored = snapshot.upload(&store, [8; 16]).await?; + Ok(Prepared { + fixture, + store, + provider, + snapshot, + stored, + indexes, + }) +} +struct Loader<'a> { + root: &'a Path, + store: &'a ArtifactStore, + budget: DiskBudget, +} +impl RunLoader for Loader<'_> { + async fn load( + &self, + run: super::super::directory::StoredRun, + ) -> std::result::Result, MetadataError> { + DirectoryRun::download(self.root, self.budget.clone(), self.store, run, limits()).await + } +} +impl SourceLoader for Loader<'_> { + async fn load( + &self, + segment: StoredSegment, + ) -> std::result::Result, MetadataError> { + MetadataSegment::download( + self.root, + self.budget.clone(), + self.store, + segment, + limits(), + ) + .await + } +} + +#[tokio::test] +async fn native_catalog_roundtrip_and_old_roots_remain_independently_readable() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let prepared = prepared(format).await?; + assert!(prepared.stored.artifact.size <= u64::from(CATALOG_BYTES)); + assert_eq!( + CatalogSnapshot::download(&prepared.store, prepared.stored).await?, + prepared.snapshot + ); + let reader = CatalogReader::open(Arc::clone(&prepared.indexes), prepared.stored).await?; + assert_eq!(reader.stored(), prepared.stored); + let before = prepared.indexes.stats(); + let next = prepared.snapshot.upload(&prepared.store, [13; 16]).await?; + let next = CatalogReader::open(Arc::clone(&prepared.indexes), next).await?; + let after = prepared.indexes.stats(); + assert_eq!(after.0.loaded_nodes, before.0.loaded_nodes); + assert_eq!(after.1.loaded_nodes, before.1.loaded_nodes); + assert!(after.1.cache_hits > before.1.cache_hits); + assert_ne!(next.stored(), reader.stored()); + let budget = DiskBudget::new(128 << 20); + let loader = Loader { + root: prepared.fixture.root.path(), + store: &prepared.store, + budget: budget.clone(), + }; + for (oid, (expected, _)) in &prepared.fixture.objects { + let resolved = reader + .lookup(*oid, &loader, &loader) + .await? + .ok_or("resolved")?; + assert_eq!(resolved.entry.header.object, *expected); + assert_eq!( + resolved.source.metadata.header(*oid)?, + Some(resolved.entry.header) + ); + assert_eq!(budget.used(), resolved.source.record.metadata.artifact.size); + drop(resolved); + assert_eq!(budget.used(), 0); + } + let missing = if format == ObjectFormat::Sha1 { + ObjectId::Sha1([255; 20]) + } else { + ObjectId::Sha256([255; 32]) + }; + assert!(reader.lookup(missing, &loader, &loader).await?.is_none()); + let directory = DirectorySnapshot::empty([1; 16], format) + .upload(&prepared.store, [9; 16]) + .await?; + let empty = CatalogSnapshot { + directory, + sources: None, + } + .upload(&prepared.store, [10; 16]) + .await?; + let empty = CatalogReader::open(Arc::clone(&prepared.indexes), empty).await?; + let oid = *prepared.fixture.objects.keys().next().ok_or("oid")?; + assert!(empty.lookup(oid, &loader, &loader).await?.is_none()); + assert!(reader.lookup(oid, &loader, &loader).await?.is_some()); + } + Ok(()) +} + +#[test] +fn catalog_codec_rejects_truncation_trailing_wrong_domains_and_oversized_descriptors() -> Result { + let format = ObjectFormat::Sha256; + let snapshot = CatalogSnapshot { + directory: StoredSnapshot { + repository: [1; 16], + operation: [2; 16], + format, + artifact: ArtifactDescriptor { + size: 128, + digest: [3; 32], + manifest_digest: [4; 32], + }, + }, + sources: None, + }; + let bytes = snapshot.encode([5; 16])?; + assert_eq!(CatalogSnapshot::decode(&bytes)?, (snapshot, [5; 16])); + for n in 0..bytes.len() { + assert!(CatalogSnapshot::decode(&bytes[..n]).is_err()); + } + let mut trailing = bytes.clone(); + trailing.push(0); + assert!(CatalogSnapshot::decode(&trailing).is_err()); + let mut domain = bytes.clone(); + domain[4] ^= 1; + assert!(CatalogSnapshot::decode(&domain).is_err()); + assert!(CatalogSnapshot::decode(&vec![0; CATALOG_BYTES as usize + 1]).is_err()); + let mut oversized = snapshot; + oversized.directory.artifact.size = 65537; + assert!(oversized.encode([5; 16]).is_err()); + let stored = StoredCatalog { + repository: [1; 16], + operation: [5; 16], + format, + artifact: ArtifactDescriptor { + size: u64::from(CATALOG_BYTES) + 1, + digest: [3; 32], + manifest_digest: [4; 32], + }, + }; + assert!(stored.validate().is_err()); + Ok(()) +} + +#[test] +fn persisted_catalog_contract_matches_independent_big_endian_golden_vectors() -> Result { + use crate::packs::directory::SegmentKey; + for (format, golden) in [ + ( + ObjectFormat::Sha1, + include_str!(concat!( + env!("CARGO_MANIFEST_DIR"), + "/../../docs/design/catalog-root-v1-sha1.hex" + )), + ), + ( + ObjectFormat::Sha256, + include_str!(concat!( + env!("CARGO_MANIFEST_DIR"), + "/../../docs/design/catalog-root-v1-sha256.hex" + )), + ), + ] { + // Codec fixtures only; these descriptors do not name populated files. + let snapshot = CatalogSnapshot { + directory: StoredSnapshot { + repository: [1; 16], + operation: [2; 16], + format, + artifact: ArtifactDescriptor { + size: 128, + digest: [3; 32], + manifest_digest: [4; 32], + }, + }, + sources: Some(SourceRoot { + operation: [12; 16], + height: 1, + artifact: ArtifactDescriptor { + size: 4096, + digest: [10; 32], + manifest_digest: [11; 32], + }, + first_key: SegmentKey { + operation: [6; 16], + digest: [7; 32], + }, + last_key: SegmentKey { + operation: [8; 16], + digest: [9; 32], + }, + record_count: 11, + object_count: 64, + }), + }; + let bytes = hex::decode(golden.trim())?; + assert_eq!(snapshot.encode([5; 16])?, bytes); + assert_eq!(CatalogSnapshot::decode(&bytes)?, (snapshot, [5; 16])); + } + Ok(()) +} + +#[tokio::test] +async fn missing_or_wrong_source_roots_cannot_resolve_a_directory_entry() -> Result { + let prepared = prepared(ObjectFormat::Sha256).await?; + let mut missing = prepared.snapshot; + missing.sources = None; + let stored = missing.upload(&prepared.store, [9; 16]).await?; + assert!( + CatalogReader::open(Arc::clone(&prepared.indexes), stored) + .await + .is_err() + ); + let index = SourceIndex::new(Arc::clone(&prepared.store), ObjectFormat::Sha256); + let unrelated = index + .insert(None, [10; 16], source(1, ObjectFormat::Sha256)) + .await?; + let mut dangling = prepared.snapshot; + dangling.sources = Some(unrelated); + let stored = dangling.upload(&prepared.store, [11; 16]).await?; + let reader = CatalogReader::open(Arc::clone(&prepared.indexes), stored).await?; + let loader = Loader { + root: prepared.fixture.root.path(), + store: &prepared.store, + budget: DiskBudget::new(128 << 20), + }; + let oid = *prepared.fixture.objects.keys().next().ok_or("oid")?; + assert!(reader.lookup(oid, &loader, &loader).await.is_err()); + assert!(reader.headers(&[oid, oid], &loader, &loader).await.is_err()); + assert_eq!(loader.budget.used(), 0); + // A valid authenticated directory root is not a source-tree node. + let mut wrong = prepared.snapshot; + let root = wrong.sources.as_mut().ok_or("source root")?; + root.artifact = wrong.directory.artifact; + root.operation = wrong.directory.operation; + let stored = wrong.upload(&prepared.store, [12; 16]).await?; + assert!( + CatalogReader::open(Arc::clone(&prepared.indexes), stored) + .await + .is_err() + ); + Ok(()) +} + +#[tokio::test] +async fn catalog_context_summary_and_artifact_corruption_fail_before_reads() -> Result { + let prepared = prepared(ObjectFormat::Sha256).await?; + let mut wrong = prepared.stored; + wrong.format = ObjectFormat::Sha1; + assert!( + CatalogReader::open(Arc::clone(&prepared.indexes), wrong) + .await + .is_err() + ); + let mut wrong = prepared.snapshot; + wrong.directory.format = ObjectFormat::Sha1; + let stored = wrong.upload(&prepared.store, [9; 16]).await?; + assert!( + CatalogReader::open(Arc::clone(&prepared.indexes), stored) + .await + .is_err() + ); + let mut wrong = prepared.snapshot; + wrong.directory.repository = [2; 16]; + assert!(wrong.upload(&prepared.store, [10; 16]).await.is_err()); + let mut wrong = prepared.snapshot; + wrong.sources.as_mut().ok_or("source root")?.object_count += 1; + let stored = wrong.upload(&prepared.store, [11; 16]).await?; + assert!( + CatalogReader::open(Arc::clone(&prepared.indexes), stored) + .await + .is_err() + ); + let path = prepared + .store + .path(prepared.stored.key(), prepared.stored.artifact.digest)?; + prepared + .provider + .put( + &canopy_object_storage::external::part(&path, 0), + bytes::Bytes::from(vec![0; prepared.stored.artifact.size as usize]).into(), + ) + .await?; + assert!( + CatalogReader::open(Arc::clone(&prepared.indexes), prepared.stored) + .await + .is_err() + ); + Ok(()) +} + +mod files; diff --git a/crates/canopy-server/src/packs/catalog/tests/files.rs b/crates/canopy-server/src/packs/catalog/tests/files.rs new file mode 100644 index 0000000..b5df889 --- /dev/null +++ b/crates/canopy-server/src/packs/catalog/tests/files.rs @@ -0,0 +1,438 @@ +use super::*; +use crate::packs::directory::StoredRun; +use std::future::Future; + +fn file_limits(open_files: u32, cached_files: usize) -> CatalogFileLimits { + CatalogFileLimits { + open_files, + cached_files, + metadata: limits(), + } +} +fn files( + prepared: &Prepared, + budget: DiskBudget, + limits: CatalogFileLimits, +) -> Result { + Ok(CatalogFiles::new( + prepared.fixture.root.path(), + budget, + Arc::clone(&prepared.store), + prepared.stored.format, + limits, + )?) +} +async fn run(prepared: &Prepared) -> Result { + let snapshot = + DirectorySnapshot::download(&prepared.store, prepared.snapshot.directory).await?; + let root = *snapshot.level_zero.first().ok_or("root")?; + Ok(prepared + .indexes + .ranges() + .find(Some(root), root.first_key) + .await? + .ok_or("run")?) +} +async fn source(prepared: &Prepared) -> Result { + let run = run(prepared).await?; + let loader = Loader { + root: prepared.fixture.root.path(), + store: &prepared.store, + budget: DiskBudget::new(128 << 20), + }; + let run = RunLoader::load(&loader, run).await?; + let entry = run.entries_after(None)?.into_iter().next().ok_or("entry")?; + Ok( + SourceIndex::new(Arc::clone(&prepared.store), prepared.stored.format) + .find(prepared.snapshot.sources, entry.source) + .await? + .ok_or("source")?, + ) +} + +#[tokio::test] +async fn worker_files_coalesce_native_lookups_and_reuse_cache_across_roots() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let prepared = prepared(format).await?; + let budget = DiskBudget::new(128 << 20); + let files = Arc::new(files(&prepared, budget.clone(), file_limits(4, 2))?); + let reader = + Arc::new(CatalogReader::open(Arc::clone(&prepared.indexes), prepared.stored).await?); + let oid = *prepared.fixture.objects.keys().next().ok_or("oid")?; + let mut jobs = Vec::new(); + for _ in 0..16 { + let files = Arc::clone(&files); + let reader = Arc::clone(&reader); + jobs.push(tokio::spawn(async move { + reader.lookup(oid, &*files, &*files).await + })); + } + let mut resolved = Vec::new(); + for job in jobs { + resolved.push(job.await??.ok_or("resolved")?); + } + for item in &resolved { + assert_eq!(item.entry.header.object, prepared.fixture.objects[&oid].0); + assert!(Arc::ptr_eq( + &item.source.metadata, + &resolved[0].source.metadata + )); + } + let stats = files.stats()?; + assert_eq!(stats.open_files, 2); + assert_eq!(stats.cached_files, 2); + assert_eq!(stats.downloaded_files, 2); + assert!(stats.cache_hits >= 30); + let next = prepared.snapshot.upload(&prepared.store, [71; 16]).await?; + let next = CatalogReader::open(Arc::clone(&prepared.indexes), next).await?; + let retained = next.lookup(oid, &*files, &*files).await?.ok_or("next")?; + assert_eq!(files.stats()?.downloaded_files, 2); + drop(resolved); + let path = retained.source.metadata.path().to_owned(); + let parent = path.parent().ok_or("parent")?.to_owned(); + let size = retained.source.record.metadata.artifact.size; + drop(files); + assert_eq!(budget.used(), size); + assert!(path.exists() && parent.exists()); + assert_eq!( + retained.source.metadata.header(oid)?, + Some(retained.entry.header) + ); + drop(retained); + assert_eq!(budget.used(), 0); + assert!(!path.exists() && !parent.exists()); + } + Ok(()) +} + +#[tokio::test] +async fn borrowed_files_hold_slots_and_eviction_compares_exact_descriptors() -> Result { + let prepared = prepared(ObjectFormat::Sha256).await?; + let budget = DiskBudget::new(128 << 20); + let files = files(&prepared, budget.clone(), file_limits(1, 1))?; + let stored_run = run(&prepared).await?; + let stored_metadata = source(&prepared).await?.metadata; + let borrowed = RunLoader::load(&files, stored_run).await?; + let path = borrowed.path().to_owned(); + assert!(matches!( + SourceLoader::load(&files, stored_metadata).await, + Err(MetadataError::Limit) + )); + assert_eq!(files.stats()?.open_files, 1); + assert_eq!(budget.used(), stored_run.artifact.size); + drop(borrowed); + let metadata = SourceLoader::load(&files, stored_metadata).await?; + assert!(!path.exists()); + assert_eq!(budget.used(), stored_metadata.artifact.size); + let mut forged = stored_metadata; + forged.segment.inventory_digest[0] ^= 1; + assert!(matches!( + SourceLoader::load(&files, forged).await, + Err(MetadataError::Integrity) + )); + let mut foreign = stored_run; + foreign.run.repository[0] ^= 1; + assert!(matches!( + RunLoader::load(&files, foreign).await, + Err(MetadataError::Integrity) + )); + assert_eq!(files.stats()?.downloaded_files, 2); + assert!(Arc::ptr_eq( + &metadata, + &SourceLoader::load(&files, stored_metadata).await? + )); + drop(metadata); + drop(files); + assert_eq!(budget.used(), 0); + Ok(()) +} + +#[tokio::test] +async fn failed_downloads_release_slots_and_cannot_enter_cache() -> Result { + let prepared = prepared(ObjectFormat::Sha1).await?; + let stored = run(&prepared).await?; + let small = DiskBudget::new(stored.artifact.size - 1); + let loader = files(&prepared, small.clone(), file_limits(1, 1))?; + assert!(matches!( + RunLoader::load(&loader, stored).await, + Err(MetadataError::Budget(_)) + )); + assert_eq!(small.used(), 0); + assert_eq!(loader.stats()?.open_files, 0); + assert_eq!(loader.stats()?.cached_files, 0); + let path = prepared.store.path( + ArtifactKey { + operation: stored.run.operation, + binding_digest: stored.run.digest, + kind: ArtifactKind::DirectoryRun, + }, + stored.artifact.digest, + )?; + prepared + .provider + .put( + &canopy_object_storage::external::part(&path, 0), + bytes::Bytes::from(vec![0; stored.artifact.size as usize]).into(), + ) + .await?; + let budget = DiskBudget::new(128 << 20); + let loader = files(&prepared, budget.clone(), file_limits(1, 1))?; + for _ in 0..2 { + assert!(RunLoader::load(&loader, stored).await.is_err()); + assert_eq!(budget.used(), 0); + assert_eq!(loader.stats()?.open_files, 0); + assert_eq!(loader.stats()?.cached_files, 0); + assert_eq!(loader.stats()?.downloaded_files, 0); + } + Ok(()) +} + +#[test] +fn canceled_queued_download_keeps_its_open_slot_and_disk_admission() -> Result { + let runtime = tokio::runtime::Builder::new_multi_thread() + .worker_threads(1) + .max_blocking_threads(1) + .enable_all() + .build()?; + let prepared = runtime.block_on(prepared(ObjectFormat::Sha256))?; + let stored = runtime.block_on(run(&prepared))?; + let budget = DiskBudget::new(128 << 20); + let files = files(&prepared, budget.clone(), file_limits(1, 0))?; + let (ready_tx, ready_rx) = std::sync::mpsc::channel(); + let (release_tx, release_rx) = std::sync::mpsc::channel(); + let blocker = runtime.spawn_blocking(move || { + ready_tx.send(()).unwrap(); + release_rx.recv().unwrap(); + }); + ready_rx.recv_timeout(std::time::Duration::from_secs(5))?; + let mut future = Box::pin(RunLoader::load(&files, stored)); + let pending = { + let _entered = runtime.enter(); + let mut context = std::task::Context::from_waker(std::task::Waker::noop()); + matches!(future.as_mut().poll(&mut context), std::task::Poll::Pending) + }; + drop(future); + let stats = files.stats()?; + let charged = budget.used(); + release_tx.send(())?; + runtime.block_on(async { + blocker.await?; + tokio::time::timeout(std::time::Duration::from_secs(5), async { + while files.stats().unwrap().open_files != 0 || budget.used() != 0 { + tokio::time::sleep(std::time::Duration::from_millis(5)).await; + } + }) + .await?; + Ok::<_, Box>(()) + })?; + assert!(pending); + assert_eq!(stats.open_files, 1); + assert_eq!(stats.cached_files, 0); + assert_eq!(charged, stored.artifact.size); + assert_eq!(files.stats()?.downloaded_files, 0); + Ok(()) +} + +#[test] +fn canceled_queued_sql_lookup_retains_file_and_private_directory_until_worker_exit() -> Result { + let runtime = tokio::runtime::Builder::new_multi_thread() + .worker_threads(1) + .max_blocking_threads(1) + .enable_all() + .build()?; + let prepared = runtime.block_on(prepared(ObjectFormat::Sha1))?; + let stored = runtime.block_on(run(&prepared))?; + let snapshot = runtime.block_on(DirectorySnapshot::download( + &prepared.store, + prepared.snapshot.directory, + ))?; + let budget = DiskBudget::new(128 << 20); + let files = files(&prepared, budget.clone(), file_limits(1, 1))?; + let borrowed = runtime.block_on(RunLoader::load(&files, stored))?; + let weak = Arc::downgrade(&borrowed); + let path = borrowed.path().to_owned(); + let parent = path.parent().ok_or("parent")?.to_owned(); + drop(borrowed); + let ranges = RangeIndex::new(Arc::clone(&prepared.store), ObjectFormat::Sha1); + let oid = *prepared.fixture.objects.keys().next().ok_or("oid")?; + // Isolate the SQL cancellation boundary: cold range-node transfer is a + // distinct async step, while this test must enqueue the file-pinned read. + assert_eq!( + runtime.block_on(snapshot.selected_runs(&ranges, oid))?, + vec![stored] + ); + let (ready_tx, ready_rx) = std::sync::mpsc::channel(); + let (release_tx, release_rx) = std::sync::mpsc::channel(); + let blocker = runtime.spawn_blocking(move || { + ready_tx.send(()).unwrap(); + release_rx.recv().unwrap(); + }); + ready_rx.recv_timeout(std::time::Duration::from_secs(5))?; + let mut future = Box::pin(snapshot.lookup(&ranges, &files, oid)); + let pending = { + let _entered = runtime.enter(); + let mut context = std::task::Context::from_waker(std::task::Waker::noop()); + matches!(future.as_mut().poll(&mut context), std::task::Poll::Pending) + }; + drop(future); + drop(files); + let charged = budget.used(); + let pinned = weak.upgrade().is_some() && path.exists() && parent.exists(); + release_tx.send(())?; + runtime.block_on(async { + blocker.await?; + tokio::time::timeout(std::time::Duration::from_secs(5), async { + while budget.used() != 0 || path.exists() || parent.exists() || weak.upgrade().is_some() + { + tokio::time::sleep(std::time::Duration::from_millis(5)).await; + } + }) + .await?; + Ok::<_, Box>(()) + })?; + assert!(pending && pinned); + assert_eq!(charged, stored.artifact.size); + assert!(weak.upgrade().is_none()); + assert!(!path.exists() && !parent.exists()); + Ok(()) +} + +struct CountingFiles { + files: CatalogFiles, + runs: std::sync::atomic::AtomicUsize, + metadata: std::sync::atomic::AtomicUsize, +} +impl RunLoader for CountingFiles { + async fn load( + &self, + stored: StoredRun, + ) -> std::result::Result, MetadataError> { + self.runs.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + RunLoader::load(&self.files, stored).await + } +} +impl SourceLoader for CountingFiles { + async fn load( + &self, + stored: StoredSegment, + ) -> std::result::Result, MetadataError> { + self.metadata + .fetch_add(1, std::sync::atomic::Ordering::Relaxed); + SourceLoader::load(&self.files, stored).await + } +} +#[tokio::test] +async fn canonical_header_batches_group_files_and_preserve_request_order() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let prepared = prepared(format).await?; + let budget = DiskBudget::new(128 << 20); + // A single slot suffices: header batches release each group before + // opening the next file, rather than pinning a file per returned OID. + let files = CountingFiles { + files: files(&prepared, budget.clone(), file_limits(1, 1))?, + runs: std::sync::atomic::AtomicUsize::new(0), + metadata: std::sync::atomic::AtomicUsize::new(0), + }; + let reader = CatalogReader::open(Arc::clone(&prepared.indexes), prepared.stored).await?; + let native: Vec<_> = prepared.fixture.objects.keys().rev().copied().collect(); + let ids: Vec<_> = (0..512) + .map(|n| { + if n % 17 == 0 { + format.zero() + } else if n % 23 == 0 { + if format == ObjectFormat::Sha1 { + ObjectFormat::Sha256.zero() + } else { + ObjectFormat::Sha1.zero() + } + } else { + native[n % native.len()] + } + }) + .collect(); + let headers = reader.headers(&ids, &files, &files).await?; + assert_eq!(headers.len(), ids.len()); + for (oid, actual) in ids.iter().zip(&headers) { + if let Some((expected, _)) = prepared.fixture.objects.get(oid) { + assert_eq!(actual.ok_or("header")?.object, *expected); + } else { + assert!(actual.is_none()); + } + } + assert_eq!(files.runs.load(std::sync::atomic::Ordering::Relaxed), 1); + assert_eq!(files.metadata.load(std::sync::atomic::Ordering::Relaxed), 1); + assert_eq!(files.files.stats()?.open_files, 1); + assert_eq!(files.files.stats()?.downloaded_files, 2); + assert!(reader.headers(&[], &files, &files).await?.is_empty()); + assert!(matches!( + reader.headers(&vec![native[0]; 513], &files, &files).await, + Err(IndexError::Limit) + )); + let stored = source(&prepared).await?.metadata; + let metadata = SourceLoader::load(&files.files, stored).await?; + assert!(matches!( + metadata.headers(&vec![native[0]; 513]), + Err(MetadataError::Limit) + )); + drop(metadata); + let stored = run(&prepared).await?; + let run = RunLoader::load(&files.files, stored).await?; + assert!(matches!( + run.find_batch(&vec![native[0]; 513]), + Err(MetadataError::Limit) + )); + drop(run); + drop(files); + assert_eq!(budget.used(), 0); + } + Ok(()) +} + +#[tokio::test] +async fn authenticated_directory_bytes_cannot_hide_a_late_source_header_conflict() -> Result { + let prepared = prepared(ObjectFormat::Sha256).await?; + let loader = Loader { + root: prepared.fixture.root.path(), + store: &prepared.store, + budget: DiskBudget::new(128 << 20), + }; + let original = RunLoader::load(&loader, run(&prepared).await?).await?; + let mut entries = original.entries_after(None)?; + entries.last_mut().ok_or("entry")?.header.object.digest[0] ^= 1; + let bad = Arc::new(crate::packs::directory::tests::inconsistent_run( + prepared.fixture.root.path(), + prepared.stored.repository, + [75; 16], + prepared.stored.format, + &entries, + )?) + .upload(&prepared.store) + .await?; + let mut snapshot = DirectorySnapshot::empty(prepared.stored.repository, prepared.stored.format); + let root = prepared + .indexes + .ranges() + .insert(None, [75; 16], bad) + .await?; + snapshot.append(prepared.indexes.ranges(), root).await?; + let catalog = CatalogSnapshot { + directory: snapshot.upload(&prepared.store, [76; 16]).await?, + sources: prepared.snapshot.sources, + } + .upload(&prepared.store, [77; 16]) + .await?; + let reader = CatalogReader::open(Arc::clone(&prepared.indexes), catalog).await?; + let budget = DiskBudget::new(128 << 20); + let files = files(&prepared, budget.clone(), file_limits(1, 1))?; + let ids: Vec<_> = entries + .iter() + .map(|entry| entry.header.object.oid) + .collect(); + assert!(matches!( + reader.headers(&ids, &files, &files).await, + Err(IndexError::Metadata(MetadataError::IdentityConflict)) + )); + drop(files); + assert_eq!(budget.used(), 0); + Ok(()) +} diff --git a/crates/canopy-server/src/packs/closure/copy.rs b/crates/canopy-server/src/packs/closure/copy.rs new file mode 100644 index 0000000..d500412 --- /dev/null +++ b/crates/canopy-server/src/packs/closure/copy.rs @@ -0,0 +1,149 @@ +use super::*; +use spool::validate_header; + +impl Spool { + pub(super) fn copy_segment( + &mut self, + segment: &MetadataSegment, + operation: [u8; 16], + ) -> Result<(), ClosureError> { + self.check_cancel()?; + let descriptor = segment.descriptor(); + let identity = descriptor.identity; + if identity.repository != self.context.repository + || identity.operation != operation + || identity.format != self.context.format + || metadata::file_digest_with::(segment.path(), descriptor.size, || { + self.check_cancel() + })? != descriptor.digest + { + return Err(ClosureError::Integrity); + } + let mut after = None; + let mut count = 0_u64; + let mut edge_count = 0_u64; + let mut first = None; + let mut inventory = metadata::inventory_seed(identity); + loop { + self.check_cancel()?; + let headers = segment.headers_after(after)?; + if headers.is_empty() { + break; + } + let canceled = Arc::clone(&self.canceled); + let (next_first, next_after, next_inventory, next_count, next_edges) = self.write(|tx| { + let mut first = first; + let mut after = after; + let mut inventory = inventory; + let mut count = count; + let mut edge_count = edge_count; + for &header in &headers { + validate_header(header, identity.format)?; + let h = header.object; + tx.execute("INSERT INTO objects(oid,kind,size,digest,edge_count,edge_digest) VALUES(?1,?2,?3,?4,?5,?6) ON CONFLICT DO NOTHING", params![h.oid.as_ref(),h.kind.git_name(),h.size as i64,h.digest.as_slice(),header.edge_count as i64,header.edge_digest.as_slice()])?; + let current = tx.query_row( + "SELECT oid,kind,size,digest,edge_count,edge_digest FROM objects WHERE oid=?1", + [h.oid.as_ref()], + metadata::header, + )?; + if current != header { + return Err(MetadataError::IdentityConflict.into()); + } + verify_edges(tx, segment, header, &canceled)?; + first.get_or_insert(h.oid); + after = Some(h.oid); + inventory = metadata::fold_header(inventory, count, header); + count = count.checked_add(1).ok_or(MetadataError::Limit)?; + edge_count = edge_count + .checked_add(header.edge_count) + .ok_or(MetadataError::Limit)?; + } + Ok((first, after, inventory, count, edge_count)) + })?; + first = next_first; + after = next_after; + inventory = next_inventory; + count = next_count; + edge_count = next_edges; + } + if count != u64::from(identity.object_count) + || edge_count != descriptor.edge_count + || first != Some(descriptor.first_oid) + || after != Some(descriptor.last_oid) + || inventory != descriptor.inventory_digest + { + return Err(ClosureError::Integrity); + } + Ok(()) + } +} +fn verify_edges( + tx: &rusqlite::Transaction<'_>, + segment: &MetadataSegment, + header: ObjectHeader, + canceled: &AtomicBool, +) -> Result<(), ClosureError> { + let parent = header.object; + let mut after = None; + let mut count = 0_u64; + let mut trees = 0; + let mut chain = metadata::edge_seed(parent.oid); + let mut insert = + tx.prepare_cached("INSERT INTO object_edges VALUES(?1,?2,?3) ON CONFLICT DO NOTHING")?; + let mut existing = + tx.prepare_cached("SELECT expected_kind FROM object_edges WHERE parent=?1 AND child=?2")?; + loop { + if canceled.load(Ordering::Acquire) { + return Err(ClosureError::Canceled); + } + let page = segment.edges_after(parent.oid, after)?; + if page.is_empty() { + break; + } + for edge in page { + if edge.child.format() != parent.oid.format() + || edge.child.is_zero() + || after.is_some_and(|oid| oid >= edge.child) + || match parent.kind { + ObjectKind::Blob => true, + ObjectKind::Tree => { + !matches!(edge.expected_kind, ObjectKind::Tree | ObjectKind::Blob) + } + ObjectKind::Commit => { + !matches!(edge.expected_kind, ObjectKind::Tree | ObjectKind::Commit) + } + ObjectKind::Tag => false, + } + { + return Err(ClosureError::Integrity); + } + insert.execute(params![ + parent.oid.as_ref(), + edge.child.as_ref(), + edge.expected_kind.git_name() + ])?; + if existing.query_row(params![parent.oid.as_ref(), edge.child.as_ref()], |row| { + metadata::kind(&row.get::<_, String>(0)?) + })? != edge.expected_kind + { + return Err(MetadataError::IdentityConflict.into()); + } + let mut record = [0; 33]; + let width = edge.child.len(); + record[..width].copy_from_slice(&edge.child); + record[width] = metadata::kind_code(edge.expected_kind); + chain = metadata::fold(chain, count, &record[..width + 1]); + count = count.checked_add(1).ok_or(MetadataError::Limit)?; + trees += u64::from(edge.expected_kind == ObjectKind::Tree); + after = Some(edge.child); + } + } + if count != header.edge_count + || chain != header.edge_digest + || (parent.kind == ObjectKind::Tag && count != 1) + || (parent.kind == ObjectKind::Commit && trees != 1) + { + return Err(ClosureError::Integrity); + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/closure/graph.rs b/crates/canopy-server/src/packs/closure/graph.rs new file mode 100644 index 0000000..7bd1eed --- /dev/null +++ b/crates/canopy-server/src/packs/closure/graph.rs @@ -0,0 +1,127 @@ +use super::*; + +impl Spool { + pub(super) fn certify_graph(&mut self) -> Result<(), ClosureError> { + self.check_cancel()?; + let wrong = self.connection.query_row("SELECT e.child,e.expected_kind,COALESCE(o.kind,b.kind) FROM object_edges e LEFT JOIN objects o ON o.oid=e.child LEFT JOIN base_objects b ON b.oid=e.child WHERE COALESCE(o.kind,b.kind) IS NULL OR e.expected_kind != COALESCE(o.kind,b.kind) LIMIT 1", [], |row| { + Ok((metadata::oid(row.get(0)?)?,metadata::kind(&row.get::<_,String>(1)?)?,row.get::<_,Option>(2)?)) + }).optional()?; + if let Some((oid, expected, kind)) = wrong { + return match kind { + Some(kind) => Err(ClosureError::Kind { + oid, + expected, + actual: metadata::kind(&kind)?, + }), + None => Err(ClosureError::Missing(oid)), + }; + } + // Only incoming vertices need topological processing. Certified external + // dependencies are anchors; historical graph edges never enter scratch. + let mut after = Vec::new(); + loop { + self.check_cancel()?; + let ids: Vec> = self + .connection + .prepare_cached("SELECT oid FROM objects WHERE oid>?1 ORDER BY oid LIMIT ?2")? + .query_map(params![after, PAGE_OBJECTS as i64], |r| r.get(0))? + .collect::>()?; + if ids.is_empty() { + break; + } + self.write(|tx| { + let mut pending = tx.prepare_cached("UPDATE objects SET pending=(SELECT count(*) FROM object_edges e JOIN objects c ON c.oid=e.child WHERE e.parent=objects.oid) WHERE oid=?1")?; + for oid in &ids { pending.execute([oid])?; } + Ok(()) + })?; + after = ids.last().ok_or(ClosureError::Integrity)?.clone(); + } + // Reuse the partial ready index as a disk-backed queue. A transaction + // performs at most 512 vertex/edge updates, including newly-ready + // vertices, so long chains do not create a journal per vertex. One wide + // reverse fanout may span transactions; its cursor has constant size. + let mut active = None; + loop { + self.check_cancel()?; + let canceled = Arc::clone(&self.canceled); + let (exhausted, next) = self.write(|tx| { + let mut next = active.clone(); + let exhausted = advance(tx, &mut next, &canceled)?; + Ok((exhausted, next)) + })?; + active = next; + if exhausted { + break; + } + } + let unfinished: bool = self.connection.query_row( + "SELECT EXISTS(SELECT 1 FROM objects WHERE done=0)", + [], + |row| row.get(0), + )?; + if unfinished { + return Err(ClosureError::Cycle); + } + Ok(()) + } +} + +fn advance( + tx: &rusqlite::Transaction<'_>, + active: &mut Option<(ObjectId, Vec)>, + canceled: &AtomicBool, +) -> Result { + let mut updates = 0; + let mut ready = tx.prepare_cached( + "SELECT oid FROM objects WHERE pending=0 AND done=0 ORDER BY oid LIMIT 1", + )?; + let mut complete = + tx.prepare_cached("UPDATE objects SET done=1 WHERE oid=?1 AND pending=0 AND done=0")?; + let mut reverse = tx.prepare_cached( + "SELECT parent FROM object_edges WHERE child=?1 AND parent>?2 ORDER BY parent LIMIT ?3", + )?; + let mut decrement = tx.prepare_cached( + "UPDATE objects SET pending=pending-1 WHERE oid=?1 AND pending>0 AND done=0", + )?; + while updates < PAGE_OBJECTS { + if canceled.load(Ordering::Acquire) { + return Err(ClosureError::Canceled); + } + if active.is_none() { + let Some(child) = ready + .query_row([], |row| metadata::oid(row.get(0)?)) + .optional()? + else { + return Ok(true); + }; + if complete.execute([child.as_ref()])? != 1 { + return Err(ClosureError::Integrity); + } + *active = Some((child, Vec::new())); + updates += 1; + if updates == PAGE_OBJECTS { + break; + } + } + let (child, after) = active.as_mut().ok_or(ClosureError::Integrity)?; + let limit = PAGE_OBJECTS - updates; + let parents: Vec = reverse + .query_map( + params![child.as_ref(), after.as_slice(), limit as i64], + |row| metadata::oid(row.get(0)?), + )? + .collect::>()?; + let exhausted = parents.len() < limit; + for parent in parents { + if decrement.execute([parent.as_ref()])? != 1 { + return Err(ClosureError::Integrity); + } + *after = parent.to_vec(); + updates += 1; + } + if exhausted { + *active = None; + } + } + Ok(false) +} diff --git a/crates/canopy-server/src/packs/closure/mod.rs b/crates/canopy-server/src/packs/closure/mod.rs new file mode 100644 index 0000000..1399199 --- /dev/null +++ b/crates/canopy-server/src/packs/closure/mod.rs @@ -0,0 +1,145 @@ +//! Admitted operation-local graph closure. Historical graph rows stay in +//! immutable metadata; only incoming objects and queried base headers enter +//! this disposable spool. A witness is conditional on a trusted certified base +//! lookup and is not an authorization, owner fence or publication certificate. +use super::{ + catalog::StoredCatalog, + directory::{self, RunDescriptor}, + metadata::{ + self, AdmittedFile, MetadataError, MetadataLimits, MetadataSegment, ObjectHeader, + PAGE_OBJECTS, + }, + verification::{PhysicalError, PhysicalPackWitness, PhysicalPartition}, +}; +use crate::{ObjectFormat, ObjectId, ObjectKind}; +use cellule_ltx::DiskBudget; +use rusqlite::{Connection, OptionalExtension, params}; +use std::{ + future::Future, + path::Path, + sync::{ + Arc, Mutex, + atomic::{AtomicBool, Ordering}, + }, +}; + +mod copy; +mod graph; +mod retained; +mod spool; +mod verifier; +mod witness; +pub(in crate::packs) use retained::RetainedClosure; +use spool::Spool; +pub use verifier::ClosureVerifier; +pub use witness::ClosureWitness; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ClosureBase { + pub catalog: StoredCatalog, + pub generation: u64, +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ClosureContext { + pub repository: [u8; 16], + pub operation: [u8; 16], + pub format: ObjectFormat, + pub base: Option, +} +impl ClosureContext { + fn validate(self) -> Result<(), ClosureError> { + if let Some(base) = self.base { + base.catalog.validate()?; + if base.catalog.repository != self.repository + || base.catalog.format != self.format + || base.generation == 0 + || base.generation > i64::MAX as u64 + { + return Err(ClosureError::Integrity); + } + } + Ok(()) + } +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct BaseObject { + pub header: ObjectHeader, + /// Must come from authoritative closure facts for the selected generation. + /// Presence in a raw catalog or physical index cannot set this to true. + pub certified: bool, +} +pub struct BaseBatch { + pub base: ClosureBase, + /// One result per requested OID in exactly the same order, at most 512. + pub objects: Vec>, +} +/// Trusted service boundary. Retain the selected generation lease throughout +/// verification and derive certification from authoritative facts. A raw +/// CatalogReader alone cannot implement this contract. Publication must CAS +/// the same base and revalidate overlaps if it changes. +pub trait BaseResolver: Sync { + fn resolve( + &self, + base: ClosureBase, + ids: &[ObjectId], + ) -> impl Future> + Send; +} +#[derive(Debug, thiserror::Error)] +pub enum ClosureError { + #[error("closure metadata failed")] + Metadata(#[from] MetadataError), + #[error("closure physical input failed")] + Physical(#[from] PhysicalError), + #[error("closure catalog lookup failed")] + Catalog(#[from] super::directory::index::IndexError), + #[error("closure worker failed")] + Task(#[from] tokio::task::JoinError), + #[error("closure input or generation binding is inconsistent")] + Integrity, + #[error("dependency {0:?} is missing")] + Missing(ObjectId), + #[error("dependency {oid:?} requires {expected:?}, found {actual:?}")] + Kind { + oid: ObjectId, + expected: ObjectKind, + actual: ObjectKind, + }, + #[error("base object {0:?} has no closure certificate")] + Uncertified(ObjectId), + #[error("incoming dependencies contain a cycle")] + Cycle, + #[error("certified base generation lease expired")] + LeaseExpired, + #[error("closure verification was canceled")] + Canceled, +} +impl From for ClosureError { + fn from(error: rusqlite::Error) -> Self { + Self::Metadata(error.into()) + } +} +struct CancelGuard { + canceled: Arc, + complete: bool, +} +impl CancelGuard { + fn new(canceled: Arc) -> Self { + Self { + canceled, + complete: false, + } + } + fn complete(&mut self) { + self.complete = true; + } +} +impl Drop for CancelGuard { + fn drop(&mut self) { + if !self.complete { + self.canceled.store(true, Ordering::Release); + } + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/canopy-server/src/packs/closure/retained.rs b/crates/canopy-server/src/packs/closure/retained.rs new file mode 100644 index 0000000..3f8e17e --- /dev/null +++ b/crates/canopy-server/src/packs/closure/retained.rs @@ -0,0 +1,290 @@ +use super::*; + +const RECONCILE_PAGE: &str = "SELECT l.oid,COALESCE(o.kind,b.kind),COALESCE(o.size,b.size),COALESCE(o.digest,b.digest),COALESCE(o.edge_count,b.edge_count),COALESCE(o.edge_digest,b.edge_digest),o.oid IS NOT NULL FROM lookups l LEFT JOIN objects o ON o.oid=l.oid LEFT JOIN base_objects b ON b.oid=l.oid WHERE l.oid>?1 ORDER BY l.oid LIMIT ?2"; + +/// Constructed only after incoming physical partitions, canonical overlap and +/// topology checks finish. Retains the already admitted scratch; no heap OID set +/// or historical graph is copied. Queued workers keep that scratch admitted. +pub(in crate::packs) struct RetainedClosure { + pub(super) spool: Arc>, + pub(super) context: ClosureContext, + pub(super) canceled: Arc, +} +impl RetainedClosure { + pub(in crate::packs) async fn reconcile( + &self, + context: ClosureContext, + resolver: &impl BaseResolver, + ) -> Result<(), ClosureError> { + context.validate()?; + if context.repository != self.context.repository + || context.operation != self.context.operation + || context.format != self.context.format + || context.base.map_or(0, |base| base.generation) + < self.context.base.map_or(0, |base| base.generation) + || (context.base.map_or(0, |base| base.generation) + == self.context.base.map_or(0, |base| base.generation) + && context.base != self.context.base) + { + return Err(ClosureError::Integrity); + } + if context == self.context { + return Ok(()); + } + let base = context.base.ok_or(ClosureError::Integrity)?; + let canceled = Arc::new(AtomicBool::new(false)); + let mut guard = CancelGuard::new(Arc::clone(&canceled)); + let mut after = Vec::new(); + loop { + let spool = Arc::clone(&self.spool); + let token = Arc::clone(&canceled); + let cursor = after.clone(); + let page = tokio::task::spawn_blocking(move || { + if token.load(Ordering::Acquire) { + return Err(ClosureError::Canceled); + } + let spool = spool.lock().map_err(|_| ClosureError::Integrity)?; + spool.check_cancel()?; + let result = spool + .connection + .prepare_cached(RECONCILE_PAGE)? + .query_map(params![cursor, PAGE_OBJECTS as i64], |row| { + Ok((metadata::header(row)?, row.get::<_, bool>(6)?)) + })? + .collect::>>()?; + if token.load(Ordering::Acquire) { + return Err(ClosureError::Canceled); + } + Ok::<_, ClosureError>(result) + }) + .await??; + if page.is_empty() { + break; + } + let ids = page + .iter() + .map(|(header, _)| header.object.oid) + .collect::>(); + let found = resolver.resolve(base, &ids).await?; + if found.base != base || found.objects.len() != page.len() { + return Err(ClosureError::Integrity); + } + for ((expected, incoming), actual) in page.into_iter().zip(found.objects) { + match actual { + Some(actual) if !actual.certified => { + return Err(ClosureError::Uncertified(expected.object.oid)); + } + Some(actual) if actual.header != expected => { + return Err(MetadataError::IdentityConflict.into()); + } + None if !incoming => return Err(ClosureError::Missing(expected.object.oid)), + _ => {} + } + after = expected.object.oid.to_vec(); + } + } + // Every external anchor has the identical canonical body and graph + // digest in a certified new base. The already verified incoming DAG can + // therefore be reused; no topological processing of history is needed. + guard.complete(); + Ok(()) + } +} + +impl Drop for RetainedClosure { + fn drop(&mut self) { + self.canceled.store(true, Ordering::Release); + } +} + +#[cfg(test)] +mod tests { + use super::*; + type Result = std::result::Result>; + struct Empty; + impl BaseResolver for Empty { + async fn resolve( + &self, + base: ClosureBase, + ids: &[ObjectId], + ) -> std::result::Result { + Ok(BaseBatch { + base, + objects: vec![None; ids.len()], + }) + } + } + #[test] + fn reconciliation_pages_use_indexed_incoming_and_anchor_lookups() -> Result { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let context = ClosureContext { + repository: [1; 16], + operation: [2; 16], + format: ObjectFormat::Sha256, + base: None, + }; + let mut spool = Spool::new( + root.path(), + budget.clone(), + context, + metadata::tests::limits(), + Arc::new(AtomicBool::new(false)), + )?; + { + spool.write(|tx| { + for n in 1u64..=10000 { + let mut oid = [0; 32]; + oid[24..].copy_from_slice(&n.to_be_bytes()); + tx.execute("INSERT INTO lookups VALUES(?1,1)", [oid.as_slice()])?; + if n.is_multiple_of(2) { + tx.execute("INSERT INTO objects(oid,kind,size,digest,edge_count,edge_digest,done) VALUES(?1,'blob',0,zeroblob(32),0,zeroblob(32),1)",[oid.as_slice()])?; + } else { + tx.execute( + "INSERT INTO base_objects VALUES(?1,'blob',0,zeroblob(32),0,zeroblob(32))", + [oid.as_slice()], + )?; + } + } + Ok(()) + })?; + } + let mut after = [0; 32]; + after[24..].copy_from_slice(&9000u64.to_be_bytes()); + let explain = format!("EXPLAIN QUERY PLAN {RECONCILE_PAGE}"); + let plan = spool + .connection + .prepare(&explain)? + .query_map(params![after.as_slice(), PAGE_OBJECTS as i64], |row| { + row.get::<_, String>(3) + })? + .collect::>>()?; + for alias in ["l", "o", "b"] { + assert!( + plan.iter() + .any(|line| line.contains(&format!("SEARCH {alias} USING PRIMARY KEY"))), + "{plan:?}" + ); + assert!( + !plan + .iter() + .any(|line| line.contains(&format!("SCAN {alias}"))), + "{plan:?}" + ); + } + let page = spool + .connection + .prepare(RECONCILE_PAGE)? + .query_map(params![after.as_slice(), PAGE_OBJECTS as i64], |row| { + Ok((metadata::header(row)?, row.get::<_, bool>(6)?)) + })? + .collect::>>()?; + assert_eq!(page.len(), PAGE_OBJECTS); + for (at, (header, incoming)) in page.into_iter().enumerate() { + let n = 9001 + at as u64; + assert_eq!(&header.object.oid[24..], &n.to_be_bytes()); + assert_eq!(incoming, n.is_multiple_of(2)); + } + drop(spool); + assert_eq!(budget.used(), 0); + assert_eq!(std::fs::read_dir(root.path())?.count(), 0); + Ok(()) + } + #[test] + fn canceled_reconciliation_keeps_admission_until_queued_reader_drains() -> Result { + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .max_blocking_threads(1) + .build()?; + runtime.block_on(async { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let context = ClosureContext { + repository: [1; 16], + operation: [2; 16], + format: ObjectFormat::Sha1, + base: None, + }; + let verifier = ClosureVerifier::new( + root.path(), + budget.clone(), + context, + metadata::tests::limits(), + ) + .await?; + let (_, retained) = verifier.finish_retained(None::<&Empty>).await?; + let stored = StoredCatalog { + repository: context.repository, + operation: [3; 16], + format: context.format, + artifact: canopy_object_storage::artifact::ArtifactDescriptor { + size: 1, + digest: [4; 32], + manifest_digest: [5; 32], + }, + }; + let selected = ClosureContext { + base: Some(ClosureBase { + generation: 1, + catalog: stored, + }), + ..context + }; + let (release, receiver) = std::sync::mpsc::channel(); + let started = Arc::new(tokio::sync::Notify::new()); + let notify = Arc::clone(&started); + let worker = tokio::task::spawn_blocking(move || { + notify.notify_one(); + receiver.recv().unwrap(); + }); + started.notified().await; + let mut pending = Box::pin(retained.reconcile(selected, &Empty)); + assert!( + tokio::time::timeout(std::time::Duration::from_millis(25), &mut pending) + .await + .is_err() + ); + drop(pending); + // Cancellation of one attempt does not poison the private original + // inventory. It may be retried after its queued reader drains. + assert!(!retained.canceled.load(Ordering::Acquire)); + let held = budget.used(); + assert_eq!(held, metadata::growth::INITIAL_BYTES * 3); + release.send(())?; + worker.await?; + retained.reconcile(selected, &Empty).await?; + let weak = Arc::downgrade(&retained.spool); + let (release, receiver) = std::sync::mpsc::channel(); + let started = Arc::new(tokio::sync::Notify::new()); + let notify = Arc::clone(&started); + let worker = tokio::task::spawn_blocking(move || { + notify.notify_one(); + receiver.recv().unwrap(); + }); + started.notified().await; + let mut pending = Box::pin(retained.reconcile(selected, &Empty)); + assert!( + tokio::time::timeout(std::time::Duration::from_millis(25), &mut pending) + .await + .is_err() + ); + drop(pending); + drop(retained); + // Now no inventory owner remains. The queued canceled worker alone + // must keep SQLite, its file and the reservation alive until drain. + assert!(weak.upgrade().is_some()); + assert_eq!(budget.used(), held); + release.send(())?; + worker.await?; + tokio::time::timeout(std::time::Duration::from_secs(5), async { + while budget.used() != 0 { + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + }) + .await?; + assert!(weak.upgrade().is_none()); + assert_eq!(std::fs::read_dir(root.path())?.count(), 0); + Ok::<_, Box>(()) + }) + } +} diff --git a/crates/canopy-server/src/packs/closure/schema.sql b/crates/canopy-server/src/packs/closure/schema.sql new file mode 100644 index 0000000..c67d217 --- /dev/null +++ b/crates/canopy-server/src/packs/closure/schema.sql @@ -0,0 +1,34 @@ +CREATE TABLE inputs ( + digest BLOB PRIMARY KEY CHECK(length(digest)=32) +) WITHOUT ROWID; +CREATE TABLE objects ( + oid BLOB PRIMARY KEY CHECK(length(oid) IN (20,32)), + kind TEXT NOT NULL CHECK(kind IN ('blob','tree','commit','tag')), + size INTEGER NOT NULL CHECK(size>=0), + digest BLOB NOT NULL CHECK(length(digest)=32), + edge_count INTEGER NOT NULL CHECK(edge_count>=0), + edge_digest BLOB NOT NULL CHECK(length(edge_digest)=32), + pending INTEGER NOT NULL DEFAULT 0 CHECK(pending>=0), + done INTEGER NOT NULL DEFAULT 0 CHECK(done IN (0,1)) +) WITHOUT ROWID; +CREATE TABLE object_edges ( + parent BLOB NOT NULL REFERENCES objects(oid), + child BLOB NOT NULL CHECK(length(child) IN (20,32)), + expected_kind TEXT NOT NULL CHECK(expected_kind IN ('blob','tree','commit','tag')), + PRIMARY KEY(parent,child) +) WITHOUT ROWID; +CREATE INDEX edge_child ON object_edges(child,parent); +CREATE INDEX ready_objects ON objects(oid) WHERE pending=0 AND done=0; +CREATE TABLE lookups ( + oid BLOB PRIMARY KEY CHECK(length(oid) IN (20,32)), + resolved INTEGER NOT NULL DEFAULT 0 CHECK(resolved IN (0,1)) +) WITHOUT ROWID; +CREATE INDEX unresolved_lookups ON lookups(oid) WHERE resolved=0; +CREATE TABLE base_objects ( + oid BLOB PRIMARY KEY CHECK(length(oid) IN (20,32)), + kind TEXT NOT NULL CHECK(kind IN ('blob','tree','commit','tag')), + size INTEGER NOT NULL CHECK(size>=0), + digest BLOB NOT NULL CHECK(length(digest)=32), + edge_count INTEGER NOT NULL CHECK(edge_count>=0), + edge_digest BLOB NOT NULL CHECK(length(edge_digest)=32) +) WITHOUT ROWID; diff --git a/crates/canopy-server/src/packs/closure/spool.rs b/crates/canopy-server/src/packs/closure/spool.rs new file mode 100644 index 0000000..7c386eb --- /dev/null +++ b/crates/canopy-server/src/packs/closure/spool.rs @@ -0,0 +1,203 @@ +use super::*; + +pub(super) struct Spool { + // SQLite closes before the journal/file are deleted and admission released. + pub(super) connection: Connection, + pub(super) _admitted: AdmittedFile, + pub(super) context: ClosureContext, + pub(super) canceled: Arc, + limits: MetadataLimits, +} +impl Spool { + pub(super) fn new( + root: &Path, + budget: DiskBudget, + context: ClosureContext, + limits: MetadataLimits, + canceled: Arc, + ) -> Result { + if canceled.load(Ordering::Acquire) { + return Err(ClosureError::Canceled); + } + context.validate()?; + if limits.cache_kib == 0 + || limits.cache_kib > i32::MAX as u32 + || limits.max_file_bytes < 16 << 10 + || limits.max_file_bytes > canopy_object_storage::external::MAX_ARTIFACT_BYTES + || !limits.max_file_bytes.is_multiple_of(4096) + { + return Err(MetadataError::Limit.into()); + } + let reservation = metadata::growth::reserve(&budget, limits.max_file_bytes)?; + let file = tempfile::Builder::new() + .prefix("canopy-closure-") + .tempfile_in(root) + .map_err(MetadataError::from)?; + let mut admitted = AdmittedFile::new(file, reservation); + let connection = Connection::open(admitted.file().path())?; + // No scratch state is an acknowledgement or recovery root. After a + // crash, rebuild it from authenticated inputs; never reopen this file. + connection.execute_batch("PRAGMA page_size=4096; PRAGMA journal_mode=DELETE; PRAGMA synchronous=OFF; PRAGMA foreign_keys=ON; PRAGMA trusted_schema=OFF; PRAGMA mmap_size=0;")?; + connection.pragma_update(None, "cache_size", -(limits.cache_kib as i64))?; + metadata::growth::configure(&connection, &mut admitted)?; + let interrupt = Arc::clone(&canceled); + connection.progress_handler(10_000, Some(move || interrupt.load(Ordering::Acquire))); + let mut spool = Self { + connection, + _admitted: admitted, + context, + canceled, + limits, + }; + spool.check_cancel()?; + spool.write(|tx| { + tx.execute_batch(include_str!("schema.sql"))?; + Ok(()) + })?; + Ok(spool) + } + pub(super) fn write( + &mut self, + mut body: impl FnMut(&rusqlite::Transaction<'_>) -> Result, + ) -> Result { + self.check_cancel()?; + let canceled = Arc::clone(&self.canceled); + metadata::growth::transaction( + &mut self.connection, + &mut self._admitted, + self.limits.max_file_bytes, + |tx| { + if canceled.load(Ordering::Acquire) { + return Err(ClosureError::Canceled); + } + let result = body(tx)?; + if canceled.load(Ordering::Acquire) { + return Err(ClosureError::Canceled); + } + Ok(result) + }, + ) + } + pub(super) fn check_cancel(&self) -> Result<(), ClosureError> { + if self.canceled.load(Ordering::Acquire) { + Err(ClosureError::Canceled) + } else { + Ok(()) + } + } + pub(super) fn input(&mut self, digest: [u8; 32]) -> Result<(), ClosureError> { + self.check_cancel()?; + self.write(|tx| { + tx.execute( + "INSERT INTO inputs VALUES(?1) ON CONFLICT DO NOTHING", + [digest.as_slice()], + )?; + Ok(()) + }) + } + pub(super) fn prepare_lookups(&mut self) -> Result<(), ClosureError> { + // Both source scans use their OID indexes; keep journals and retries + // page-sized even for a full-history import. + for sql in [ + "SELECT oid FROM objects WHERE oid>?1 ORDER BY oid LIMIT ?2", + "SELECT DISTINCT child FROM object_edges WHERE child>?1 ORDER BY child LIMIT ?2", + ] { + let mut after = Vec::new(); + loop { + self.check_cancel()?; + let ids: Vec> = self + .connection + .prepare_cached(sql)? + .query_map(params![after, PAGE_OBJECTS as i64], |r| r.get(0))? + .collect::>()?; + if ids.is_empty() { + break; + } + self.write(|tx| { + let mut insert = + tx.prepare_cached("INSERT OR IGNORE INTO lookups(oid) VALUES(?1)")?; + for oid in &ids { + insert.execute([oid])?; + } + Ok(()) + })?; + after = ids.last().ok_or(ClosureError::Integrity)?.clone(); + } + } + Ok(()) + } + pub(super) fn lookup_page(&self) -> Result, ClosureError> { + self.check_cancel()?; + Ok(self + .connection + .prepare_cached("SELECT oid FROM lookups WHERE resolved=0 ORDER BY oid LIMIT ?1")? + .query_map([PAGE_OBJECTS as i64], |row| metadata::oid(row.get(0)?))? + .collect::>()?) + } + pub(super) fn apply_base( + &mut self, + ids: &[ObjectId], + batch: BaseBatch, + ) -> Result<(), ClosureError> { + self.check_cancel()?; + if self.context.base != Some(batch.base) + || ids.is_empty() + || ids.len() > PAGE_OBJECTS + || batch.objects.len() != ids.len() + { + return Err(ClosureError::Integrity); + } + let format = self.context.format; + self.write(|tx| { + for (oid, found) in ids.iter().zip(&batch.objects) { + if let Some(found) = found { + if found.header.object.oid != *oid { + return Err(ClosureError::Integrity); + } + validate_header(found.header, format)?; + let local = tx.query_row("SELECT oid,kind,size,digest,edge_count,edge_digest FROM objects WHERE oid=?1", [oid.as_ref()], metadata::header).optional()?; + if local.is_some_and(|header| header != found.header) { + return Err(MetadataError::IdentityConflict.into()); + } + if !found.certified { + return Err(ClosureError::Uncertified(*oid)); + } + let h = found.header; + tx.execute( + "INSERT INTO base_objects VALUES(?1,?2,?3,?4,?5,?6)", + params![ + oid.as_ref(), + h.object.kind.git_name(), + h.object.size as i64, + h.object.digest.as_slice(), + h.edge_count as i64, + h.edge_digest.as_slice() + ], + )?; + } + if tx.execute( + "UPDATE lookups SET resolved=1 WHERE oid=?1 AND resolved=0", + [oid.as_ref()], + )? != 1 + { + return Err(ClosureError::Integrity); + } + } + Ok(()) + }) + } +} +pub(super) fn validate_header( + header: ObjectHeader, + format: ObjectFormat, +) -> Result<(), ClosureError> { + if header.object.oid.format() != format + || header.object.oid.is_zero() + || header.object.size > i64::MAX as u64 + || header.edge_count > i64::MAX as u64 + || (header.object.kind == ObjectKind::Blob && header.edge_count != 0) + { + return Err(ClosureError::Integrity); + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/closure/tests.rs b/crates/canopy-server/src/packs/closure/tests.rs new file mode 100644 index 0000000..cccd352 --- /dev/null +++ b/crates/canopy-server/src/packs/closure/tests.rs @@ -0,0 +1,453 @@ +use super::*; +use crate::packs::{ + directory::DirectoryBuilder, + metadata::{CanonicalObject, TypedEdge, tests::limits}, + verification::{ + PhysicalVerifier, + physical::tests::{ + independence::{git_input, upload_pair}, + physical_limits, prepared, + }, + }, +}; +use canopy_object_storage::artifact::ArtifactDescriptor; +use std::{collections::BTreeMap, time::Duration}; + +mod graph; +mod resources; + +type Result = std::result::Result>; +fn context(format: ObjectFormat) -> ClosureContext { + ClosureContext { + repository: [1; 16], + operation: [2; 16], + format, + base: None, + } +} +fn base(context: ClosureContext) -> ClosureBase { + // Test-only catalog identity; production derives this from leased Cell facts. + ClosureBase { + catalog: StoredCatalog { + repository: context.repository, + operation: [8; 16], + format: context.format, + artifact: ArtifactDescriptor { + size: 1, + digest: [3; 32], + manifest_digest: [4; 32], + }, + }, + generation: 7, + } +} +fn header(object: CanonicalObject, edges: &[TypedEdge]) -> ObjectHeader { + let edges: BTreeMap<_, _> = edges.iter().map(|e| (e.child, e.expected_kind)).collect(); + let mut chain = metadata::edge_seed(object.oid); + for (ordinal, (oid, kind)) in edges.iter().enumerate() { + let mut record = oid.to_vec(); + record.push(metadata::kind_code(*kind)); + chain = metadata::fold(chain, ordinal as u64, &record); + } + ObjectHeader { + object, + edge_count: edges.len() as u64, + edge_digest: chain, + } +} +struct Resolver { + headers: BTreeMap, + requests: Mutex>, +} +impl Resolver { + fn empty() -> Self { + Self { + headers: BTreeMap::new(), + requests: Mutex::new(Vec::new()), + } + } +} +impl BaseResolver for Resolver { + async fn resolve( + &self, + base: ClosureBase, + ids: &[ObjectId], + ) -> std::result::Result { + assert!(!ids.is_empty() && ids.len() <= PAGE_OBJECTS); + assert!(ids.windows(2).all(|pair| pair[0] < pair[1])); + self.requests.lock().unwrap().extend_from_slice(ids); + Ok(BaseBatch { + base, + objects: ids + .iter() + .map(|oid| self.headers.get(oid).copied()) + .collect(), + }) + } +} +async fn cleanup(root: &Path, budget: &DiskBudget) -> Result { + tokio::time::timeout(Duration::from_secs(5), async { + while budget.used() != 0 { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + assert_eq!(std::fs::read_dir(root)?.count(), 0); + Ok(()) +} + +#[tokio::test] +async fn physical_shards_close_and_bind_the_reused_directory_inventory() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let prepared = prepared(format, 1600).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(256 << 20); + let mut ctx = context(format); + ctx.base = Some(base(ctx)); + let mut closure = ClosureVerifier::new(root.path(), budget.clone(), ctx, limits()).await?; + let mut directory = DirectoryBuilder::new( + root.path(), + budget.clone(), + ctx.repository, + ctx.operation, + format, + limits(), + )?; + // Repeated exact physical inputs deduplicate objects, edges and input proofs. + for _ in 0..2 { + let mut physical = PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + prepared.descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + let mut segments = Vec::new(); + for count in [1, 800, prepared.descriptor.object_count - 801] { + segments.push(physical.inspect_next_shard(count).await?); + } + closure.begin_pack(physical.finish().await?)?; + for segment in segments { + directory.add_segment(&segment)?; + closure.add_segment(segment).await?; + } + closure.finish_pack().await?; + } + let mut resolver = Resolver::empty(); + let expected: Vec<_> = prepared + .fixture + .objects + .values() + .map(|(o, e)| header(*o, e)) + .collect(); + // An overlap from the base must be compared even though all children are local. + let overlapping = expected[0]; + resolver.headers.insert( + overlapping.object.oid, + BaseObject { + header: overlapping, + certified: true, + }, + ); + let witness = closure.finish(Some(&resolver)).await?; + assert_eq!(witness.context(), ctx); + assert_eq!(witness.input_count(), 1); + assert_eq!(witness.object_count(), expected.len() as u64); + assert_eq!( + witness.edge_count(), + expected.iter().map(|h| h.edge_count).sum::() + ); + witness.verify_headers(expected.iter().copied())?; + { + let requested = resolver.requests.lock().unwrap(); + assert_eq!( + *requested, + expected.iter().map(|h| h.object.oid).collect::>() + ); + } + let run = directory.seal()?; + witness.verify_run(run.descriptor())?; + let mut forged = run.descriptor(); + forged.inventory_digest[0] ^= 1; + assert!(matches!( + witness.verify_run(forged), + Err(ClosureError::Integrity) + )); + assert!(matches!( + witness.verify_headers(expected.iter().copied().skip(1)), + Err(ClosureError::Integrity) + )); + let mut conflicting = expected.clone(); + conflicting[0].object.digest[0] ^= 1; + assert!(matches!( + witness.verify_headers(conflicting), + Err(ClosureError::Integrity) + )); + drop(run); + cleanup(root.path(), &budget).await?; + } + Ok(()) +} + +#[tokio::test] +async fn commit_in_a_separate_pack_requires_the_exact_certified_base_dependency() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let prepared = prepared(format, 16).await?; + let (commit, edges) = prepared + .fixture + .objects + .values() + .find(|(o, _)| o.kind == ObjectKind::Commit) + .ok_or("commit")?; + let tree = edges + .iter() + .find(|e| e.expected_kind == ObjectKind::Tree) + .ok_or("tree")? + .child; + let (tree_object, tree_edges) = &prepared.fixture.objects[&tree]; + let tree_header = header(*tree_object, tree_edges); + let pack = git_input( + prepared.fixture.root.path(), + &["pack-objects", "--stdout", "--no-reuse-delta"], + format!("{}\n", hex::encode(commit.oid)).as_bytes(), + ) + .await?; + let descriptor = upload_pair(&prepared, commit.oid, &pack).await?; + // Success, missing base, uncertified base, and the wrong typed child. + for case in 0..4 { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let mut physical = PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + let segment = physical.inspect_next_shard(1).await?; + let mut ctx = context(format); + ctx.operation = descriptor.operation; + ctx.base = Some(base(ctx)); + let mut closure = + ClosureVerifier::new(root.path(), budget.clone(), ctx, limits()).await?; + closure.begin_pack(physical.finish().await?)?; + closure.add_segment(segment).await?; + closure.finish_pack().await?; + let mut resolver = Resolver::empty(); + if case != 1 { + let mut h = tree_header; + if case == 3 { + h.object.kind = ObjectKind::Tag; + } + resolver.headers.insert( + tree, + BaseObject { + header: h, + certified: case != 2, + }, + ); + } + let result = closure.finish(Some(&resolver)).await; + match case { + 0 => { + let witness = result?; + assert_eq!(witness.object_count(), 1); + assert_eq!(witness.edge_count(), 1); + witness.verify_headers([header(*commit, edges)])?; + } + 1 => assert!(matches!(result,Err(ClosureError::Missing(oid)) if oid==tree)), + 2 => assert!(matches!(result,Err(ClosureError::Uncertified(oid)) if oid==tree)), + _ => assert!( + matches!(result,Err(ClosureError::Kind { oid,expected:ObjectKind::Tree,actual:ObjectKind::Tag }) if oid==tree) + ), + } + assert_eq!(resolver.requests.lock().unwrap().len(), 2); + cleanup(root.path(), &budget).await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn incomplete_mismatched_or_interrupted_physical_partitions_poison_closure() -> Result { + let prepared = prepared(ObjectFormat::Sha256, 4).await?; + for case in 0..4 { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let mut physical = PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + prepared.descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + let first = physical.inspect_next_shard(1).await?; + let second = physical + .inspect_next_shard(prepared.descriptor.object_count - 1) + .await?; + let proof = physical.finish().await?; + let mut ctx = context(ObjectFormat::Sha256); + if case == 0 { + ctx.operation[0] ^= 1; + } + let mut closure = ClosureVerifier::new(root.path(), budget.clone(), ctx, limits()).await?; + if case == 0 { + assert!(matches!( + closure.begin_pack(proof), + Err(ClosureError::Integrity) + )); + } else { + closure.begin_pack(proof)?; + if case == 1 { + assert!(closure.add_segment(second.clone()).await.is_err()); + } + if case == 2 { + closure.add_segment(first.clone()).await?; + assert!(closure.finish_pack().await.is_err()); + } + if case == 3 { + closure.add_segment(first.clone()).await?; + } + } + assert!(closure.finish(None::<&Resolver>).await.is_err()); + drop((first, second)); + cleanup(root.path(), &budget).await?; + } + // Empty operations are allowed; they certify no pack or incoming object. + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let closure = ClosureVerifier::new( + root.path(), + budget.clone(), + context(ObjectFormat::Sha256), + limits(), + ) + .await?; + let witness = closure.finish(None::<&Resolver>).await?; + assert_eq!( + ( + witness.object_count(), + witness.input_count(), + witness.edge_count() + ), + (0, 0, 0) + ); + witness.verify_headers([])?; + cleanup(root.path(), &budget).await?; + Ok(()) +} + +#[tokio::test] +async fn changed_sealed_metadata_bytes_cannot_enter_closure_or_be_retried() -> Result { + let prepared = prepared(ObjectFormat::Sha256, 4).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let mut physical = PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + prepared.descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + let segment = physical + .inspect_next_shard(prepared.descriptor.object_count) + .await?; + let proof = physical.finish().await?; + let mut closure = ClosureVerifier::new( + root.path(), + budget.clone(), + context(ObjectFormat::Sha256), + limits(), + ) + .await?; + closure.begin_pack(proof)?; + // Mutate an unused tail byte: cached SQLite headers might remain readable, + // but copying must check all sealed bytes before trusting any cached row. + let mut bytes = std::fs::read(segment.path())?; + *bytes.last_mut().ok_or("metadata bytes")? ^= 1; + std::fs::write(segment.path(), bytes)?; + assert!(matches!( + closure.add_segment(segment.clone()).await, + Err(ClosureError::Integrity) + )); + assert!(matches!( + closure.add_segment(segment.clone()).await, + Err(ClosureError::Integrity) + )); + assert!(matches!( + closure.finish(None::<&Resolver>).await, + Err(ClosureError::Integrity) + )); + drop(segment); + cleanup(root.path(), &budget).await?; + Ok(()) +} + +#[tokio::test] +async fn separate_native_packs_merge_their_graphs_and_identical_canonical_overlaps() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(256 << 20); + let ctx = context(format); + let mut closure = ClosureVerifier::new(root.path(), budget.clone(), ctx, limits()).await?; + let mut directory = DirectoryBuilder::new( + root.path(), + budget.clone(), + ctx.repository, + ctx.operation, + format, + limits(), + )?; + let mut expected = BTreeMap::new(); + let mut physical_count = 0; + for blobs in [16, 8] { + let prepared = prepared(format, blobs).await?; + for (object, edges) in prepared.fixture.objects.values() { + let h = header(*object, edges); + if let Some(prior) = expected.insert(object.oid, h) { + assert_eq!(prior, h); + } + } + let mut physical = PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + prepared.descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + physical_count += prepared.descriptor.object_count as u64; + let segment = physical + .inspect_next_shard(prepared.descriptor.object_count) + .await?; + closure.begin_pack(physical.finish().await?)?; + directory.add_segment(&segment)?; + closure.add_segment(segment).await?; + closure.finish_pack().await?; + } + let witness = closure.finish(None::<&Resolver>).await?; + assert_eq!(witness.input_count(), 2); + assert_eq!(witness.object_count(), expected.len() as u64); + assert!(physical_count > witness.object_count()); + witness.verify_headers(expected.values().copied())?; + let run = directory.seal()?; + witness.verify_run(run.descriptor())?; + drop(run); + cleanup(root.path(), &budget).await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/closure/tests/graph.rs b/crates/canopy-server/src/packs/closure/tests/graph.rs new file mode 100644 index 0000000..8a0e9ac --- /dev/null +++ b/crates/canopy-server/src/packs/closure/tests/graph.rs @@ -0,0 +1,285 @@ +use super::*; + +pub(super) fn oid(format: ObjectFormat, n: u64) -> ObjectId { + let mut bytes = vec![0; format.bytes()]; + let start = bytes.len() - 8; + bytes[start..].copy_from_slice(&n.to_be_bytes()); + ObjectId::try_from(bytes.as_slice()).unwrap() +} +pub(super) fn synthetic( + format: ObjectFormat, + n: u64, + kind: ObjectKind, + edges: &[(u64, ObjectKind)], +) -> (ObjectHeader, Vec) { + let edges: Vec<_> = edges + .iter() + .map(|(n, k)| TypedEdge { + child: oid(format, *n), + expected_kind: *k, + }) + .collect(); + let object = CanonicalObject { + oid: oid(format, n), + kind, + size: n, + digest: *blake3::hash(&n.to_le_bytes()).as_bytes(), + }; + (header(object, &edges), edges) +} +pub(super) fn insert(spool: &mut Spool, objects: &[(ObjectHeader, Vec)]) -> Result { + // Private synthetic graph fixtures bypass physical verification to exercise + // cycles and type faults. No production constructor can accept these rows. + for page in objects.chunks(PAGE_OBJECTS) { + spool.write(|tx| { + for (h, edges) in page { + tx.execute("INSERT INTO objects(oid,kind,size,digest,edge_count,edge_digest) VALUES(?1,?2,?3,?4,?5,?6)",params![h.object.oid.as_ref(),h.object.kind.git_name(),h.object.size as i64,h.object.digest.as_slice(),h.edge_count as i64,h.edge_digest.as_slice()])?; + for e in edges { + tx.execute( + "INSERT INTO object_edges VALUES(?1,?2,?3)", + params![ + h.object.oid.as_ref(), + e.child.as_ref(), + e.expected_kind.git_name() + ], + )?; + } + } + Ok(()) + })?; + } + Ok(()) +} +fn spool(root: &Path, budget: DiskBudget, ctx: ClosureContext) -> Result { + Ok(Spool::new( + root, + budget, + ctx, + limits(), + Arc::new(AtomicBool::new(false)), + )?) +} + +#[test] +fn deep_chain_wide_fanout_and_many_ready_leaves_do_not_require_a_heap_graph() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let mut spool = spool(root.path(), budget.clone(), context(format))?; + let leaf = synthetic(format, 1, ObjectKind::Blob, &[]); + insert(&mut spool, &[leaf])?; + // 10,000-deep chain, then 1,200 parents of a common leaf. + for page in (2..10_002).collect::>().chunks(PAGE_OBJECTS) { + let nodes: Vec<_> = page + .iter() + .map(|n| { + synthetic( + format, + *n, + ObjectKind::Tag, + &[( + *n - 1, + if *n == 2 { + ObjectKind::Blob + } else { + ObjectKind::Tag + }, + )], + ) + }) + .collect(); + insert(&mut spool, &nodes)?; + } + let fanout: Vec<_> = (10_002..11_202) + .map(|n| synthetic(format, n, ObjectKind::Tree, &[(1, ObjectKind::Blob)])) + .collect(); + insert(&mut spool, &fanout)?; + let leaves: Vec<_> = (11_202..12_402) + .map(|n| synthetic(format, n, ObjectKind::Blob, &[])) + .collect(); + insert(&mut spool, &leaves)?; + let edges: Vec<_> = (11_202..12_402).map(|n| (n, ObjectKind::Blob)).collect(); + insert( + &mut spool, + &[synthetic(format, 12_402, ObjectKind::Tree, &edges)], + )?; + spool.certify_graph()?; + let proof = spool.witness()?; + assert_eq!((proof.object_count(), proof.edge_count()), (12_402, 12_400)); + assert_eq!( + spool.connection.query_row( + "SELECT count(*) FROM objects WHERE done=1 AND pending=0", + [], + |r| r.get::<_, u64>(0) + )?, + 12_402 + ); + // Critical paged queries must use their index and avoid a temp sort. + for (sql, index) in [ + ( + "SELECT oid FROM objects WHERE oid>x'00' ORDER BY oid LIMIT 512", + "PRIMARY KEY", + ), + ( + "SELECT DISTINCT child FROM object_edges WHERE child>x'00' ORDER BY child LIMIT 512", + "edge_child", + ), + ( + "SELECT oid FROM objects WHERE pending=0 AND done=0 ORDER BY oid LIMIT 512", + "ready_objects", + ), + ( + "SELECT oid FROM lookups WHERE resolved=0 ORDER BY oid LIMIT 512", + "unresolved_lookups", + ), + ( + "SELECT parent FROM object_edges WHERE child=x'00' AND parent>x'00' ORDER BY parent LIMIT 512", + "edge_child", + ), + ] { + let rows: Vec = spool + .connection + .prepare(&format!("EXPLAIN QUERY PLAN {sql}"))? + .query_map([], |r| r.get(3))? + .collect::>()?; + assert!(rows.iter().any(|r| r.contains(index)), "{rows:?}"); + assert!(!rows.iter().any(|r| r.contains("TEMP B-TREE")), "{rows:?}"); + } + drop(spool); + assert_eq!(budget.used(), 0); + assert_eq!(std::fs::read_dir(root.path())?.count(), 0); + } + Ok(()) +} + +#[test] +fn cycles_missing_children_and_wrong_types_cannot_certify_even_with_base_overlaps() -> Result { + let format = ObjectFormat::Sha256; + for case in 0..4 { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let mut ctx = context(format); + ctx.base = Some(base(ctx)); + let mut spool = spool(root.path(), budget.clone(), ctx)?; + let nodes = match case { + 0 => vec![ + synthetic(format, 1, ObjectKind::Tag, &[(1, ObjectKind::Tag)]), + synthetic(format, 3, ObjectKind::Blob, &[]), + ], + 1 => vec![ + synthetic(format, 1, ObjectKind::Tag, &[(2, ObjectKind::Tag)]), + synthetic(format, 2, ObjectKind::Tag, &[(1, ObjectKind::Tag)]), + synthetic(format, 3, ObjectKind::Blob, &[]), + ], + 2 => vec![synthetic( + format, + 1, + ObjectKind::Tree, + &[(2, ObjectKind::Blob)], + )], + _ => vec![ + synthetic(format, 1, ObjectKind::Tree, &[(2, ObjectKind::Blob)]), + synthetic(format, 2, ObjectKind::Tree, &[]), + ], + }; + insert(&mut spool, &nodes)?; + spool.prepare_lookups()?; + let ids = spool.lookup_page()?; + let objects = ids + .iter() + .map(|id| { + nodes + .iter() + .find(|(h, _)| h.object.oid == *id) + .map(|(h, _)| BaseObject { + header: *h, + certified: true, + }) + }) + .collect(); + spool.apply_base( + &ids, + BaseBatch { + base: ctx.base.unwrap(), + objects, + }, + )?; + let result = spool.certify_graph(); + match case { + 0 | 1 => assert!(matches!(result, Err(ClosureError::Cycle))), + 2 => assert!(matches!(result,Err(ClosureError::Missing(id)) if id==oid(format,2))), + _ => assert!( + matches!(result,Err(ClosureError::Kind {oid:id,expected:ObjectKind::Blob,actual:ObjectKind::Tree}) if id==oid(format,2)) + ), + } + drop(spool); + assert_eq!(budget.used(), 0); + } + Ok(()) +} + +#[test] +fn base_batches_bind_generation_order_shape_and_the_complete_canonical_header() -> Result { + let format = ObjectFormat::Sha256; + for case in 0..10 { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let mut ctx = context(format); + ctx.base = Some(base(ctx)); + let mut spool = spool(root.path(), budget.clone(), ctx)?; + let nodes = vec![ + synthetic(format, 1, ObjectKind::Blob, &[]), + synthetic(format, 2, ObjectKind::Tree, &[(1, ObjectKind::Blob)]), + ]; + insert(&mut spool, &nodes)?; + spool.prepare_lookups()?; + let ids = spool.lookup_page()?; + assert_eq!(ids.len(), 2); // Shared children are looked up only once. + let mut batch = BaseBatch { + base: ctx.base.unwrap(), + objects: nodes + .iter() + .map(|(h, _)| { + Some(BaseObject { + header: *h, + certified: true, + }) + }) + .collect(), + }; + match case { + 0 => batch.base.generation += 1, + 1 => { + batch.objects.pop(); + } + 2 => batch.objects.reverse(), + 3 => batch.objects[0].as_mut().unwrap().header.object.digest[0] ^= 1, + 4 => batch.objects[0].as_mut().unwrap().header.object.size += 1, + 5 => batch.objects[1].as_mut().unwrap().header.edge_count += 1, + 6 => batch.objects[1].as_mut().unwrap().header.edge_digest[0] ^= 1, + 7 => batch.objects[1].as_mut().unwrap().header.object.kind = ObjectKind::Tag, + 8 => batch.objects[0].as_mut().unwrap().certified = false, + _ => batch.base.catalog.artifact.digest[0] ^= 1, + } + let result = spool.apply_base(&ids, batch); + match case { + 3..=7 => assert!(matches!( + result, + Err(ClosureError::Metadata(MetadataError::IdentityConflict)) + )), + 8 => assert!(matches!(result, Err(ClosureError::Uncertified(_)))), + _ => assert!(matches!(result, Err(ClosureError::Integrity))), + } + assert_eq!(spool.lookup_page()?, ids); // Failed batch is atomic. + assert_eq!( + spool + .connection + .query_row("SELECT count(*) FROM base_objects", [], |r| r + .get::<_, u64>(0))?, + 0 + ); + drop(spool); + assert_eq!(budget.used(), 0); + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/closure/tests/resources.rs b/crates/canopy-server/src/packs/closure/tests/resources.rs new file mode 100644 index 0000000..2b94ad1 --- /dev/null +++ b/crates/canopy-server/src/packs/closure/tests/resources.rs @@ -0,0 +1,140 @@ +use super::*; +use std::sync::mpsc; + +#[tokio::test] +async fn disk_admission_precedes_files_and_sqlite_full_releases_the_spool() -> Result { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(1); + assert!( + ClosureVerifier::new( + root.path(), + budget.clone(), + context(ObjectFormat::Sha256), + limits() + ) + .await + .is_err() + ); + assert_eq!(budget.used(), 0); + assert_eq!(std::fs::read_dir(root.path())?.count(), 0); + let limits = MetadataLimits { + max_file_bytes: 64 << 10, + cache_kib: 16, + }; + let budget = DiskBudget::new(limits.max_file_bytes * 3); + let mut spool = Spool::new( + root.path(), + budget.clone(), + context(ObjectFormat::Sha256), + limits, + Arc::new(AtomicBool::new(false)), + )?; + let page: Vec<_> = (1..=512) + .map(|n| graph::synthetic(ObjectFormat::Sha256, n, ObjectKind::Blob, &[])) + .collect(); + let error = graph::insert(&mut spool, &page).unwrap_err(); + assert!(matches!( + error.downcast_ref::(), + Some(ClosureError::Metadata(MetadataError::Limit)) + )); + let pages: u64 = spool + .connection + .pragma_query_value(None, "page_count", |r| r.get(0))?; + assert!(pages * 4096 <= limits.max_file_bytes); + assert_eq!(budget.used(), limits.max_file_bytes * 3); + drop(spool); + cleanup(root.path(), &budget).await?; + Ok(()) +} + +#[test] +fn canceled_queued_work_retains_ownership_and_admission_until_the_worker_exits() -> Result { + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .max_blocking_threads(1) + .build()?; + runtime.block_on(async { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let closure = ClosureVerifier::new( + root.path(), + budget.clone(), + context(ObjectFormat::Sha256), + limits(), + ) + .await?; + let (weak, canceled) = closure.test_ownership(); + let used = budget.used(); + assert!(used > 0); + let (release_tx, release_rx) = mpsc::channel(); + let started = Arc::new(tokio::sync::Notify::new()); + let notify = started.clone(); + let blocked = tokio::task::spawn_blocking(move || { + notify.notify_one(); + release_rx.recv().unwrap(); + }); + started.notified().await; + let mut finish = Box::pin(closure.finish(None::<&Resolver>)); + assert!( + tokio::time::timeout(Duration::from_millis(25), &mut finish) + .await + .is_err() + ); + drop(finish); + assert!(canceled.load(Ordering::Acquire)); + assert!(weak.upgrade().is_some()); + assert_eq!(budget.used(), used); + release_tx.send(())?; + blocked.await?; + cleanup(root.path(), &budget).await?; + assert!(weak.upgrade().is_none()); + Ok::<_, Box>(()) + }) +} + +#[tokio::test] +async fn cancellation_interrupts_active_sql_and_bounded_file_hashing() -> Result { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let canceled = Arc::new(AtomicBool::new(false)); + let spool = Spool::new( + root.path(), + budget.clone(), + context(ObjectFormat::Sha256), + limits(), + canceled.clone(), + )?; + let (started_tx, started_rx) = tokio::sync::oneshot::channel(); + let work = tokio::task::spawn_blocking(move || { + let _ = started_tx.send(()); + // A long SQLite VM loop avoids unbounded fixture allocation and proves + // cancellation interrupts within a query, rather than only between pages. + let result=spool.connection.query_row("WITH RECURSIVE numbers(n) AS (VALUES(1) UNION ALL SELECT n+1 FROM numbers WHERE n<1000000000) SELECT sum(n) FROM numbers",[],|row|row.get::<_,i64>(0)); + assert!( + matches!(result,Err(rusqlite::Error::SqliteFailure(code,_)) if code.code==rusqlite::ErrorCode::OperationInterrupted) + ); + drop(spool); + }); + started_rx.await?; + canceled.store(true, Ordering::Release); + tokio::time::timeout(Duration::from_secs(5), work).await??; + cleanup(root.path(), &budget).await?; + let file = tempfile::NamedTempFile::new()?; + std::fs::write(file.path(), vec![5; 1 << 20])?; + let mut checkpoints = 0; + let result = metadata::file_digest_with::(file.path(), 1 << 20, || { + checkpoints += 1; + if checkpoints == 5 { + Err(ClosureError::Canceled) + } else { + Ok(()) + } + }); + assert!(matches!(result, Err(ClosureError::Canceled))); + assert_eq!(checkpoints, 5); + assert_eq!( + metadata::file_digest(file.path(), 1 << 20)?, + *blake3::hash(&vec![5; 1 << 20]).as_bytes() + ); + Ok(()) +} diff --git a/crates/canopy-server/src/packs/closure/verifier.rs b/crates/canopy-server/src/packs/closure/verifier.rs new file mode 100644 index 0000000..e062053 --- /dev/null +++ b/crates/canopy-server/src/packs/closure/verifier.rs @@ -0,0 +1,235 @@ +use super::*; + +struct ActivePack { + partition: PhysicalPartition, + digest: [u8; 32], + native: crate::packs::sources::NativePackDescriptor, + custody: Option, +} +pub struct ClosureVerifier { + spool: Arc>, + context: ClosureContext, + canceled: Arc, + active: Option, + failed: bool, + completed: bool, +} +impl ClosureVerifier { + #[cfg(test)] + pub(super) fn test_ownership(&self) -> (std::sync::Weak>, Arc) { + (Arc::downgrade(&self.spool), self.canceled.clone()) + } + pub async fn new( + root: &Path, + budget: DiskBudget, + context: ClosureContext, + limits: MetadataLimits, + ) -> Result { + Self::new_inner(root, budget, context, limits, None).await + } + pub(in crate::packs) async fn new_in_workspace( + workspace: Arc, + budget: DiskBudget, + context: ClosureContext, + limits: MetadataLimits, + ) -> Result { + let root = workspace.path().to_owned(); + Self::new_inner(&root, budget, context, limits, Some(workspace)).await + } + async fn new_inner( + root: &Path, + budget: DiskBudget, + context: ClosureContext, + limits: MetadataLimits, + workspace: Option>, + ) -> Result { + let canceled = Arc::new(AtomicBool::new(false)); + let mut guard = CancelGuard::new(canceled.clone()); + let root = root.to_owned(); + let token = canceled.clone(); + let spool = tokio::task::spawn_blocking(move || { + let mut spool = Spool::new(&root, budget, context, limits, token)?; + if let Some(workspace) = workspace { + spool._admitted.retain_workspace(workspace); + } + Ok::<_, ClosureError>(spool) + }) + .await??; + guard.complete(); + Ok(Self { + spool: Arc::new(Mutex::new(spool)), + context, + canceled, + active: None, + failed: false, + completed: false, + }) + } + fn healthy(&self) -> Result<(), ClosureError> { + if self.failed { + Err(ClosureError::Integrity) + } else if self.canceled.load(Ordering::Acquire) { + Err(ClosureError::Canceled) + } else { + Ok(()) + } + } + /// Consume only a complete isolated physical witness. Copy its metadata + /// shards sequentially; a missing or mismatched shard prevents finishing. + pub fn begin_pack(&mut self, witness: PhysicalPackWitness) -> Result<(), ClosureError> { + self.begin_inner(witness, None) + } + pub(in crate::packs) fn begin_retained_pack( + &mut self, + witness: PhysicalPackWitness, + custody: crate::packs::publication::RetainedNativeInput, + ) -> Result<(), ClosureError> { + self.begin_inner(witness, Some(custody)) + } + fn begin_inner( + &mut self, + witness: PhysicalPackWitness, + custody: Option, + ) -> Result<(), ClosureError> { + self.healthy()?; + self.failed = true; + let native = witness.native(); + if let Some(custody) = &custody { + custody.authorize(self.context, native)?; + } + if self.active.is_some() + || native.repository != self.context.repository + || (native.operation != self.context.operation && custody.is_none()) + || native.format != self.context.format + { + return Err(ClosureError::Integrity); + } + self.active = Some(ActivePack { + partition: witness.partition(), + digest: witness.metadata_digest(), + native, + custody, + }); + self.failed = false; + Ok(()) + } + pub async fn add_segment(&mut self, segment: Arc) -> Result<(), ClosureError> { + self.healthy()?; + self.failed = true; + let mut guard = CancelGuard::new(self.canceled.clone()); + let active = self.active.as_mut().ok_or(ClosureError::Integrity)?; + if let Some(custody) = &active.custody { + custody.authorize(self.context, active.native)?; + } + active.partition.add(segment.descriptor())?; + let operation = active.native.operation; + let spool = self.spool.clone(); + tokio::task::spawn_blocking(move || { + spool + .lock() + .map_err(|_| ClosureError::Integrity)? + .copy_segment(&segment, operation) + }) + .await??; + self.failed = false; + guard.complete(); + Ok(()) + } + pub async fn finish_pack(&mut self) -> Result<(), ClosureError> { + self.healthy()?; + self.failed = true; + let mut guard = CancelGuard::new(self.canceled.clone()); + let active = self.active.take().ok_or(ClosureError::Integrity)?; + if let Some(custody) = &active.custody { + custody.authorize(self.context, active.native)?; + } + active.partition.finish()?; + let spool = self.spool.clone(); + tokio::task::spawn_blocking(move || { + spool + .lock() + .map_err(|_| ClosureError::Integrity)? + .input(active.digest) + }) + .await??; + self.failed = false; + guard.complete(); + Ok(()) + } + /// No base is permitted only for an empty published dependency catalog. + /// External lookup status is conditional on the trusted resolver contract; + /// final publication must recheck the bound catalog generation and fence. + pub async fn finish( + self, + resolver: Option<&impl BaseResolver>, + ) -> Result { + let (witness, _) = self.finish_retained(resolver).await?; + Ok(witness) + } + pub(in crate::packs) async fn finish_retained( + mut self, + resolver: Option<&impl BaseResolver>, + ) -> Result<(ClosureWitness, RetainedClosure), ClosureError> { + self.healthy()?; + if self.active.is_some() || self.context.base.is_some() != resolver.is_some() { + return Err(ClosureError::Integrity); + } + let mut guard = CancelGuard::new(self.canceled.clone()); + let spool = self.spool.clone(); + tokio::task::spawn_blocking(move || { + spool + .lock() + .map_err(|_| ClosureError::Integrity)? + .prepare_lookups() + }) + .await??; + if let (Some(base), Some(resolver)) = (self.context.base, resolver) { + loop { + let spool = self.spool.clone(); + let ids = tokio::task::spawn_blocking(move || { + spool + .lock() + .map_err(|_| ClosureError::Integrity)? + .lookup_page() + }) + .await??; + if ids.is_empty() { + break; + } + let batch = resolver.resolve(base, &ids).await?; + let spool = self.spool.clone(); + tokio::task::spawn_blocking(move || { + spool + .lock() + .map_err(|_| ClosureError::Integrity)? + .apply_base(&ids, batch) + }) + .await??; + } + } + let spool = self.spool.clone(); + let witness = tokio::task::spawn_blocking(move || { + let mut spool = spool.lock().map_err(|_| ClosureError::Integrity)?; + spool.certify_graph()?; + spool.witness() + }) + .await??; + guard.complete(); + self.completed = true; + Ok(( + witness, + RetainedClosure { + spool: Arc::clone(&self.spool), + context: self.context, + canceled: Arc::clone(&self.canceled), + }, + )) + } +} +impl Drop for ClosureVerifier { + fn drop(&mut self) { + if !self.completed { + self.canceled.store(true, Ordering::Release); + } + } +} diff --git a/crates/canopy-server/src/packs/closure/witness.rs b/crates/canopy-server/src/packs/closure/witness.rs new file mode 100644 index 0000000..61642c2 --- /dev/null +++ b/crates/canopy-server/src/packs/closure/witness.rs @@ -0,0 +1,135 @@ +use super::*; + +/// Complete incoming closure conditional on the bound certified base. Reuse the +/// directory's canonical inventory encoding to bind the incoming directory. +/// The coordinator must verify catalog inputs/coverage and CAS context.base +/// under the admitted owner fence before any durable acknowledgement. +pub struct ClosureWitness { + context: ClosureContext, + object_count: u64, + edge_count: u64, + input_count: u64, + inputs_digest: [u8; 32], + inventory_digest: [u8; 32], + first: Option, + last: Option, +} +impl ClosureWitness { + pub fn context(&self) -> ClosureContext { + self.context + } + pub fn object_count(&self) -> u64 { + self.object_count + } + pub fn edge_count(&self) -> u64 { + self.edge_count + } + pub fn input_count(&self) -> u64 { + self.input_count + } + pub fn inputs_digest(&self) -> [u8; 32] { + self.inputs_digest + } + pub fn inventory_digest(&self) -> [u8; 32] { + self.inventory_digest + } + pub fn verify_run(&self, run: RunDescriptor) -> Result<(), ClosureError> { + run.validate()?; + if run.repository != self.context.repository + || run.operation != self.context.operation + || run.format != self.context.format + || run.object_count != self.object_count + || run.inventory_digest != self.inventory_digest + || Some(run.first_oid) != self.first + || Some(run.last_oid) != self.last + { + return Err(ClosureError::Integrity); + } + Ok(()) + } + /// For partitioned incoming directories, stream their complete canonical + /// union in raw OID order. This does not verify descriptor/source provenance. + pub fn verify_headers( + &self, + headers: impl IntoIterator, + ) -> Result<(), ClosureError> { + let mut count = 0_u64; + let mut chain = directory::inventory_seed(self.context.format); + let mut last = None; + for header in headers { + spool::validate_header(header, self.context.format)?; + if last.is_some_and(|oid| oid >= header.object.oid) { + return Err(ClosureError::Integrity); + } + chain = metadata::fold_header(chain, count, header); + count = count.checked_add(1).ok_or(MetadataError::Limit)?; + last = Some(header.object.oid); + } + if count != self.object_count || chain != self.inventory_digest { + return Err(ClosureError::Integrity); + } + Ok(()) + } +} +impl Spool { + pub(super) fn witness(&self) -> Result { + self.check_cancel()?; + let mut after = Vec::new(); + let mut count = 0_u64; + let mut edges = 0_u64; + let mut first = None; + let mut last = None; + let mut inventory = directory::inventory_seed(self.context.format); + loop { + self.check_cancel()?; + let headers: Vec = self.connection.prepare_cached("SELECT oid,kind,size,digest,edge_count,edge_digest FROM objects WHERE oid>?1 ORDER BY oid LIMIT ?2")? + .query_map(params![after,PAGE_OBJECTS as i64], metadata::header)?.collect::>()?; + if headers.is_empty() { + break; + } + for header in headers { + spool::validate_header(header, self.context.format)?; + inventory = metadata::fold_header(inventory, count, header); + edges = edges + .checked_add(header.edge_count) + .ok_or(MetadataError::Limit)?; + count = count.checked_add(1).ok_or(MetadataError::Limit)?; + first.get_or_insert(header.object.oid); + last = Some(header.object.oid); + after = header.object.oid.to_vec(); + } + } + let actual_edges: u64 = + self.connection + .query_row("SELECT count(*) FROM object_edges", [], |row| row.get(0))?; + if edges != actual_edges { + return Err(ClosureError::Integrity); + } + let mut inputs = blake3::Hasher::new(); + inputs.update(b"canopy.closure-inputs.v1\0"); + inputs.update(&self.context.repository); + inputs.update(&self.context.operation); + inputs.update(&[self.context.format.bytes() as u8]); + let mut rows = self + .connection + .prepare_cached("SELECT digest FROM inputs ORDER BY digest")?; + let mut cursor = rows.query([])?; + let mut input_count = 0; + while let Some(row) = cursor.next()? { + self.check_cancel()?; + let digest = metadata::digest(row.get(0)?)?; + inputs.update(&digest); + input_count += 1; + } + Ok(ClosureWitness { + context: self.context, + object_count: count, + edge_count: edges, + input_count, + inputs_digest: *inputs.finalize().as_bytes(), + inventory_digest: inventory, + first, + last, + }) + } +} diff --git a/crates/canopy-server/src/packs/directory/coverage.rs b/crates/canopy-server/src/packs/directory/coverage.rs new file mode 100644 index 0000000..58ece4e --- /dev/null +++ b/crates/canopy-server/src/packs/directory/coverage.rs @@ -0,0 +1,166 @@ +use super::*; + +/// A contiguous logical projection of an authenticated immutable run. Its fold +/// uses the existing canonical encoding and excludes physical placement, exactly +/// like a whole run. It is not independently an authorization/closure proof. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct RunCoverage { + pub object_count: u64, + pub first_oid: ObjectId, + pub last_oid: ObjectId, + pub inventory_digest: [u8; 32], +} +impl RunDescriptor { + pub fn coverage(self) -> RunCoverage { + RunCoverage { + object_count: self.object_count, + first_oid: self.first_oid, + last_oid: self.last_oid, + inventory_digest: self.inventory_digest, + } + } +} +impl RunCoverage { + pub fn validate(self, run: RunDescriptor) -> Result<(), MetadataError> { + if self.object_count == 0 + || self.object_count > run.object_count + || self.first_oid.format() != run.format + || self.last_oid.format() != run.format + || self.first_oid < run.first_oid + || self.last_oid > run.last_oid + || self.first_oid > self.last_oid + || (self.object_count == 1) != (self.first_oid == self.last_oid) + || ((self.object_count == run.object_count + || (self.first_oid == run.first_oid && self.last_oid == run.last_oid)) + && self != run.coverage()) + { + return Err(MetadataError::Integrity); + } + Ok(()) + } +} +struct Fold { + count: u64, + edges: u64, + inventory: [u8; 32], + first: Option, + last: Option, +} +impl Fold { + fn new(format: ObjectFormat) -> Self { + Self { + count: 0, + edges: 0, + inventory: inventory_seed(format), + first: None, + last: None, + } + } + fn add(&mut self, entry: DirectoryEntry) -> Result<(), MetadataError> { + let oid = entry.header.object.oid; + if self.last.is_some_and(|last| last >= oid) { + return Err(MetadataError::Integrity); + } + self.inventory = fold_header(self.inventory, self.count, entry.header); + self.count = self.count.checked_add(1).ok_or(MetadataError::Limit)?; + self.edges = self + .edges + .checked_add(entry.header.edge_count) + .ok_or(MetadataError::Limit)?; + self.first.get_or_insert(oid); + self.last = Some(oid); + Ok(()) + } + fn coverage(&self) -> Option { + Some(RunCoverage { + object_count: self.count, + first_oid: self.first?, + last_oid: self.last?, + inventory_digest: self.inventory, + }) + } +} +impl DirectoryRun { + fn entries_in_coverage( + &self, + coverage: RunCoverage, + after: Option, + ) -> Result, MetadataError> { + let connection = self.connection()?; + let mut statement = connection.prepare_cached("SELECT oid,kind,size,digest,edge_count,edge_digest,source_operation,source_digest,location_version FROM objects WHERE oid>=?1 AND oid<=?2 AND oid>?3 ORDER BY oid LIMIT ?4")?; + Ok(statement + .query_map( + params![ + coverage.first_oid.as_ref(), + coverage.last_oid.as_ref(), + after.as_ref().map_or(&[][..], AsRef::<[u8]>::as_ref), + PAGE_OBJECTS as i64 + ], + entry, + )? + .collect::>()?) + } + #[cfg(test)] + pub(in crate::packs) fn verify_coverage( + &self, + coverage: RunCoverage, + ) -> Result { + self.copy_coverage(coverage, |_| Ok(())) + } + pub(super) fn copy_coverage( + &self, + coverage: RunCoverage, + mut consume: impl FnMut(&[DirectoryEntry]) -> Result<(), MetadataError>, + ) -> Result { + coverage.validate(self.descriptor)?; + let mut fold = Fold::new(self.descriptor.format); + loop { + let page = self.entries_in_coverage(coverage, fold.last)?; + if page.is_empty() { + break; + } + for value in &page { + fold.add(*value)?; + } + consume(&page)?; + } + if fold.coverage() != Some(coverage) { + return Err(MetadataError::Integrity); + } + Ok(fold.edges) + } + /// Verify the complete input projection while folding both resulting parts + /// in that same bounded pass. Empty parts are absent. Neither part escapes + /// until exact parent inventory/count/endpoints have matched. + pub(in crate::packs) fn split_coverage( + &self, + coverage: RunCoverage, + through: ObjectId, + ) -> Result<(Option, Option, u64), MetadataError> { + coverage.validate(self.descriptor)?; + if through.format() != self.descriptor.format { + return Err(MetadataError::Integrity); + } + let mut parent = Fold::new(self.descriptor.format); + let mut left = Fold::new(self.descriptor.format); + let mut right = Fold::new(self.descriptor.format); + loop { + let page = self.entries_in_coverage(coverage, parent.last)?; + if page.is_empty() { + break; + } + for value in page { + parent.add(value)?; + if value.header.object.oid <= through { + left.add(value)?; + } else { + right.add(value)?; + } + } + } + if parent.coverage() != Some(coverage) { + return Err(MetadataError::Integrity); + } + Ok((left.coverage(), right.coverage(), left.edges)) + } +} diff --git a/crates/canopy-server/src/packs/directory/index/bulk.rs b/crates/canopy-server/src/packs/directory/index/bulk.rs new file mode 100644 index 0000000..ad345a6 --- /dev/null +++ b/crates/canopy-server/src/packs/directory/index/bulk.rs @@ -0,0 +1,299 @@ +//! Streaming ordered construction. One completed block is held behind the +//! active block at each level so small final tails can be balanced before reuse. +use super::*; + +struct Level { + pending: Option>>, + refs: Vec>, + bytes: usize, +} +impl Level { + fn empty(&self) -> bool { + self.pending.is_none() && self.refs.is_empty() + } +} +/// Drain at most two valid blocks, balancing a final underfilled tail by both +/// encoded bytes and fanout. The same helper handles leaves and child groups. +fn groups( + pending: Option>, + mut active: Vec, + header: usize, + limit: u32, + fanout: usize, + encode: impl Fn(&T, &mut BoundedEncoder) -> Result<(), CodecError>, +) -> Result>, IndexError> { + let Some(mut pending) = pending else { + return Ok(if active.is_empty() { + Vec::new() + } else { + vec![active] + }); + }; + if active.is_empty() { + return Ok(vec![pending]); + } + let capacity = (limit as usize) + .checked_sub(header) + .ok_or(IndexError::Limit)?; + let size = |item: &T| -> Result { + let mut e = BoundedEncoder::new(limit)?; + encode(item, &mut e)?; + Ok(e.finish().len()) + }; + let active_bytes = active + .iter() + .map(&size) + .try_fold(0usize, |sum, n| Ok::<_, IndexError>(sum + n?))?; + if active.len() * 2 >= fanout || active_bytes * 2 >= capacity { + return Ok(vec![pending, active]); + } + pending.append(&mut active); + let sizes = pending.iter().map(size).collect::, _>>()?; + let total: usize = sizes.iter().sum(); + let mut prefix = 0; + let mut best = None; + for (at, size) in sizes.iter().enumerate().take(sizes.len() - 1) { + prefix += size; + let left = at + 1; + let right = sizes.len() - left; + if left > fanout || right > fanout || prefix > capacity || total - prefix > capacity { + continue; + } + let left_load = (left as u64 * capacity as u64).max(prefix as u64 * fanout as u64); + let right_load = + (right as u64 * capacity as u64).max((total - prefix) as u64 * fanout as u64); + let difference = left_load.abs_diff(right_load); + if best.is_none_or(|(_, score)| difference < score) { + best = Some((left, difference)); + } + } + let (at, _) = best.ok_or(IndexError::Limit)?; + let second = pending.split_off(at); + Ok(vec![pending, second]) +} +pub(super) struct Builder<'a, R: IndexRecord> { + index: &'a RangeIndex, + operation: [u8; 16], + header: usize, + levels: Vec>, + pending_leaf: Option>, + leaf: Vec, + used: usize, + last: Option, +} +impl<'a, R: IndexRecord> Builder<'a, R> { + pub(super) fn new(index: &'a RangeIndex, operation: [u8; 16]) -> Result { + let header = Node:: { + repository: index.repository(), + operation, + format: index.format, + height: 0, + contents: Contents::Runs(Vec::new()), + } + .header_size()?; + Ok(Self { + index, + operation, + header, + levels: Vec::new(), + pending_leaf: None, + leaf: Vec::new(), + used: header, + last: None, + }) + } + async fn emit_leaf(&mut self, records: Vec) -> Result<(), IndexError> { + let reference = self + .index + .persist(self.operation, 0, Contents::Runs(records)) + .await?; + self.index + .carry(self.operation, self.header, &mut self.levels, reference) + .await + } + async fn flush_leaf(&mut self) -> Result<(), IndexError> { + let groups = groups( + self.pending_leaf.take(), + std::mem::take(&mut self.leaf), + self.header, + R::NODE_BYTES, + R::FANOUT, + |record, e| record.encode_record(e), + )?; + self.used = self.header; + for group in groups { + self.emit_leaf(group).await?; + } + Ok(()) + } + fn drain_level(&mut self, height: usize) -> Result>>, IndexError> { + let Some(level) = self.levels.get_mut(height) else { + return Ok(Vec::new()); + }; + let groups = groups( + level.pending.take(), + std::mem::take(&mut level.refs), + self.header, + R::NODE_BYTES, + R::FANOUT, + |reference, e| codec::reference(e, reference.clone()), + )?; + level.bytes = self.header; + Ok(groups) + } + pub(super) async fn record(&mut self, record: R) -> Result<(), IndexError> { + record.validate_record(self.index.repository(), self.index.format)?; + if self + .last + .as_ref() + .is_some_and(|last| *last >= record.first_key()) + { + return Err(IndexError::RangeOverlap); + } + let mut e = BoundedEncoder::new(R::NODE_BYTES)?; + record.encode_record(&mut e)?; + let size = e.finish().len(); + if self.header + size > R::NODE_BYTES as usize { + return Err(IndexError::Limit); + } + if !self.leaf.is_empty() + && (self.leaf.len() == R::FANOUT || self.used + size > R::NODE_BYTES as usize) + { + let full = std::mem::take(&mut self.leaf); + if let Some(previous) = self.pending_leaf.replace(full) { + self.emit_leaf(previous).await?; + } + self.used = self.header; + } + self.last = Some(record.last_key()); + self.leaf.push(record); + self.used += size; + Ok(()) + } + /// Caller authenticated the root or its containing parent. Reuse itself + /// does not issue a new descendant/existence certificate. + pub(super) async fn subtree(&mut self, reference: NodeRef) -> Result<(), IndexError> { + reference.validate(self.index.format)?; + if self + .last + .as_ref() + .is_some_and(|last| *last >= reference.first_key) + { + return Err(IndexError::RangeOverlap); + } + self.flush_leaf().await?; + // Lower groups are the later suffix of emitted content. Lift them + // first so a reused higher subtree cannot precede that suffix. + for height in 0..usize::from(reference.height) { + for children in self.drain_level(height)? { + let parent = self + .index + .persist( + self.operation, + (height + 1) as u8, + Contents::Children(children), + ) + .await?; + self.index + .carry(self.operation, self.header, &mut self.levels, parent) + .await?; + } + } + self.last = Some(reference.last_key.clone()); + self.index + .carry(self.operation, self.header, &mut self.levels, reference) + .await + } + pub(super) async fn finish(mut self) -> Result>, IndexError> { + self.flush_leaf().await?; + let mut height = 0; + while height < self.levels.len() { + let mut groups = self.drain_level(height)?; + if groups.len() == 1 && groups[0].len() == 1 && self.levels.iter().all(Level::empty) { + return Ok(groups.pop().and_then(|mut children| children.pop())); + } + for children in groups { + let parent_height = u8::try_from(height + 1).map_err(|_| IndexError::Limit)?; + if parent_height > R::MAX_HEIGHT { + return Err(IndexError::Limit); + } + let reference = self + .index + .persist(self.operation, parent_height, Contents::Children(children)) + .await?; + self.index + .carry(self.operation, self.header, &mut self.levels, reference) + .await?; + } + height += 1; + } + Ok(None) + } +} +impl RangeIndex { + async fn carry( + &self, + operation: [u8; 16], + header: usize, + levels: &mut Vec>, + mut reference: NodeRef, + ) -> Result<(), IndexError> { + loop { + let height = reference.height as usize; + if height > R::MAX_HEIGHT as usize { + return Err(IndexError::Limit); + } + while levels.len() <= height { + levels.push(Level { + pending: None, + refs: Vec::new(), + bytes: header, + }); + } + let mut e = BoundedEncoder::new(R::NODE_BYTES)?; + codec::reference(&mut e, reference.clone())?; + let size = e.finish().len(); + if header + size > R::NODE_BYTES as usize { + return Err(IndexError::Limit); + } + let level = &mut levels[height]; + if !level.refs.is_empty() + && (level.refs.len() == R::FANOUT || level.bytes + size > R::NODE_BYTES as usize) + { + let full = std::mem::take(&mut level.refs); + let previous = level.pending.replace(full); + level.refs.push(reference); + level.bytes = header + size; + let Some(children) = previous else { + return Ok(()); + }; + let parent_height = u8::try_from(height + 1).map_err(|_| IndexError::Limit)?; + if parent_height > R::MAX_HEIGHT { + return Err(IndexError::Limit); + } + reference = self + .persist(operation, parent_height, Contents::Children(children)) + .await?; + } else { + level.refs.push(reference); + level.bytes += size; + return Ok(()); + } + } + } + pub async fn build_sorted( + &self, + operation: [u8; 16], + records: I, + ) -> Result>, IndexError> + where + I: IntoIterator>, + I::IntoIter: Send, + { + let mut builder = Builder::new(self, operation)?; + for record in records { + builder.record(record?).await?; + } + builder.finish().await + } +} diff --git a/crates/canopy-server/src/packs/directory/index/codec.rs b/crates/canopy-server/src/packs/directory/index/codec.rs new file mode 100644 index 0000000..4f6d7d7 --- /dev/null +++ b/crates/canopy-server/src/packs/directory/index/codec.rs @@ -0,0 +1,183 @@ +use super::*; + +pub(in crate::packs) fn fixed( + decoder: &mut BoundedDecoder<'_>, +) -> Result<[u8; N], CodecError> { + decoder + .read_bytes()? + .try_into() + .map_err(|_| CodecError::Invalid("invalid fixed-width directory field")) +} +fn oid(decoder: &mut BoundedDecoder<'_>, format: ObjectFormat) -> Result { + let oid = ObjectId::try_from(decoder.read_bytes()?) + .map_err(|_| CodecError::Invalid("invalid directory OID"))?; + if oid.format() != format { + return Err(CodecError::Invalid("directory OID format mismatch")); + } + Ok(oid) +} +pub(in crate::packs) fn artifact( + encoder: &mut BoundedEncoder, + value: ArtifactDescriptor, +) -> Result<(), CodecError> { + encoder.write_u64(value.size)?; + encoder.write_bytes(&value.digest)?; + encoder.write_bytes(&value.manifest_digest) +} +pub(in crate::packs) fn read_artifact( + decoder: &mut BoundedDecoder<'_>, +) -> Result { + Ok(ArtifactDescriptor { + size: decoder.read_u64()?, + digest: fixed(decoder)?, + manifest_digest: fixed(decoder)?, + }) +} +pub(in crate::packs) fn reference( + encoder: &mut BoundedEncoder, + value: NodeRef, +) -> Result<(), CodecError> { + encoder.write_bytes(&value.operation)?; + artifact(encoder, value.artifact)?; + encoder.write_u8(value.height)?; + value.first_key.encode(encoder)?; + value.last_key.encode(encoder)?; + encoder.write_u64(value.record_count)?; + encoder.write_u64(value.object_count) +} +pub(in crate::packs) fn read_reference( + decoder: &mut BoundedDecoder<'_>, + format: ObjectFormat, +) -> Result, CodecError> { + Ok(NodeRef { + operation: fixed(decoder)?, + artifact: read_artifact(decoder)?, + height: decoder.read_u8()?, + first_key: R::Key::decode(decoder, format)?, + last_key: R::Key::decode(decoder, format)?, + record_count: decoder.read_u64()?, + object_count: decoder.read_u64()?, + }) +} +pub(in crate::packs) fn write_run( + encoder: &mut BoundedEncoder, + stored: StoredRun, +) -> Result<(), CodecError> { + let run = stored.run; + encoder.write_bytes(&run.operation)?; + encoder.write_u64(run.object_count)?; + encoder.write_bytes(&run.first_oid)?; + encoder.write_bytes(&run.last_oid)?; + encoder.write_bytes(&run.inventory_digest)?; + artifact(encoder, stored.artifact)?; + encoder.write_u64(stored.coverage.object_count)?; + encoder.write_bytes(&stored.coverage.first_oid)?; + encoder.write_bytes(&stored.coverage.last_oid)?; + encoder.write_bytes(&stored.coverage.inventory_digest) +} +pub(in crate::packs) fn read_run( + decoder: &mut BoundedDecoder<'_>, + repository: [u8; 16], + format: ObjectFormat, +) -> Result { + let operation = fixed(decoder)?; + let object_count = decoder.read_u64()?; + let first_oid = oid(decoder, format)?; + let last_oid = oid(decoder, format)?; + let inventory_digest = fixed(decoder)?; + let artifact = read_artifact(decoder)?; + let coverage = RunCoverage { + object_count: decoder.read_u64()?, + first_oid: oid(decoder, format)?, + last_oid: oid(decoder, format)?, + inventory_digest: fixed(decoder)?, + }; + Ok(StoredRun { + run: RunDescriptor { + repository, + operation, + format, + object_count, + first_oid, + last_oid, + inventory_digest, + size: artifact.size, + digest: artifact.digest, + }, + artifact, + coverage, + }) +} +impl Node { + fn prefix(&self) -> Result { + let mut encoder = BoundedEncoder::new(R::NODE_BYTES)?; + encoder.write_bytes(R::DOMAIN)?; + encoder.write_bytes(&self.repository)?; + encoder.write_bytes(&self.operation)?; + encoder.write_u8(self.format.bytes() as u8)?; + encoder.write_u8(self.height)?; + encoder.write_count(self.contents.len())?; + Ok(encoder) + } + pub(super) fn header_size(&self) -> Result { + Ok(self.prefix()?.finish().len()) + } + pub(super) fn encode(&self) -> Result, IndexError> { + self.validate()?; + let mut encoder = self.prefix()?; + match &self.contents { + Contents::Runs(runs) => { + for stored in runs { + stored.encode_record(&mut encoder)?; + } + } + Contents::Children(children) => { + for child in children { + reference(&mut encoder, child.clone())?; + } + } + } + Ok(encoder.finish()) + } + pub(super) fn decode(bytes: &[u8]) -> Result { + let mut decoder = BoundedDecoder::new(bytes, R::NODE_BYTES)?; + if decoder.read_bytes()? != R::DOMAIN { + return Err(IndexError::Integrity); + } + let repository = fixed(&mut decoder)?; + let operation = fixed(&mut decoder)?; + let format = match decoder.read_u8()? { + 20 => ObjectFormat::Sha1, + 32 => ObjectFormat::Sha256, + _ => return Err(IndexError::Integrity), + }; + let height = decoder.read_u8()?; + let count = decoder.read_count()?; + if !(1..=R::FANOUT).contains(&count) || height > R::MAX_HEIGHT { + return Err(IndexError::Limit); + } + let contents = if height == 0 { + let mut runs = Vec::with_capacity(count); + for _ in 0..count { + runs.push(R::decode_record(&mut decoder, repository, format)?); + } + Contents::Runs(runs) + } else { + let mut children = Vec::with_capacity(count); + for _ in 0..count { + children.push(read_reference(&mut decoder, format)?); + } + Contents::Children(children) + }; + decoder.finish()?; + let node = Self { + repository, + operation, + format, + height, + contents, + }; + node.validate()?; + Ok(node) + } +} diff --git a/crates/canopy-server/src/packs/directory/index/cursor.rs b/crates/canopy-server/src/packs/directory/index/cursor.rs new file mode 100644 index 0000000..9bd70f7 --- /dev/null +++ b/crates/canopy-server/src/packs/directory/index/cursor.rs @@ -0,0 +1,161 @@ +use super::*; + +/// Forward cursor retaining only one bounded-height path. A canceled/failed +/// `next` poisons the cursor; restart from a known last returned OID instead. +pub struct RangeCursor<'a, R: IndexRecord = StoredRun> { + index: &'a RangeIndex, + root: Option>, + after: Option, + path: Vec<(Arc>, usize)>, + leaf: Option>>, + position: usize, + initialized: bool, + poisoned: bool, + positive_only: bool, +} +impl RangeIndex { + pub fn cursor( + &self, + root: Option>, + after: Option, + ) -> Result, IndexError> { + self.make_cursor(root, after, false) + } + /// Skip zero-weight subtrees and records, while preserving ordered seek. + pub fn positive_cursor( + &self, + root: Option>, + after: Option, + ) -> Result, IndexError> { + self.make_cursor(root, after, true) + } + fn make_cursor( + &self, + root: Option>, + after: Option, + positive_only: bool, + ) -> Result, IndexError> { + if let Some(root) = &root { + root.validate(self.format)?; + } + if after.as_ref().is_some_and(|oid| !oid.valid(self.format)) { + return Err(IndexError::Integrity); + } + Ok(RangeCursor { + index: self, + root, + after, + path: Vec::new(), + leaf: None, + position: 0, + initialized: false, + poisoned: false, + positive_only, + }) + } +} +impl RangeCursor<'_, R> { + async fn descend(&mut self, mut reference: NodeRef, seek: bool) -> Result<(), IndexError> { + loop { + let weight = reference.object_count; + let node = self.index.load(reference).await?; + if self.positive_only && weight == 0 { + self.leaf = None; + return Ok(()); + } + match &node.contents { + Contents::Runs(runs) => { + self.position = if seek { + self.after + .as_ref() + .map_or(0, |oid| runs.partition_point(|run| run.first_key() <= *oid)) + } else { + 0 + }; + self.leaf = Some(node); + return Ok(()); + } + Contents::Children(children) => { + let at = if seek { + self.after.as_ref().map_or(0, |oid| { + children + .partition_point(|child| child.last_key < *oid) + .min(children.len() - 1) + }) + } else { + 0 + }; + let Some(at) = (at..children.len()) + .find(|at| !self.positive_only || children[*at].object_count != 0) + else { + self.leaf = None; + return Ok(()); + }; + reference = children[at].clone(); + self.path.push((node, at)); + } + } + } + } + async fn advance_leaf(&mut self) -> Result { + while let Some((node, at)) = self.path.pop() { + let Contents::Children(children) = &node.contents else { + return Err(IndexError::Integrity); + }; + if let Some(next) = (at + 1..children.len()) + .find(|at| !self.positive_only || children[*at].object_count != 0) + { + let reference = children[next].clone(); + self.path.push((node, next)); + self.descend(reference, false).await?; + return Ok(true); + } + } + self.leaf = None; + Ok(false) + } + async fn next_inner(&mut self) -> Result, IndexError> { + if !self.initialized { + self.initialized = true; + if let Some(root) = self.root.clone() + && self.after.as_ref().is_none_or(|oid| *oid < root.last_key) + { + self.descend(root, true).await?; + } + } + loop { + let Some(leaf) = &self.leaf else { + // A seek may exhaust a positive subtree after its final live + // child. Its ancestors can still have later live siblings. + if self.advance_leaf().await? { + continue; + } + return Ok(None); + }; + let Contents::Runs(runs) = &leaf.contents else { + return Err(IndexError::Integrity); + }; + if let Some(run) = runs.get(self.position) { + self.position += 1; + if !self.positive_only || run.object_count() != 0 { + return Ok(Some(run.clone())); + } + continue; + } + if !self.advance_leaf().await? { + return Ok(None); + } + } + } + pub async fn next(&mut self) -> Result, IndexError> { + if self.poisoned { + return Err(IndexError::Integrity); + } + self.poisoned = true; + let result = self.next_inner().await; + if result.is_ok() { + self.poisoned = false; + } + result + } +} diff --git a/crates/canopy-server/src/packs/directory/index/mod.rs b/crates/canopy-server/src/packs/directory/index/mod.rs new file mode 100644 index 0000000..7fe9cf5 --- /dev/null +++ b/crates/canopy-server/src/packs/directory/index/mod.rs @@ -0,0 +1,480 @@ +//! Shared persistent bounded-fanout tree for nonoverlapping directory ranges +//! and metadata source incarnation keys. Leaf types have distinct codec domains. +//! Every path update writes only changed nodes. A selected root never requires +//! walking earlier generations. Node integrity is not a publication certificate. + +use super::*; +use cellule_runtime::codec::{BoundedDecoder, BoundedEncoder, CodecError}; +use std::{ + collections::VecDeque, + sync::atomic::{AtomicU64, Ordering}, +}; + +pub(in crate::packs) mod codec; +pub(in crate::packs) mod record; +pub use record::{IndexKey, IndexRecord}; +mod bulk; +mod cursor; +mod rewrite; +mod update; +pub use cursor::RangeCursor; + +pub const FANOUT: usize = 128; +pub const NODE_BYTES: u32 = 64 << 10; +pub const MAX_HEIGHT: u8 = 7; +const CACHE_NODES: usize = 64; +pub const MAX_RANGE_RECORDS: usize = 4096; + +#[derive(Debug, thiserror::Error)] +pub enum IndexError { + #[error("directory metadata failed")] + Metadata(#[from] MetadataError), + #[error("catalog node transfer failed")] + Artifact(#[from] canopy_object_storage::artifact::ArtifactError), + #[error("catalog node encoding failed")] + Codec(#[from] CodecError), + #[error("catalog node or reference is invalid")] + Integrity, + #[error("catalog record keys or ranges overlap")] + RangeOverlap, + #[error("the expected catalog record is absent or changed")] + Stale, + #[error("range index exceeds its bounded fanout, height or byte limit")] + Limit, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct NodeRef { + pub operation: [u8; 16], + pub artifact: ArtifactDescriptor, + pub height: u8, + pub first_key: R::Key, + pub last_key: R::Key, + pub record_count: u64, + /// Sum of record weights: represented objects for object/source records, + /// live refs for ref-state records. Object counts include overlaps between + /// shards or logical projections; never a unique-object coverage proof. + pub object_count: u64, +} +impl Copy for NodeRef where R::Key: Copy {} +impl NodeRef { + pub fn validate(&self, format: ObjectFormat) -> Result<(), IndexError> { + if self.height > R::MAX_HEIGHT + || !self.first_key.valid(format) + || !self.last_key.valid(format) + || self.first_key > self.last_key + || self.record_count == 0 + || !R::valid_counts(self.record_count, self.object_count) + || self.artifact.size == 0 + || self.artifact.size > u64::from(R::NODE_BYTES) + { + return Err(IndexError::Integrity); + } + Ok(()) + } + fn key(&self) -> ArtifactKey { + ArtifactKey { + operation: self.operation, + binding_digest: self.artifact.digest, + kind: ArtifactKind::CatalogNode, + } + } +} + +#[derive(Clone)] +enum Contents { + Runs(Vec), + Children(Vec>), +} +impl Contents { + fn len(&self) -> usize { + match self { + Self::Runs(v) => v.len(), + Self::Children(v) => v.len(), + } + } + fn is_empty(&self) -> bool { + self.len() == 0 + } + fn first(&self) -> Option { + match self { + Self::Runs(v) => v.first().map(|r| r.first_key()), + Self::Children(v) => v.first().map(|r| r.first_key.clone()), + } + } + fn last(&self) -> Option { + match self { + Self::Runs(v) => v.last().map(|r| r.last_key()), + Self::Children(v) => v.last().map(|r| r.last_key.clone()), + } + } + fn split(&mut self) -> Option { + if self.len() <= R::FANOUT { + return None; + } + let at = self.len() / 2; + Some(match self { + Self::Runs(v) => Self::Runs(v.split_off(at)), + Self::Children(v) => Self::Children(v.split_off(at)), + }) + } + fn byte_parts(self, header: usize) -> Result, IndexError> { + fn chunks( + items: Vec, + header: usize, + limit: u32, + fanout: usize, + encode: impl Fn(&T, &mut BoundedEncoder) -> Result<(), CodecError>, + ) -> Result>, IndexError> { + let capacity = (limit as usize) + .checked_sub(header) + .ok_or(IndexError::Limit)?; + let mut groups = Vec::new(); + let mut group = Vec::new(); + let mut used = 0; + for item in items { + let mut e = BoundedEncoder::new(limit)?; + encode(&item, &mut e)?; + let size = e.finish().len(); + if size > capacity { + return Err(IndexError::Limit); + } + if !group.is_empty() && (used + size > capacity || group.len() == fanout) { + groups.push(std::mem::take(&mut group)); + used = 0; + } + group.push(item); + used += size; + } + if !group.is_empty() { + groups.push(group); + } + if groups.len() < 2 { + return Err(IndexError::Limit); + } + Ok(groups) + } + match self { + Self::Runs(v) => Ok(chunks(v, header, R::NODE_BYTES, R::FANOUT, |r, e| { + r.encode_record(e) + })? + .into_iter() + .map(Self::Runs) + .collect()), + Self::Children(v) => Ok(chunks(v, header, R::NODE_BYTES, R::FANOUT, |r, e| { + codec::reference(e, r.clone()) + })? + .into_iter() + .map(Self::Children) + .collect()), + } + } +} +#[derive(Clone)] +struct Node { + repository: [u8; 16], + operation: [u8; 16], + format: ObjectFormat, + height: u8, + contents: Contents, +} +impl Node { + fn validate(&self) -> Result<(), IndexError> { + if self.contents.is_empty() + || self.contents.len() > R::FANOUT + || self.height > R::MAX_HEIGHT + { + return Err(IndexError::Limit); + } + let mut previous = None; + match &self.contents { + Contents::Runs(runs) => { + if self.height != 0 { + return Err(IndexError::Integrity); + } + for stored in runs { + stored.validate_record(self.repository, self.format)?; + if previous + .as_ref() + .is_some_and(|last| *last >= stored.first_key()) + { + return Err(IndexError::Integrity); + } + previous = Some(stored.last_key()); + } + } + Contents::Children(children) => { + if self.height == 0 { + return Err(IndexError::Integrity); + } + for child in children { + child.validate(self.format)?; + if child.height + 1 != self.height + || previous + .as_ref() + .is_some_and(|last| *last >= child.first_key) + { + return Err(IndexError::Integrity); + } + previous = Some(child.last_key.clone()); + } + } + } + let (records, weight) = self.counts()?; + if !R::valid_counts(records, weight) { + return Err(IndexError::Limit); + } + Ok(()) + } + fn counts(&self) -> Result<(u64, u64), IndexError> { + let mut count = (0_u64, 0_u64); + let mut add = |runs: u64, objects: u64| -> Result<(), IndexError> { + count.0 = count.0.checked_add(runs).ok_or(IndexError::Limit)?; + count.1 = count + .1 + .checked_add(objects) + .filter(|n| *n <= i64::MAX as u64) + .ok_or(IndexError::Limit)?; + Ok(()) + }; + match &self.contents { + Contents::Runs(runs) => { + for run in runs { + add(1, run.object_count())?; + } + } + Contents::Children(children) => { + for child in children { + add(child.record_count, child.object_count)?; + } + } + } + Ok(count) + } + fn reference(&self, artifact: ArtifactDescriptor) -> Result, IndexError> { + self.validate()?; + let (record_count, object_count) = self.counts()?; + Ok(NodeRef { + operation: self.operation, + artifact, + height: self.height, + first_key: self.contents.first().ok_or(IndexError::Integrity)?, + last_key: self.contents.last().ok_or(IndexError::Integrity)?, + record_count, + object_count, + }) + } +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ReadStats { + pub loaded_nodes: u64, + pub cache_hits: u64, +} +type CachedNode = (NodeRef, Arc>); +/// One repository/format's index client with a fixed-size node cache. Reader +/// admission and a retained generation pin are supplied by the service layer. +pub struct RangeIndex { + store: Arc, + format: ObjectFormat, + cache: Mutex>>, + loaded: AtomicU64, + hits: AtomicU64, +} +impl RangeIndex { + pub fn new(store: Arc, format: ObjectFormat) -> Self { + Self { + store, + format, + cache: Mutex::new(VecDeque::new()), + loaded: AtomicU64::new(0), + hits: AtomicU64::new(0), + } + } + pub fn repository(&self) -> [u8; 16] { + self.store.repository() + } + pub fn format(&self) -> ObjectFormat { + self.format + } + pub fn stats(&self) -> ReadStats { + ReadStats { + loaded_nodes: self.loaded.load(Ordering::Relaxed), + cache_hits: self.hits.load(Ordering::Relaxed), + } + } + pub fn clear_cache(&self) -> Result<(), IndexError> { + self.cache + .lock() + .map_err(|_| IndexError::Integrity)? + .clear(); + Ok(()) + } + /// Authenticates the selected root node and its exact summary/context. + /// Descendant existence, inventory and closure remain verifier obligations. + pub async fn validate_root(&self, root: NodeRef) -> Result<(), IndexError> { + self.load(root).await?; + Ok(()) + } + fn cache(&self, reference: NodeRef, node: Arc>) -> Result<(), IndexError> { + let mut cache = self.cache.lock().map_err(|_| IndexError::Integrity)?; + if let Some(at) = cache.iter().position(|(cached, _)| cached == &reference) { + cache.remove(at); + } + if cache.len() == CACHE_NODES { + cache.pop_front(); + } + cache.push_back((reference, node)); + Ok(()) + } + async fn load(&self, reference: NodeRef) -> Result>, IndexError> { + reference.validate(self.format)?; + { + let mut cache = self.cache.lock().map_err(|_| IndexError::Integrity)?; + if let Some(at) = cache.iter().position(|(cached, _)| cached == &reference) { + let value = cache.remove(at).ok_or(IndexError::Integrity)?; + let node = Arc::clone(&value.1); + cache.push_back(value); + self.hits.fetch_add(1, Ordering::Relaxed); + return Ok(node); + } + } + let mut read = self.store.read(reference.key(), reference.artifact).await?; + let bytes = read.next().await?.ok_or(IndexError::Integrity)?; + if read.next().await?.is_some() { + return Err(IndexError::Integrity); + } + let node = Arc::new(Node::::decode(&bytes)?); + if node.repository != self.store.repository() + || node.format != self.format + || node.reference(reference.artifact)? != reference + { + return Err(IndexError::Integrity); + } + self.loaded.fetch_add(1, Ordering::Relaxed); + self.cache(reference, Arc::clone(&node))?; + Ok(node) + } + async fn persist( + &self, + operation: [u8; 16], + height: u8, + contents: Contents, + ) -> Result, IndexError> { + let node = Arc::new(Node { + repository: self.store.repository(), + operation, + format: self.format, + height, + contents, + }); + let bytes = node.encode()?; + self.persist_encoded(node, bytes).await + } + async fn persist_encoded( + &self, + node: Arc>, + bytes: Vec, + ) -> Result, IndexError> { + let digest = *blake3::hash(&bytes).as_bytes(); + let key = ArtifactKey { + operation: node.operation, + binding_digest: digest, + kind: ArtifactKind::CatalogNode, + }; + let artifact = self + .store + .put(key, bytes.len() as u64, digest, &mut bytes.as_slice()) + .await?; + let reference = node.reference(artifact)?; + self.cache(reference.clone(), node)?; + Ok(reference) + } + /// First run whose last OID is at least `oid`. It may start after `oid`, so + /// callers use this to detect overlap even when both endpoints lie in gaps. + pub async fn successor( + &self, + root: Option>, + oid: R::Key, + ) -> Result, IndexError> { + if !oid.valid(self.format) { + return Ok(None); + } + let Some(mut reference) = root else { + return Ok(None); + }; + reference.validate(self.format)?; + if oid > reference.last_key { + return Ok(None); + } + loop { + let node = self.load(reference).await?; + match &node.contents { + Contents::Runs(runs) => { + return Ok(runs + .get(runs.partition_point(|run| run.last_key() < oid)) + .cloned()); + } + Contents::Children(children) => { + let at = children.partition_point(|child| child.last_key < oid); + reference = children.get(at).ok_or(IndexError::Integrity)?.clone(); + } + } + } + } + /// All records intersecting an inclusive interval, including an enclosing + /// record whose first key precedes the interval. Fail rather than truncate + /// when the caller's bounded selection budget is exceeded. Seeking reads + /// one path; subsequent cursor work is proportional to intersecting leaves. + pub async fn overlapping( + &self, + root: Option>, + first: R::Key, + last: R::Key, + limit: usize, + ) -> Result, IndexError> { + if !first.valid(self.format) || !last.valid(self.format) || first > last { + return Err(IndexError::Integrity); + } + if limit > MAX_RANGE_RECORDS { + return Err(IndexError::Limit); + } + let mut selected = Vec::new(); + let Some(start) = self + .successor(root.clone(), first) + .await? + .filter(|run| run.first_key() <= last) + else { + return Ok(selected); + }; + if limit == 0 { + return Err(IndexError::Limit); + } + let after = start.first_key(); + selected.push(start); + let mut cursor = self.cursor(root, Some(after))?; + while let Some(run) = cursor.next().await? { + if run.first_key() > last { + break; + } + if selected.len() == limit { + return Err(IndexError::Limit); + } + selected.push(run); + } + Ok(selected) + } + pub async fn find( + &self, + root: Option>, + oid: R::Key, + ) -> Result, IndexError> { + Ok(self + .successor(root, oid.clone()) + .await? + .filter(|run| run.first_key() <= oid)) + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/canopy-server/src/packs/directory/index/record.rs b/crates/canopy-server/src/packs/directory/index/record.rs new file mode 100644 index 0000000..a234936 --- /dev/null +++ b/crates/canopy-server/src/packs/directory/index/record.rs @@ -0,0 +1,93 @@ +use super::*; + +pub(in crate::packs) mod sealed { + pub trait Key {} + pub trait Record {} +} + +/// Persisted key types are sealed: their byte order and validation are part of +/// the catalog contract. Source incarnation keys are not fabricated Git OIDs. +pub trait IndexKey: Clone + Ord + std::fmt::Debug + Send + Sync + 'static + sealed::Key { + fn valid(&self, format: ObjectFormat) -> bool; + fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError>; + fn decode(decoder: &mut BoundedDecoder<'_>, format: ObjectFormat) -> Result; +} + +/// One immutable leaf descriptor in the shared path-copy range tree. +pub trait IndexRecord: + Clone + Eq + std::fmt::Debug + Send + Sync + 'static + sealed::Record +{ + type Key: IndexKey; + const FANOUT: usize; + const DOMAIN: &'static [u8]; + const NODE_BYTES: u32 = super::NODE_BYTES; + const MAX_HEIGHT: u8 = super::MAX_HEIGHT; + fn valid_counts(records: u64, weight: u64) -> bool { + weight >= records && weight <= i64::MAX as u64 + } + fn first_key(&self) -> Self::Key; + fn last_key(&self) -> Self::Key; + fn object_count(&self) -> u64; + fn validate_record(&self, repository: [u8; 16], format: ObjectFormat) + -> Result<(), IndexError>; + fn encode_record(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError>; + fn decode_record( + decoder: &mut BoundedDecoder<'_>, + repository: [u8; 16], + format: ObjectFormat, + ) -> Result; +} + +impl sealed::Key for ObjectId {} +impl IndexKey for ObjectId { + fn valid(&self, format: ObjectFormat) -> bool { + self.format() == format && !self.is_zero() + } + fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { + encoder.write_bytes(self) + } + fn decode(decoder: &mut BoundedDecoder<'_>, format: ObjectFormat) -> Result { + let oid = ObjectId::try_from(decoder.read_bytes()?) + .map_err(|_| CodecError::Invalid("invalid catalog OID"))?; + if !oid.valid(format) { + return Err(CodecError::Invalid("invalid catalog OID format or zero ID")); + } + Ok(oid) + } +} +impl sealed::Record for StoredRun {} +impl IndexRecord for StoredRun { + type Key = ObjectId; + const FANOUT: usize = FANOUT; + const DOMAIN: &'static [u8] = b"canopy.range-index.v2\0"; + fn first_key(&self) -> ObjectId { + self.coverage.first_oid + } + fn last_key(&self) -> ObjectId { + self.coverage.last_oid + } + fn object_count(&self) -> u64 { + self.coverage.object_count + } + fn validate_record( + &self, + repository: [u8; 16], + format: ObjectFormat, + ) -> Result<(), IndexError> { + self.validate()?; + if self.run.repository != repository || self.run.format != format { + return Err(IndexError::Integrity); + } + Ok(()) + } + fn encode_record(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { + codec::write_run(encoder, *self) + } + fn decode_record( + decoder: &mut BoundedDecoder<'_>, + repository: [u8; 16], + format: ObjectFormat, + ) -> Result { + codec::read_run(decoder, repository, format) + } +} diff --git a/crates/canopy-server/src/packs/directory/index/rewrite.rs b/crates/canopy-server/src/packs/directory/index/rewrite.rs new file mode 100644 index 0000000..a780807 --- /dev/null +++ b/crates/canopy-server/src/packs/directory/index/rewrite.rs @@ -0,0 +1,136 @@ +//! Conditional sorted upserts. Only affected paths are loaded; unchanged +//! authenticated child references join the shared streaming builder directly. +use super::*; +use bulk::Builder; + +struct Changes { + input: I, + pending: Option, + last: Option, + done: bool, + repository: [u8; 16], + format: ObjectFormat, +} +impl>, R: IndexRecord> Changes { + fn peek(&mut self) -> Result, IndexError> { + if self.pending.is_none() && !self.done { + if let Some(record) = self.input.next() { + let record = record?; + record.validate_record(self.repository, self.format)?; + if self + .last + .as_ref() + .is_some_and(|last| *last >= record.first_key()) + { + return Err(IndexError::RangeOverlap); + } + self.last = Some(record.last_key()); + self.pending = Some(record); + } else { + self.done = true; + } + } + Ok(self.pending.as_ref()) + } + fn within(&mut self, upper: Option<&R::Key>) -> Result { + Ok(self + .peek()? + .is_some_and(|record| upper.is_none_or(|upper| record.first_key() <= *upper))) + } + fn pop(&mut self) -> Result { + self.peek()?; + self.pending.take().ok_or(IndexError::Integrity) + } +} +impl RangeIndex { + /// Upsert sorted exact-key/range records through one streaming rewrite. + /// Ref callers validate every expectation/namespace against this immutable + /// base first. This structural API confers no publication authority. + pub async fn upsert_sorted( + &self, + root: Option>, + operation: [u8; 16], + records: I, + ) -> Result>, IndexError> + where + I: IntoIterator>, + I::IntoIter: Send, + { + let Some(root) = root else { + return self.build_sorted(operation, records).await; + }; + self.validate_root(root.clone()).await?; + let mut changes = Changes { + input: records.into_iter(), + pending: None, + last: None, + done: false, + repository: self.repository(), + format: self.format, + }; + if changes.peek()?.is_none() { + return Ok(Some(root)); + } + let mut builder = Builder::new(self, operation)?; + self.rewrite_walk(root, None, &mut changes, &mut builder) + .await?; + if changes.peek()?.is_some() { + return Err(IndexError::Integrity); + } + builder.finish().await + } + async fn rewrite_walk> + Send>( + &self, + root: NodeRef, + upper: Option, + changes: &mut Changes, + builder: &mut Builder<'_, R>, + ) -> Result<(), IndexError> { + if !changes.within(upper.as_ref())? { + return builder.subtree(root).await; + } + let node = self.load(root).await?; + match &node.contents { + Contents::Runs(records) => { + for old in records { + while changes.within(upper.as_ref())? + && changes + .peek()? + .is_some_and(|new| new.first_key() < old.first_key()) + { + builder.record(changes.pop()?).await?; + } + if changes.within(upper.as_ref())? + && changes + .peek()? + .is_some_and(|new| new.first_key() == old.first_key()) + { + let new = changes.pop()?; + if new.last_key() != old.last_key() { + return Err(IndexError::RangeOverlap); + } + builder.record(new).await?; + } else { + builder.record(old.clone()).await?; + } + } + while changes.within(upper.as_ref())? { + builder.record(changes.pop()?).await?; + } + } + Contents::Children(children) => { + for (at, child) in children.iter().enumerate() { + // Route gaps to the next child; the last child inherits the + // enclosing interval so insertions can extend its old fence. + let bound = if at + 1 == children.len() { + upper.clone() + } else { + Some(child.last_key.clone()) + }; + Box::pin(self.rewrite_walk(child.clone(), bound, changes, builder)).await?; + } + } + } + Ok(()) + } +} diff --git a/crates/canopy-server/src/packs/directory/index/tests.rs b/crates/canopy-server/src/packs/directory/index/tests.rs new file mode 100644 index 0000000..74d77cf --- /dev/null +++ b/crates/canopy-server/src/packs/directory/index/tests.rs @@ -0,0 +1,506 @@ +use super::*; +use object_store::{ObjectStore, ObjectStoreExt, memory::InMemory}; + +type Result = std::result::Result>; +fn oid(n: u64, format: ObjectFormat) -> ObjectId { + let mut bytes = vec![0; format.bytes()]; + bytes[..8].copy_from_slice(&n.to_be_bytes()); + bytes.try_into().unwrap() +} +// These fixtures test the descriptor index, not native object verification or +// publication. Their directory files are intentionally not populated. +fn run(n: u64, format: ObjectFormat) -> StoredRun { + let digest = *blake3::hash(&n.to_be_bytes()).as_bytes(); + let artifact = ArtifactDescriptor { + size: 16 << 10, + digest, + manifest_digest: [3; 32], + }; + StoredRun { + run: RunDescriptor { + repository: [1; 16], + operation: [2; 16], + format, + object_count: 2, + first_oid: oid(3 * n + 1, format), + last_oid: oid(3 * n + 2, format), + inventory_digest: [4; 32], + size: artifact.size, + digest, + }, + artifact, + coverage: RunCoverage { + object_count: 2, + first_oid: oid(3 * n + 1, format), + last_oid: oid(3 * n + 2, format), + inventory_digest: [4; 32], + }, + } +} +fn index(format: ObjectFormat) -> (RangeIndex, Arc) { + let store: Arc = Arc::new(InMemory::new()); + ( + RangeIndex::new( + Arc::new(ArtifactStore::new(Arc::clone(&store), [1; 16])), + format, + ), + store, + ) +} + +#[tokio::test] +async fn incremental_split_and_removal_preserve_old_roots_for_both_formats() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (index, _store) = index(format); + let mut root = None; + let count = FANOUT + 20; + for n in 0..count { + root = Some(index.insert(root, [5; 16], run(n as u64, format)).await?); + } + let old = root.ok_or("root")?; + assert_eq!(old.height, 1); + assert_eq!(old.record_count, count as u64); + assert_eq!(old.object_count, 2 * count as u64); + assert_eq!(index.insert(root, [6; 16], run(0, format)).await?, old); + let mut cursor = index.cursor(root, None)?; + for n in 0..count { + assert_eq!(cursor.next().await?, Some(run(n as u64, format))); + } + assert!(cursor.next().await?.is_none()); + let missing = run(127, format); + root = index.remove(root, [7; 16], missing).await?; + assert!(index.find(root, missing.run.first_oid).await?.is_none()); + assert_eq!( + index.find(Some(old), missing.run.first_oid).await?, + Some(missing) + ); + assert!(matches!( + index.remove(root, [7; 16], missing).await, + Err(IndexError::Stale) + )); + let mut stale = run(0, format); + stale.artifact.manifest_digest[0] ^= 1; + assert!(matches!( + index.remove(root, [7; 16], stale).await, + Err(IndexError::Stale) + )); + for n in 0..count { + if n != 127 { + root = index.remove(root, [8; 16], run(n as u64, format)).await?; + } + } + assert!(root.is_none()); + assert_eq!( + index + .find(Some(old), run((count - 1) as u64, format).run.first_oid) + .await?, + Some(run((count - 1) as u64, format)) + ); + assert!(index.cache.lock().unwrap().len() <= CACHE_NODES); + } + Ok(()) +} + +#[tokio::test] +async fn overlap_detection_includes_ranges_enclosing_existing_runs() -> Result { + let format = ObjectFormat::Sha256; + let (index, _store) = index(format); + let root = Some(index.insert(None, [5; 16], run(20, format)).await?); + let mut enclosing = run(0, format); + enclosing.run.first_oid = oid(1, format); + enclosing.run.last_oid = oid(100, format); + enclosing.coverage = enclosing.run.coverage(); + assert!(index.find(root, enclosing.run.first_oid).await?.is_none()); + assert!(index.find(root, enclosing.run.last_oid).await?.is_none()); + assert!(matches!( + index.insert(root, [6; 16], enclosing).await, + Err(IndexError::RangeOverlap) + )); + let mut touching = run(19, format); + touching.run.last_oid = run(20, format).run.first_oid; + touching.coverage = touching.run.coverage(); + assert!(matches!( + index.insert(root, [6; 16], touching).await, + Err(IndexError::RangeOverlap) + )); + let before = index.insert(root, [6; 16], run(0, format)).await?; + let after = index.insert(Some(before), [6; 16], run(40, format)).await?; + assert_eq!(after.record_count, 3); + Ok(()) +} + +#[tokio::test] +async fn cold_point_lookup_reads_only_one_bounded_path_and_cursor_seeks() -> Result { + let format = ObjectFormat::Sha1; + let (index, _store) = index(format); + let runs = (0..16).map(|n| run(n, format)).collect::>(); + let mut level = Vec::new(); + for chunk in runs.chunks(2) { + level.push( + index + .persist([5; 16], 0, Contents::Runs(chunk.to_vec())) + .await?, + ); + } + let mut height = 0; + while level.len() > 1 { + height += 1; + let mut next = Vec::new(); + for chunk in level.chunks(2) { + next.push( + index + .persist([5; 16], height, Contents::Children(chunk.to_vec())) + .await?, + ); + } + level = next; + } + let root = level[0]; + assert_eq!(root.height, 3); + index.clear_cache()?; + let before = index.stats(); + assert_eq!( + index.find(Some(root), runs[10].run.first_oid).await?, + Some(runs[10]) + ); + let after = index.stats(); + assert_eq!( + after.loaded_nodes - before.loaded_nodes, + u64::from(root.height) + 1 + ); + assert_eq!( + index.find(Some(root), runs[10].run.last_oid).await?, + Some(runs[10]) + ); + assert_eq!(index.stats().loaded_nodes, after.loaded_nodes); + assert!(index.stats().cache_hits > after.cache_hits); + assert!(index.find(Some(root), oid(33, format)).await?.is_none()); + let mut cursor = index.cursor(Some(root), Some(runs[8].run.first_oid))?; + for expected in runs.iter().skip(9) { + assert_eq!(cursor.next().await?, Some(*expected)); + } + assert!(cursor.next().await?.is_none()); + let mut cursor = index.cursor(Some(root), Some(root.last_key))?; + assert!(cursor.next().await?.is_none()); + Ok(()) +} + +#[test] +fn node_codec_rejects_bad_count_ranges_height_framing_and_large_input() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let node = Node { + repository: [1; 16], + operation: [5; 16], + format, + height: 0, + contents: Contents::Runs(vec![run(0, format), run(1, format)]), + }; + let bytes = node.encode()?; + let decoded = Node::::decode(&bytes)?; + assert_eq!(decoded.encode()?, bytes); + for length in [0, 1, bytes.len() - 1] { + assert!(Node::::decode(&bytes[..length]).is_err()); + } + let mut trailing = bytes.clone(); + trailing.push(0); + assert!(Node::::decode(&trailing).is_err()); + assert!(Node::::decode(&vec![0; NODE_BYTES as usize + 1]).is_err()); + let count_at = 4 + b"canopy.range-index.v2\0".len() + 4 + 16 + 4 + 16 + 1 + 1; + for count in [0_u32, FANOUT as u32 + 1, u32::MAX] { + let mut bad = bytes.clone(); + bad[count_at..count_at + 4].copy_from_slice(&count.to_be_bytes()); + assert!(Node::::decode(&bad).is_err()); + } + let mut reversed = node.clone(); + reversed.contents = Contents::Runs(vec![run(1, format), run(0, format)]); + assert!(reversed.encode().is_err()); + let mut overlap = node.clone(); + overlap.contents = Contents::Runs(vec![run(0, format), run(0, format)]); + assert!(overlap.encode().is_err()); + let mut height = node.clone(); + height.height = MAX_HEIGHT + 1; + assert!(height.encode().is_err()); + let many = Node { + contents: Contents::Runs((0..FANOUT).map(|n| run(n as u64, format)).collect()), + ..node + }; + assert!(many.encode()?.len() <= NODE_BYTES as usize); + } + Ok(()) +} + +#[tokio::test] +async fn authenticated_node_bytes_and_reference_summaries_are_checked_before_lookup() -> Result { + let format = ObjectFormat::Sha256; + let (index, store) = index(format); + let reference = index.insert(None, [5; 16], run(0, format)).await?; + let mut forged = reference; + forged.object_count += 1; + assert!(matches!( + index.find(Some(forged), run(0, format).run.first_oid).await, + Err(IndexError::Integrity) + )); + let path = index + .store + .path(reference.key(), reference.artifact.digest)?; + store + .put( + &canopy_object_storage::external::part(&path, 0), + bytes::Bytes::from(vec![0; reference.artifact.size as usize]).into(), + ) + .await?; + index.clear_cache()?; + assert!( + index + .find(Some(reference), run(0, format).run.first_oid) + .await + .is_err() + ); + let mut cursor = index.cursor(Some(reference), None)?; + assert!(cursor.next().await.is_err()); + assert!(matches!(cursor.next().await, Err(IndexError::Integrity))); + Ok(()) +} + +#[tokio::test] +async fn repository_operation_and_object_format_binding_cannot_be_forged() -> Result { + let format = ObjectFormat::Sha256; + let (index, _store) = index(format); + let mut foreign = run(0, format); + foreign.run.repository = [9; 16]; + assert!(matches!( + index.insert(None, [5; 16], foreign).await, + Err(IndexError::Integrity) + )); + assert!(matches!( + index + .insert(None, [5; 16], run(0, ObjectFormat::Sha1)) + .await, + Err(IndexError::Integrity) + )); + let node = Node { + repository: [9; 16], + operation: [5; 16], + format, + height: 0, + contents: Contents::Runs(vec![foreign]), + }; + let bytes = node.encode()?; + let digest = *blake3::hash(&bytes).as_bytes(); + let key = ArtifactKey { + operation: [5; 16], + binding_digest: digest, + kind: ArtifactKind::CatalogNode, + }; + let artifact = index + .store + .put(key, bytes.len() as u64, digest, &mut bytes.as_slice()) + .await?; + let reference = node.reference(artifact)?; + assert!(matches!( + index.find(Some(reference), foreign.run.first_oid).await, + Err(IndexError::Integrity) + )); + Ok(()) +} + +#[test] +fn source_nodes_use_typed_keys_separate_domains_and_bounded_leaf_codecs() -> Result { + use crate::packs::sources::{SOURCE_FANOUT, SourceRecord, tests::source}; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let node = Node:: { + repository: [1; 16], + operation: [5; 16], + format, + height: 0, + contents: Contents::Runs( + (0..SOURCE_FANOUT) + .map(|n| source(n as u64, format)) + .collect(), + ), + }; + let bytes = node.encode()?; + assert!(bytes.len() <= NODE_BYTES as usize); + assert_eq!(Node::::decode(&bytes)?.encode()?, bytes); + assert!(Node::::decode(&bytes).is_err()); + let directory = Node:: { + repository: [1; 16], + operation: [5; 16], + format, + height: 0, + contents: Contents::Runs(vec![run(0, format)]), + }; + assert!(Node::::decode(&directory.encode()?).is_err()); + for length in [0, 1, bytes.len() - 1] { + assert!(Node::::decode(&bytes[..length]).is_err()); + } + let mut trailing = bytes.clone(); + trailing.push(0); + assert!(Node::::decode(&trailing).is_err()); + let count_at = 4 + SourceRecord::DOMAIN.len() + 4 + 16 + 4 + 16 + 1 + 1; + for count in [0_u32, SOURCE_FANOUT as u32 + 1, u32::MAX] { + let mut bad = bytes.clone(); + bad[count_at..count_at + 4].copy_from_slice(&count.to_be_bytes()); + assert!(Node::::decode(&bad).is_err()); + } + assert!(Node::::decode(&vec![0; NODE_BYTES as usize + 1]).is_err()); + let mut unsorted = node.clone(); + let Contents::Runs(records) = &mut unsorted.contents else { + unreachable!() + }; + records.reverse(); + assert!(unsorted.encode().is_err()); + for width in [0_usize, 47, 49, 64] { + let mut encoder = BoundedEncoder::new(128)?; + encoder.write_bytes(&vec![0; width])?; + let bytes = encoder.finish(); + let mut decoder = BoundedDecoder::new(&bytes, 128)?; + assert!(SegmentKey::decode(&mut decoder, format).is_err()); + } + } + Ok(()) +} + +#[tokio::test] +async fn bounded_overlap_seek_includes_enclosing_ranges_and_rejects_truncation() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (index, _store) = index(format); + let mut root = None; + for n in 0..600 { + root = Some(index.insert(root, [5; 16], run(n, format)).await?); + } + index.clear_cache()?; + let before = index.stats(); + assert_eq!( + index + .overlapping(root, oid(898, format), oid(904, format), 3) + .await?, + vec![run(299, format), run(300, format), run(301, format)] + ); + assert!(index.stats().loaded_nodes - before.loaded_nodes <= 4); + // Endpoint inside a run, endpoint in a gap, exact touching endpoints. + assert_eq!( + index + .overlapping(root, oid(901, format), oid(901, format), 1) + .await?, + vec![run(300, format)] + ); + assert_eq!( + index + .overlapping(root, oid(900, format), oid(900, format), 0) + .await?, + vec![] + ); + assert!(matches!( + index + .overlapping(root, oid(898, format), oid(904, format), 2) + .await, + Err(IndexError::Limit) + )); + assert!(matches!( + index + .overlapping(root, oid(901, format), oid(902, format), 0) + .await, + Err(IndexError::Limit) + )); + assert!(matches!( + index + .overlapping(root, oid(904, format), oid(898, format), 3) + .await, + Err(IndexError::Integrity) + )); + assert!(matches!( + index + .overlapping(root, oid(1, format), oid(2, format), MAX_RANGE_RECORDS + 1) + .await, + Err(IndexError::Limit) + )); + assert!( + index + .overlapping(root, oid(1801, format), oid(1802, format), 0) + .await? + .is_empty() + ); + assert!( + index + .overlapping(None, oid(1, format), oid(2, format), 0) + .await? + .is_empty() + ); + } + Ok(()) +} + +#[test] +fn covered_run_codec_matches_independent_vectors_and_rejects_old_leaf_domain() -> Result { + use cellule_runtime::codec::{BoundedDecoder, BoundedEncoder}; + for (format, golden) in [ + ( + ObjectFormat::Sha1, + include_str!(concat!( + env!("CARGO_MANIFEST_DIR"), + "/../../docs/design/directory-run-v2-sha1.hex" + )), + ), + ( + ObjectFormat::Sha256, + include_str!(concat!( + env!("CARGO_MANIFEST_DIR"), + "/../../docs/design/directory-run-v2-sha256.hex" + )), + ), + ] { + // Codec-only descriptors: this never asserts native inventory/closure. + let stored = StoredRun { + run: RunDescriptor { + repository: [1; 16], + operation: [2; 16], + format, + object_count: 10, + first_oid: oid(1, format), + last_oid: oid(20, format), + inventory_digest: [4; 32], + size: 16 << 10, + digest: [5; 32], + }, + artifact: ArtifactDescriptor { + size: 16 << 10, + digest: [5; 32], + manifest_digest: [6; 32], + }, + coverage: RunCoverage { + object_count: 2, + first_oid: oid(4, format), + last_oid: oid(7, format), + inventory_digest: [9; 32], + }, + }; + stored.validate()?; + let expected = hex::decode(golden.trim())?; + let mut encoder = BoundedEncoder::new(1024)?; + codec::write_run(&mut encoder, stored)?; + assert_eq!(encoder.finish(), expected); + let mut decoder = BoundedDecoder::new(&expected, 1024)?; + assert_eq!(codec::read_run(&mut decoder, [1; 16], format)?, stored); + decoder.finish()?; + let node = Node { + repository: [1; 16], + operation: [3; 16], + format, + height: 0, + contents: Contents::Runs(vec![stored]), + }; + let mut old = node.encode()?; + let domain = b"canopy.range-index.v2\0"; + let at = old + .windows(domain.len()) + .position(|bytes| bytes == domain) + .ok_or("domain")?; + old[at + domain.len() - 2] = b'1'; + assert!(matches!( + Node::::decode(&old), + Err(IndexError::Integrity) + )); + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/directory/index/update.rs b/crates/canopy-server/src/packs/directory/index/update.rs new file mode 100644 index 0000000..bd5d26b --- /dev/null +++ b/crates/canopy-server/src/packs/directory/index/update.rs @@ -0,0 +1,216 @@ +use super::*; + +impl RangeIndex { + async fn path( + &self, + root: NodeRef, + oid: R::Key, + ) -> Result<(Vec<(Arc>, usize)>, Arc>), IndexError> { + let mut path = Vec::new(); + let mut reference = root; + loop { + let node = self.load(reference).await?; + match &node.contents { + Contents::Runs(_) => return Ok((path, node)), + Contents::Children(children) => { + let at = children + .partition_point(|child| child.last_key < oid) + .min(children.len() - 1); + reference = children[at].clone(); + path.push((node, at)); + } + } + } + } + async fn persist_split( + &self, + operation: [u8; 16], + height: u8, + contents: Contents, + ) -> Result>, IndexError> { + let mut pending = vec![contents]; + let mut references = Vec::new(); + while let Some(mut contents) = pending.pop() { + if let Some(second) = contents.split() { + pending.push(second); + pending.push(contents); + continue; + } + let node = Arc::new(Node { + repository: self.repository(), + operation, + format: self.format, + height, + contents, + }); + match node.encode() { + Ok(bytes) => references.push(self.persist_encoded(node, bytes).await?), + Err(IndexError::Codec(CodecError::Limit)) => { + let header = node.header_size()?; + let node = Arc::try_unwrap(node).map_err(|_| IndexError::Integrity)?; + pending.extend(node.contents.byte_parts(header)?.into_iter().rev()); + } + Err(error) => return Err(error), + } + } + Ok(references) + } + /// Replace one exact same-key incarnation in one path copy. This is a + /// structural CAS; current authority and root publication are separate. + pub async fn replace( + &self, + root: NodeRef, + operation: [u8; 16], + expected: R, + replacement: R, + ) -> Result, IndexError> { + expected.validate_record(self.repository(), self.format)?; + replacement.validate_record(self.repository(), self.format)?; + if expected.first_key() != replacement.first_key() + || expected.last_key() != replacement.last_key() + { + return Err(IndexError::RangeOverlap); + } + let (path, leaf) = self.path(root.clone(), expected.first_key()).await?; + let Contents::Runs(mut runs) = leaf.contents.clone() else { + return Err(IndexError::Integrity); + }; + let at = runs.partition_point(|run| run.first_key() < expected.first_key()); + if runs.get(at) != Some(&expected) { + return Err(IndexError::Stale); + } + if expected == replacement { + return Ok(root); + } + runs[at] = replacement; + let mut changed = self + .persist_split(operation, 0, Contents::Runs(runs)) + .await?; + for (parent, at) in path.into_iter().rev() { + let Contents::Children(mut children) = parent.contents.clone() else { + return Err(IndexError::Integrity); + }; + children.splice(at..=at, changed); + changed = self + .persist_split(operation, parent.height, Contents::Children(children)) + .await?; + } + if changed.len() == 1 { + return Ok(changed.remove(0)); + } + let height = root + .height + .checked_add(1) + .filter(|height| *height <= R::MAX_HEIGHT) + .ok_or(IndexError::Limit)?; + self.persist(operation, height, Contents::Children(changed)) + .await + } + /// Structural path-copy insertion. A trusted compaction/publication verifier + /// still certifies the run's headers, dependencies and artifact existence. + pub async fn insert( + &self, + root: Option>, + operation: [u8; 16], + run: R, + ) -> Result, IndexError> { + run.validate_record(self.store.repository(), self.format)?; + if let Some(existing) = self.successor(root.clone(), run.first_key()).await? { + if existing == run { + return root.ok_or(IndexError::Integrity); + } + if existing.first_key() <= run.last_key() { + return Err(IndexError::RangeOverlap); + } + } + let Some(root) = root else { + return self.persist(operation, 0, Contents::Runs(vec![run])).await; + }; + let (path, leaf) = self.path(root.clone(), run.first_key()).await?; + let Contents::Runs(mut runs) = leaf.contents.clone() else { + return Err(IndexError::Integrity); + }; + let at = runs.partition_point(|existing| existing.first_key() < run.first_key()); + runs.insert(at, run); + let mut changed = self + .persist_split(operation, 0, Contents::Runs(runs)) + .await?; + for (parent, at) in path.into_iter().rev() { + let Contents::Children(mut children) = parent.contents.clone() else { + return Err(IndexError::Integrity); + }; + children.splice(at..=at, changed); + changed = self + .persist_split(operation, parent.height, Contents::Children(children)) + .await?; + } + if changed.len() == 1 { + return Ok(changed[0].clone()); + } + let height = root + .height + .checked_add(1) + .filter(|height| *height <= R::MAX_HEIGHT) + .ok_or(IndexError::Limit)?; + self.persist(operation, height, Contents::Children(changed)) + .await + } + /// Remove exactly the expected incarnation. Old roots remain immutable and + /// usable for retained readers until the service releases their generation. + pub async fn remove( + &self, + root: Option>, + operation: [u8; 16], + expected: R, + ) -> Result>, IndexError> { + expected.validate_record(self.store.repository(), self.format)?; + let root = root.ok_or(IndexError::Stale)?; + let (path, leaf) = self.path(root, expected.first_key()).await?; + let Contents::Runs(mut runs) = leaf.contents.clone() else { + return Err(IndexError::Integrity); + }; + let at = runs.partition_point(|run| run.first_key() < expected.first_key()); + if runs.get(at) != Some(&expected) { + return Err(IndexError::Stale); + } + runs.remove(at); + let mut changed = if runs.is_empty() { + None + } else { + Some(self.persist(operation, 0, Contents::Runs(runs)).await?) + }; + let depth = path.len(); + for (n, (parent, at)) in path.into_iter().rev().enumerate() { + let Contents::Children(mut children) = parent.contents.clone() else { + return Err(IndexError::Integrity); + }; + children.splice(at..=at, changed); + changed = if children.is_empty() { + None + } else if n + 1 == depth && children.len() == 1 { + Some(children[0].clone()) + } else { + Some( + self.persist(operation, parent.height, Contents::Children(children)) + .await?, + ) + }; + } + // A retained unary subtree may become the root after removal. Collapse + // it without changing any run or rewriting all remaining descriptors. + while let Some(reference) = changed.clone() { + if reference.height == 0 { + break; + } + let node = self.load(reference).await?; + let Contents::Children(children) = &node.contents else { + return Err(IndexError::Integrity); + }; + if children.len() != 1 { + break; + } + changed = Some(children[0].clone()); + } + Ok(changed) + } +} diff --git a/crates/canopy-server/src/packs/directory/mod.rs b/crates/canopy-server/src/packs/directory/mod.rs new file mode 100644 index 0000000..6a2f9d5 --- /dev/null +++ b/crates/canopy-server/src/packs/directory/mod.rs @@ -0,0 +1,308 @@ +//! Immutable canonical object directory runs. A published leveled index chooses +//! at most one run per nonoverlapping level, plus bounded overlapping level 0. +//! Runs bind source descriptors, but do not themselves attest graph closure. + +use super::metadata::{ + self, AdmittedFile, MetadataError, MetadataLimits, MetadataSegment, ObjectHeader, PAGE_OBJECTS, + ReaderAdmission, digest, file_digest, fold_header, header, + transport::{PinnedFile, download_file_for_reader, upload_file}, +}; +use crate::{ObjectFormat, ObjectId}; +use canopy_object_storage::artifact::{ + ArtifactDescriptor, ArtifactKey, ArtifactKind, ArtifactStore, +}; +use cellule_ltx::{DiskBudget, DiskReservation}; +use rusqlite::{Connection, OpenFlags, OptionalExtension, params}; +use std::{ + path::Path, + sync::{Arc, Mutex}, +}; + +mod coverage; +pub use coverage::RunCoverage; +mod writer; +pub use writer::DirectoryBuilder; +mod partition; +pub use partition::DirectoryPartitioner; +pub mod index; +pub mod snapshot; + +pub const RUN_TARGET_BYTES: u64 = 64 << 20; +const APPLICATION_ID: u32 = 1_128_353_358; +const SCHEMA: &str = include_str!("schema.sql"); + +/// Identifies a metadata artifact incarnation, not only its content digest. +#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord)] +pub struct SegmentKey { + pub operation: [u8; 16], + pub digest: [u8; 32], +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct DirectoryEntry { + pub header: ObjectHeader, + pub source: SegmentKey, + /// Changes only when a trusted compaction switches physical placement. + /// Canonical inventory digests deliberately exclude this version. + pub location_version: u64, +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct RunDescriptor { + pub repository: [u8; 16], + pub operation: [u8; 16], + pub format: ObjectFormat, + pub object_count: u64, + pub first_oid: ObjectId, + pub last_oid: ObjectId, + pub inventory_digest: [u8; 32], + pub size: u64, + pub digest: [u8; 32], +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct StoredRun { + /// Complete authenticated physical file identity; projections never change it. + pub run: RunDescriptor, + pub artifact: ArtifactDescriptor, + /// Exact contiguous OID coverage indexed in this catalog. Always present, + /// including whole-file runs; no legacy/sliced representation switch. + pub coverage: RunCoverage, +} +impl RunDescriptor { + pub fn validate(self) -> Result<(), MetadataError> { + if self.object_count == 0 + || self.object_count > i64::MAX as u64 + || self.first_oid.format() != self.format + || self.last_oid.format() != self.format + || self.first_oid.is_zero() + || self.first_oid > self.last_oid + || (self.object_count == 1) != (self.first_oid == self.last_oid) + || self.size < 12 << 10 + || !self.size.is_multiple_of(4096) + || self.size > canopy_object_storage::external::MAX_ARTIFACT_BYTES + { + return Err(MetadataError::Integrity); + } + Ok(()) + } + fn key(self) -> ArtifactKey { + ArtifactKey { + operation: self.operation, + binding_digest: self.digest, + kind: ArtifactKind::DirectoryRun, + } + } +} +impl StoredRun { + pub fn validate(self) -> Result<(), MetadataError> { + self.run.validate()?; + self.coverage.validate(self.run)?; + if self.run.size != self.artifact.size || self.run.digest != self.artifact.digest { + return Err(MetadataError::Integrity); + } + Ok(()) + } +} +fn entry(row: &rusqlite::Row<'_>) -> rusqlite::Result { + Ok(DirectoryEntry { + header: header(row)?, + source: SegmentKey { + operation: row + .get::<_, Vec>(6)? + .try_into() + .map_err(|_| rusqlite::Error::InvalidQuery)?, + digest: digest(row.get(7)?)?, + }, + location_version: metadata::unsigned(row.get(8)?)?, + }) +} +pub(super) fn inventory_seed(format: ObjectFormat) -> [u8; 32] { + let mut hash = blake3::Hasher::new(); + hash.update(b"canopy.directory.v1\0"); + hash.update(&[format.bytes() as u8]); + *hash.finalize().as_bytes() +} + +/// A verified private file, bounded SQLite page cache, and owned disk charge. +/// The connection closes before file cleanup; cancellation retains file pins. +pub struct DirectoryRun { + connection: Mutex, + admitted: AdmittedFile, + descriptor: RunDescriptor, +} +impl PinnedFile for DirectoryRun { + fn open(&self) -> std::io::Result { + std::fs::File::open(self.path()) + } +} +impl DirectoryRun { + pub fn open( + file: tempfile::NamedTempFile, + reservation: DiskReservation, + descriptor: RunDescriptor, + cache_kib: u32, + ) -> Result { + Self::open_admitted(AdmittedFile::new(file, reservation), descriptor, cache_kib) + } + fn open_admitted( + mut admitted: AdmittedFile, + descriptor: RunDescriptor, + cache_kib: u32, + ) -> Result { + descriptor.validate()?; + if cache_kib == 0 + || cache_kib > i32::MAX as u32 + || descriptor.size > admitted.reservation().bytes() + || file_digest(admitted.file().path(), descriptor.size)? != descriptor.digest + { + return Err(MetadataError::Integrity); + } + let connection = Connection::open_with_flags( + admitted.file().path(), + OpenFlags::SQLITE_OPEN_READ_ONLY | OpenFlags::SQLITE_OPEN_NO_MUTEX, + )?; + connection.execute_batch( + "PRAGMA query_only=ON; PRAGMA trusted_schema=OFF; PRAGMA mmap_size=0;", + )?; + connection.pragma_update(None, "cache_size", -(cache_kib as i64))?; + let app: u32 = connection.pragma_query_value(None, "application_id", |row| row.get(0))?; + let version: u32 = connection.pragma_query_value(None, "user_version", |row| row.get(0))?; + if app != APPLICATION_ID || version != 1 { + return Err(MetadataError::Integrity); + } + let stored = connection.query_row("SELECT repository_id,operation_id,object_format,object_count,first_oid,last_oid,inventory_digest FROM directory_identity WHERE singleton=1", [], |row| { + Ok((row.get::<_,Vec>(0)?,row.get::<_,Vec>(1)?,row.get::<_,String>(2)?,row.get::<_,u64>(3)?,row.get::<_,Vec>(4)?,row.get::<_,Vec>(5)?,row.get::<_,Vec>(6)?)) + })?; + if stored + != ( + descriptor.repository.to_vec(), + descriptor.operation.to_vec(), + descriptor.format.as_str().to_owned(), + descriptor.object_count, + descriptor.first_oid.to_vec(), + descriptor.last_oid.to_vec(), + descriptor.inventory_digest.to_vec(), + ) + { + return Err(MetadataError::Integrity); + } + Ok(Self { + connection: Mutex::new(connection), + admitted, + descriptor, + }) + } + pub fn descriptor(&self) -> RunDescriptor { + self.descriptor + } + pub fn path(&self) -> &Path { + self.admitted.file().path() + } + fn connection(&self) -> Result, MetadataError> { + self.connection.lock().map_err(|_| MetadataError::Integrity) + } + pub fn find(&self, oid: ObjectId) -> Result, MetadataError> { + Ok(self.find_batch(&[oid])?.pop().flatten()) + } + /// One prepared indexed statement and connection lock for a bounded batch. + pub fn find_batch( + &self, + ids: &[ObjectId], + ) -> Result>, MetadataError> { + if ids.len() > PAGE_OBJECTS { + return Err(MetadataError::Limit); + } + let connection = self.connection()?; + let mut statement = connection.prepare_cached("SELECT oid,kind,size,digest,edge_count,edge_digest,source_operation,source_digest,location_version FROM objects WHERE oid=?1")?; + ids.iter() + .map(|oid| { + if oid.format() != self.descriptor.format + || *oid < self.descriptor.first_oid + || *oid > self.descriptor.last_oid + { + return Ok(None); + } + Ok(statement.query_row([oid.as_ref()], entry).optional()?) + }) + .collect() + } + pub fn entries_after( + &self, + after: Option, + ) -> Result, MetadataError> { + if after.is_some_and(|oid| oid.format() != self.descriptor.format) { + return Err(MetadataError::Integrity); + } + let connection = self.connection()?; + let mut statement = connection.prepare_cached("SELECT oid,kind,size,digest,edge_count,edge_digest,source_operation,source_digest,location_version FROM objects WHERE oid>?1 ORDER BY oid LIMIT ?2")?; + Ok(statement + .query_map( + params![ + after.as_ref().map_or(&[][..], AsRef::<[u8]>::as_ref), + PAGE_OBJECTS as i64 + ], + entry, + )? + .collect::>()?) + } + #[cfg(test)] + pub(in crate::packs) fn verify_inventory(&self) -> Result { + self.verify_coverage(self.descriptor.coverage()) + } + pub async fn upload( + self: Arc, + store: &ArtifactStore, + ) -> Result { + let run = self.descriptor; + if run.repository != store.repository() { + return Err(MetadataError::Integrity); + } + let artifact = upload_file(self, store, run.key(), run.size, run.digest).await?; + Ok(StoredRun { + run, + artifact, + coverage: run.coverage(), + }) + } + pub async fn download( + root: &Path, + budget: DiskBudget, + store: &ArtifactStore, + stored: StoredRun, + limits: MetadataLimits, + ) -> Result, MetadataError> { + Self::download_for_reader(root, budget, store, stored, limits, None).await + } + pub(in crate::packs) async fn download_for_reader( + root: &Path, + budget: DiskBudget, + store: &ArtifactStore, + stored: StoredRun, + limits: MetadataLimits, + reader: Option, + ) -> Result, MetadataError> { + stored.validate()?; + if stored.run.repository != store.repository() { + return Err(MetadataError::Integrity); + } + let admitted = download_file_for_reader( + root, + budget, + store, + stored.run.key(), + stored.artifact, + limits, + reader, + ) + .await?; + tokio::task::spawn_blocking(move || { + Ok(Arc::new(Self::open_admitted( + admitted, + stored.run, + limits.cache_kib, + )?)) + }) + .await? + } +} + +#[cfg(test)] +pub(in crate::packs) mod tests; diff --git a/crates/canopy-server/src/packs/directory/partition.rs b/crates/canopy-server/src/packs/directory/partition.rs new file mode 100644 index 0000000..d794d0d --- /dev/null +++ b/crates/canopy-server/src/packs/directory/partition.rs @@ -0,0 +1,140 @@ +//! Stream disjoint immutable runs from an admitted canonical spool. Callers must +//! exhaust the stream before publishing its root: exhaustion checks the entire +//! output inventory against the input, including count and range endpoints. +use super::*; + +pub struct DirectoryPartitioner { + input: Arc, + budget: DiskBudget, + limits: MetadataLimits, + pending: Vec, + position: usize, + after: Option, + count: u64, + inventory: [u8; 32], + first: Option, + last: Option, + finished: bool, + failed: bool, +} +impl DirectoryPartitioner { + pub(in crate::packs) fn validate_limits(limits: MetadataLimits) -> Result<(), MetadataError> { + if limits.max_file_bytes < 16 << 10 + || limits.max_file_bytes > RUN_TARGET_BYTES + || !limits.max_file_bytes.is_multiple_of(4096) + || limits.cache_kib == 0 + || limits.cache_kib > i32::MAX as u32 + { + return Err(MetadataError::Limit); + } + Ok(()) + } + pub fn new( + input: Arc, + budget: DiskBudget, + limits: MetadataLimits, + ) -> Result { + Self::validate_limits(limits)?; + Ok(Self { + inventory: inventory_seed(input.descriptor.format), + input, + budget, + limits, + pending: Vec::new(), + position: 0, + after: None, + count: 0, + first: None, + last: None, + finished: false, + failed: false, + }) + } + /// Blocking disk work. At most one input page (512 entries) and one output + /// builder are live. Input/output workspace pins follow queued work and files. + /// A failed call permanently poisons the stream; no partial root is complete. + pub fn next_run(&mut self) -> Result>, MetadataError> { + if self.failed { + return Err(MetadataError::Integrity); + } + if self.finished { + return Ok(None); + } + self.failed = true; + let result = self.next_inner()?; + self.failed = false; + Ok(result) + } + fn next_inner(&mut self) -> Result>, MetadataError> { + // Small pushes reuse their verified file without copying or reserving a + // second builder. Its exact descriptor already binds the whole inventory. + if self.input.descriptor.size <= self.limits.max_file_bytes { + self.finished = true; + return Ok(Some(Arc::clone(&self.input))); + } + let descriptor = self.input.descriptor; + let root = self.input.path().parent().ok_or(MetadataError::Integrity)?; + let mut output = DirectoryBuilder::new( + root, + self.budget.clone(), + descriptor.repository, + descriptor.operation, + descriptor.format, + self.limits, + )?; + if let Some(workspace) = self.input.admitted.workspace() { + output.retain_workspace(workspace); + } + let mut copied = 0_u64; + loop { + if self.position == self.pending.len() { + self.pending = self.input.entries_after(self.after)?; + self.position = 0; + self.after = self.pending.last().map(|entry| entry.header.object.oid); + if self.pending.is_empty() { + if self.count != descriptor.object_count + || self.inventory != descriptor.inventory_digest + || self.first != Some(descriptor.first_oid) + || self.last != Some(descriptor.last_oid) + { + return Err(MetadataError::Integrity); + } + self.finished = true; + return if copied == 0 { + Ok(None) + } else { + Ok(Some(Arc::new(output.seal()?))) + }; + } + } + let entries = &self.pending[self.position..]; + let mut length = entries.len(); + loop { + match output.put_entries(&entries[..length]) { + Ok(()) => break, + // SQLite rolls back the entire attempted transaction when + // max_page_count is reached. Retry bounded smaller prefixes; + // advance only entries that actually committed. Ordinary + // pages use one transaction, rather than one per object. + Err(MetadataError::Limit) if length > 1 => length /= 2, + Err(MetadataError::Limit) if copied > 0 => { + return Ok(Some(Arc::new(output.seal()?))); + } + Err(error) => return Err(error), + } + } + for entry in &entries[..length] { + let oid = entry.header.object.oid; + if self.last.is_some_and(|last| last >= oid) { + return Err(MetadataError::Integrity); + } + self.inventory = fold_header(self.inventory, self.count, entry.header); + self.count = self.count.checked_add(1).ok_or(MetadataError::Limit)?; + self.first.get_or_insert(oid); + self.last = Some(oid); + } + copied += length as u64; + self.position += length; + } + } +} diff --git a/crates/canopy-server/src/packs/directory/schema.sql b/crates/canopy-server/src/packs/directory/schema.sql new file mode 100644 index 0000000..e4855c5 --- /dev/null +++ b/crates/canopy-server/src/packs/directory/schema.sql @@ -0,0 +1,33 @@ +PRAGMA application_id = 1128353358; +PRAGMA user_version = 1; + +CREATE TABLE directory_identity ( + singleton INTEGER PRIMARY KEY CHECK(singleton = 1), + repository_id BLOB NOT NULL CHECK(length(repository_id) = 16), + operation_id BLOB NOT NULL CHECK(length(operation_id) = 16), + object_format TEXT NOT NULL CHECK(object_format IN ('sha1', 'sha256')), + object_count INTEGER NOT NULL CHECK(object_count > 0), + first_oid BLOB NOT NULL, + last_oid BLOB NOT NULL, + inventory_digest BLOB NOT NULL CHECK(length(inventory_digest) = 32), + CHECK(length(first_oid) = CASE object_format WHEN 'sha1' THEN 20 ELSE 32 END), + CHECK(length(last_oid) = length(first_oid)), + CHECK(first_oid <= last_oid) +) WITHOUT ROWID; + +-- Reuse canonical metadata and graph inventory, never native pack offsets. +-- The selected catalog's descriptor tree resolves the source key. +CREATE TABLE objects ( + oid BLOB PRIMARY KEY CHECK(length(oid) IN (20, 32)), + kind TEXT NOT NULL CHECK(kind IN ('blob', 'tree', 'commit', 'tag')), + size INTEGER NOT NULL CHECK(size >= 0), + digest BLOB NOT NULL CHECK(length(digest) = 32), + edge_count INTEGER NOT NULL CHECK(edge_count >= 0), + edge_digest BLOB NOT NULL CHECK(length(edge_digest) = 32), + source_operation BLOB NOT NULL CHECK(length(source_operation) = 16), + source_digest BLOB NOT NULL CHECK(length(source_digest) = 32), + location_version INTEGER NOT NULL CHECK(location_version > 0), + CHECK(kind != 'blob' OR edge_count = 0), + CHECK(kind != 'commit' OR edge_count > 0), + CHECK(kind != 'tag' OR edge_count = 1) +) WITHOUT ROWID; diff --git a/crates/canopy-server/src/packs/directory/snapshot.rs b/crates/canopy-server/src/packs/directory/snapshot.rs new file mode 100644 index 0000000..3397c29 --- /dev/null +++ b/crates/canopy-server/src/packs/directory/snapshot.rs @@ -0,0 +1,183 @@ +//! Bounded selection and canonical conflict checking across directory levels. +//! This snapshot is an input to catalog publication, not an authorization token. + +use super::{ + index::{IndexError, NodeRef, RangeIndex}, + *, +}; +use std::future::Future; + +mod codec; +pub use codec::StoredSnapshot; + +pub const LEVEL_ZERO_ROOTS: usize = 32; +pub const MAX_LEVELS: usize = 16; +pub const MAX_SELECTED_RUNS: usize = LEVEL_ZERO_ROOTS + MAX_LEVELS; + +/// Every root indexes disjoint OID ranges. Different level-zero roots may +/// overlap, but a partitioned ingress batch consumes only one root slot. +/// No point lookup materializes every run descriptor in a root. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct DirectorySnapshot { + pub repository: [u8; 16], + pub format: ObjectFormat, + pub level_zero: Vec, + pub levels: Vec>, +} +pub trait RunLoader: Sync { + fn load( + &self, + run: StoredRun, + ) -> impl Future, MetadataError>> + Send; +} +impl DirectorySnapshot { + pub fn empty(repository: [u8; 16], format: ObjectFormat) -> Self { + Self { + repository, + format, + level_zero: Vec::new(), + levels: Vec::new(), + } + } + pub fn validate(&self) -> Result<(), IndexError> { + if self.level_zero.len() > LEVEL_ZERO_ROOTS || self.levels.len() > MAX_LEVELS { + return Err(IndexError::Limit); + } + for (at, root) in self.level_zero.iter().enumerate() { + root.validate(self.format)?; + if self.level_zero[..at].contains(root) { + return Err(IndexError::Integrity); + } + } + for root in self.levels.iter().flatten() { + root.validate(self.format)?; + } + Ok(()) + } + /// Bounded ingress. When full, preparation must wait/reject until admitted + /// compaction publishes a replacement root; it cannot append extra roots. + /// Authenticate the run-set root in this repository before accepting it. + pub async fn append(&mut self, index: &RangeIndex, root: NodeRef) -> Result<(), IndexError> { + self.validate()?; + if index.repository() != self.repository || index.format() != self.format { + return Err(IndexError::Integrity); + } + index.validate_root(root).await?; + if self.level_zero.contains(&root) { + return Ok(()); + } + if self.level_zero.len() == LEVEL_ZERO_ROOTS { + return Err(IndexError::Limit); + } + self.level_zero.push(root); + Ok(()) + } + pub async fn selected_runs( + &self, + index: &RangeIndex, + oid: ObjectId, + ) -> Result, IndexError> { + self.validate()?; + if self.repository != index.repository() || self.format != index.format() { + return Err(IndexError::Integrity); + } + if oid.format() != self.format { + return Ok(Vec::new()); + } + let mut selected = Vec::new(); + for root in self + .level_zero + .iter() + .copied() + .map(Some) + .chain(self.levels.iter().copied()) + { + if let Some(run) = index.find(root, oid).await? + && !selected.contains(&run) + { + selected.push(run); + } + } + debug_assert!(selected.len() <= MAX_SELECTED_RUNS); + Ok(selected) + } + /// Every matching header must agree, even when its physical source has been + /// superseded. A newer placement never conceals an OID/graph conflict. + pub async fn lookup( + &self, + index: &RangeIndex, + loader: &impl RunLoader, + oid: ObjectId, + ) -> Result, IndexError> { + Ok(self + .lookup_batch(index, loader, &[oid]) + .await? + .pop() + .flatten()) + } + /// Group only requested IDs by selected file. At most 512 IDs and 48 file + /// candidates per ID are retained; no historical inventory is loaded. + pub async fn lookup_batch( + &self, + index: &RangeIndex, + loader: &impl RunLoader, + ids: &[ObjectId], + ) -> Result>, IndexError> { + if ids.len() > PAGE_OBJECTS { + return Err(IndexError::Limit); + } + self.validate()?; + if self.repository != index.repository() || self.format != index.format() { + return Err(IndexError::Integrity); + } + let mut groups = std::collections::BTreeMap::)>::new(); + for (at, oid) in ids.iter().enumerate() { + for stored in self.selected_runs(index, *oid).await? { + let key = SegmentKey { + operation: stored.run.operation, + digest: stored.artifact.digest, + }; + let group = groups.entry(key).or_insert_with(|| (stored, Vec::new())); + if group.0.run != stored.run || group.0.artifact != stored.artifact { + return Err(IndexError::Integrity); + } + group.1.push(at); + } + } + let mut chosen: Vec> = vec![None; ids.len()]; + for (_, (stored, positions)) in groups { + let run = loader.load(stored).await?; + let requested: Vec<_> = positions.iter().map(|at| ids[*at]).collect(); + let candidates = tokio::task::spawn_blocking(move || { + if run.descriptor() != stored.run { + return Err(MetadataError::Integrity); + } + run.find_batch(&requested) + }) + .await + .map_err(MetadataError::from)??; + if candidates.len() != positions.len() { + return Err(IndexError::Integrity); + } + for (at, candidate) in positions.into_iter().zip(candidates) { + let Some(candidate) = candidate else { + continue; + }; + if let Some(current) = chosen[at] { + if current.header != candidate.header { + return Err(MetadataError::IdentityConflict.into()); + } + if candidate.location_version > current.location_version + || (candidate.location_version == current.location_version + && candidate.source < current.source) + { + chosen[at] = Some(candidate); + } + } else { + chosen[at] = Some(candidate); + } + } + } + Ok(chosen) + } +} diff --git a/crates/canopy-server/src/packs/directory/snapshot/codec.rs b/crates/canopy-server/src/packs/directory/snapshot/codec.rs new file mode 100644 index 0000000..d9688c9 --- /dev/null +++ b/crates/canopy-server/src/packs/directory/snapshot/codec.rs @@ -0,0 +1,142 @@ +use super::super::index::{ + NODE_BYTES, + codec::{fixed, read_reference, reference}, +}; +use super::*; +use cellule_runtime::codec::{BoundedDecoder, BoundedEncoder}; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct StoredSnapshot { + pub repository: [u8; 16], + pub operation: [u8; 16], + pub format: ObjectFormat, + pub artifact: ArtifactDescriptor, +} +impl StoredSnapshot { + fn key(self) -> ArtifactKey { + ArtifactKey { + operation: self.operation, + binding_digest: self.artifact.digest, + kind: ArtifactKind::CatalogNode, + } + } +} +impl DirectorySnapshot { + pub(in crate::packs::directory) fn encode( + &self, + operation: [u8; 16], + ) -> Result, IndexError> { + self.validate()?; + let mut encoder = BoundedEncoder::new(NODE_BYTES)?; + encoder.write_bytes(b"canopy.directory-root.v3\0")?; + encoder.write_bytes(&self.repository)?; + encoder.write_bytes(&operation)?; + encoder.write_u8(self.format.bytes() as u8)?; + encoder.write_count(self.level_zero.len())?; + for root in &self.level_zero { + reference(&mut encoder, *root)?; + } + encoder.write_count(self.levels.len())?; + for root in &self.levels { + encoder.write_bool(root.is_some())?; + if let Some(root) = root { + reference(&mut encoder, *root)?; + } + } + Ok(encoder.finish()) + } + pub(in crate::packs::directory) fn decode( + bytes: &[u8], + ) -> Result<(Self, [u8; 16]), IndexError> { + let mut decoder = BoundedDecoder::new(bytes, NODE_BYTES)?; + if decoder.read_bytes()? != b"canopy.directory-root.v3\0" { + return Err(IndexError::Integrity); + } + let repository = fixed(&mut decoder)?; + let operation = fixed(&mut decoder)?; + let format = match decoder.read_u8()? { + 20 => ObjectFormat::Sha1, + 32 => ObjectFormat::Sha256, + _ => return Err(IndexError::Integrity), + }; + let count = decoder.read_count()?; + if count > LEVEL_ZERO_ROOTS { + return Err(IndexError::Limit); + } + let mut level_zero = Vec::with_capacity(count); + for _ in 0..count { + level_zero.push(read_reference(&mut decoder, format)?); + } + let count = decoder.read_count()?; + if count > MAX_LEVELS { + return Err(IndexError::Limit); + } + let mut levels = Vec::with_capacity(count); + for _ in 0..count { + levels.push(if decoder.read_bool()? { + Some(read_reference(&mut decoder, format)?) + } else { + None + }); + } + decoder.finish()?; + let snapshot = Self { + repository, + format, + level_zero, + levels, + }; + snapshot.validate()?; + Ok((snapshot, operation)) + } + /// Publishes immutable directory bytes, not refs or a successful push. + pub async fn upload( + &self, + store: &ArtifactStore, + operation: [u8; 16], + ) -> Result { + if self.repository != store.repository() { + return Err(IndexError::Integrity); + } + let bytes = self.encode(operation)?; + let digest = *blake3::hash(&bytes).as_bytes(); + let key = ArtifactKey { + operation, + binding_digest: digest, + kind: ArtifactKind::CatalogNode, + }; + let artifact = store + .put(key, bytes.len() as u64, digest, &mut bytes.as_slice()) + .await?; + Ok(StoredSnapshot { + repository: self.repository, + operation, + format: self.format, + artifact, + }) + } + pub async fn download( + store: &ArtifactStore, + stored: StoredSnapshot, + ) -> Result { + if stored.repository != store.repository() + || stored.artifact.size == 0 + || stored.artifact.size > u64::from(NODE_BYTES) + { + return Err(IndexError::Integrity); + } + let mut reader = store.read(stored.key(), stored.artifact).await?; + let bytes = reader.next().await?.ok_or(IndexError::Integrity)?; + if reader.next().await?.is_some() { + return Err(IndexError::Integrity); + } + let (snapshot, operation) = Self::decode(&bytes)?; + if snapshot.repository != stored.repository + || snapshot.format != stored.format + || operation != stored.operation + { + return Err(IndexError::Integrity); + } + Ok(snapshot) + } +} diff --git a/crates/canopy-server/src/packs/directory/tests.rs b/crates/canopy-server/src/packs/directory/tests.rs new file mode 100644 index 0000000..fccd0df --- /dev/null +++ b/crates/canopy-server/src/packs/directory/tests.rs @@ -0,0 +1,568 @@ +use super::*; +use crate::packs::metadata::tests::{Fixture, builder, fill, fixture, limits}; +use object_store::{ObjectStore, ObjectStoreExt, memory::InMemory}; + +mod compaction_inventory; +mod coverage; +mod partition; +mod partition_lifetime; + +type Result = std::result::Result>; +fn segment(fixture: &Fixture, budget: DiskBudget, operation: [u8; 16]) -> Result { + let mut identity = fixture.identity; + identity.operation = operation; + let mut writer = builder(fixture, budget, identity)?; + fill( + &mut writer, + &fixture.objects.values().cloned().collect::>(), + )?; + Ok(writer.seal(&fixture.index)?) +} +fn directory(fixture: &Fixture, budget: DiskBudget) -> Result { + Ok(DirectoryBuilder::new( + fixture.root.path(), + budget, + fixture.identity.repository, + [3; 16], + fixture.identity.format, + limits(), + )?) +} + +#[tokio::test] +async fn canonical_runs_deduplicate_native_shards_and_choose_stable_sources() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = fixture(format, 40).await?; + let budget = DiskBudget::new(128 << 20); + let first = segment(&fixture, budget.clone(), [1; 16])?; + let second = segment(&fixture, budget.clone(), [2; 16])?; + let mut a = directory(&fixture, budget.clone())?; + a.add_segment(&second)?; + a.add_segment(&first)?; + a.add_segment(&first)?; + let a = a.seal()?; + let mut b = directory(&fixture, budget.clone())?; + b.add_segment(&first)?; + b.add_segment(&second)?; + let b = b.seal()?; + assert_eq!(a.descriptor().object_count, fixture.objects.len() as u64); + assert_eq!( + a.descriptor().inventory_digest, + b.descriptor().inventory_digest + ); + for oid in fixture.objects.keys() { + let entry = a.find(*oid)?.ok_or("entry")?; + assert_eq!(entry.header, first.header(*oid)?.ok_or("header")?); + assert_eq!( + entry.source, + SegmentKey { + operation: [1; 16], + digest: first.descriptor().digest + } + ); + assert_eq!(Some(entry), b.find(*oid)?); + } + assert!(a.find(format.zero())?.is_none()); + assert!( + a.find(if format == ObjectFormat::Sha1 { + ObjectFormat::Sha256.zero() + } else { + ObjectFormat::Sha1.zero() + })? + .is_none() + ); + let mut compact = directory(&fixture, budget.clone())?; + compact.add_run(&a)?; + compact.add_run(&b)?; + let compact = compact.seal()?; + assert_eq!( + compact.descriptor().inventory_digest, + a.descriptor().inventory_digest + ); + assert_eq!(compact.entries_after(None)?, a.entries_after(None)?); + drop((first, second, a, b, compact)); + assert_eq!(budget.used(), 0); + } + Ok(()) +} + +#[tokio::test] +async fn conflicting_body_or_graph_identity_rolls_back_a_complete_batch() -> Result { + let fixture = fixture(ObjectFormat::Sha256, 4).await?; + let budget = DiskBudget::new(128 << 20); + let source = segment(&fixture, budget.clone(), [2; 16])?; + for graph in [false, true] { + let mut writer = directory(&fixture, budget.clone())?; + writer.add_segment(&source)?; + let oid = *fixture.objects.keys().next().ok_or("oid")?; + let original = DirectoryEntry { + location_version: 1, + header: source.header(oid)?.ok_or("header")?, + source: SegmentKey { + operation: [2; 16], + digest: source.descriptor().digest, + }, + }; + let mut extra = original; + extra.header.object.oid = ObjectId::Sha256([91; 32]); + assert!(!fixture.objects.contains_key(&extra.header.object.oid)); + let mut conflict = original; + if graph { + conflict.header.edge_digest[0] ^= 1; + } else { + conflict.header.object.digest[0] ^= 1; + } + assert!(matches!( + writer.put_entries(&[extra, conflict]), + Err(MetadataError::IdentityConflict) + )); + let sealed = writer.seal()?; + assert_eq!( + sealed.descriptor().object_count, + fixture.objects.len() as u64 + ); + assert!(sealed.find(extra.header.object.oid)?.is_none()); + assert_eq!(sealed.find(oid)?, Some(original)); + } + drop(source); + assert_eq!(budget.used(), 0); + Ok(()) +} + +#[tokio::test] +async fn source_copy_failure_poisoning_prevents_partial_run_publication() -> Result { + let fixture = fixture(ObjectFormat::Sha1, 4).await?; + let budget = DiskBudget::new(128 << 20); + let source = segment(&fixture, budget.clone(), [1; 16])?; + let mut bad = builder(&fixture, budget.clone(), fixture.identity)?; + let mut objects = fixture.objects.values().cloned().collect::>(); + objects[0].0.digest[0] ^= 1; // Model a conflicting trusted verifier result. + fill(&mut bad, &objects)?; + let bad = bad.seal(&fixture.index)?; + let mut writer = directory(&fixture, budget.clone())?; + writer.add_segment(&source)?; + assert!(matches!( + writer.add_segment(&bad), + Err(MetadataError::IdentityConflict) + )); + assert!(matches!( + writer.add_segment(&source), + Err(MetadataError::Integrity) + )); + assert!(matches!(writer.seal(), Err(MetadataError::Integrity))); + drop((source, bad)); + assert_eq!(budget.used(), 0); + Ok(()) +} + +#[tokio::test] +async fn large_run_pages_and_compaction_stay_within_disk_admission() -> Result { + let fixture = fixture(ObjectFormat::Sha256, 1600).await?; + let budget = DiskBudget::new(128 << 20); + let source = segment(&fixture, budget.clone(), [2; 16])?; + let mut writer = directory(&fixture, budget.clone())?; + writer.add_segment(&source)?; + let run = writer.seal()?; + assert_eq!( + budget.used(), + source.descriptor().size + run.descriptor().size + ); + let mut after = None; + let mut count = 0; + loop { + let page = run.entries_after(after)?; + if page.is_empty() { + break; + } + assert!(page.len() <= PAGE_OBJECTS); + for entry in &page { + assert!(after.is_none_or(|oid| oid < entry.header.object.oid)); + after = Some(entry.header.object.oid); + } + count += page.len() as u64; + } + assert_eq!(count, run.descriptor().object_count); + let mut compact = directory(&fixture, budget.clone())?; + compact.add_run(&run)?; + let compact = compact.seal()?; + assert_eq!( + compact.descriptor().inventory_digest, + run.descriptor().inventory_digest + ); + assert!(matches!( + DirectoryBuilder::new( + fixture.root.path(), + DiskBudget::new(3 * metadata::growth::INITIAL_BYTES - 1), + fixture.identity.repository, + [4; 16], + fixture.identity.format, + limits() + ), + Err(MetadataError::Budget(_)) + )); + drop((source, run, compact)); + assert_eq!(budget.used(), 0); + Ok(()) +} + +#[tokio::test] +async fn directory_artifacts_roundtrip_and_reject_corruption_and_wrong_repository() -> Result { + let fixture = fixture(ObjectFormat::Sha256, 4).await?; + let budget = DiskBudget::new(128 << 20); + let source = segment(&fixture, budget.clone(), [2; 16])?; + let mut writer = directory(&fixture, budget.clone())?; + writer.add_segment(&source)?; + let run = Arc::new(writer.seal()?); + let store: Arc = Arc::new(InMemory::new()); + let artifacts = ArtifactStore::new(Arc::clone(&store), fixture.identity.repository); + let stored = Arc::clone(&run).upload(&artifacts).await?; + let downloaded = DirectoryRun::download( + fixture.root.path(), + budget.clone(), + &artifacts, + stored, + limits(), + ) + .await?; + assert_eq!(downloaded.descriptor(), run.descriptor()); + assert_eq!(downloaded.entries_after(None)?, run.entries_after(None)?); + drop(downloaded); + let charged = budget.used(); + let other = ArtifactStore::new(Arc::clone(&store), [9; 16]); + assert!(matches!( + DirectoryRun::download( + fixture.root.path(), + budget.clone(), + &other, + stored, + limits() + ) + .await, + Err(MetadataError::Integrity) + )); + assert!(matches!( + Arc::clone(&run).upload(&other).await, + Err(MetadataError::Integrity) + )); + let mut wrong = stored; + wrong.artifact.size += 1; + assert!(matches!( + DirectoryRun::download( + fixture.root.path(), + budget.clone(), + &artifacts, + wrong, + limits() + ) + .await, + Err(MetadataError::Integrity) + )); + let path = artifacts.path(stored.run.key(), stored.artifact.digest)?; + assert!(path.as_ref().contains("/git-catalogs/")); + assert!(path.as_ref().contains("/directory/")); + store + .put( + &canopy_object_storage::external::part(&path, 0), + bytes::Bytes::from(vec![0; stored.artifact.size as usize]).into(), + ) + .await?; + assert!( + DirectoryRun::download( + fixture.root.path(), + budget.clone(), + &artifacts, + stored, + limits() + ) + .await + .is_err() + ); + assert_eq!(budget.used(), charged); + drop((source, run)); + assert_eq!(budget.used(), 0); + Ok(()) +} + +#[tokio::test] +async fn empty_directory_and_mismatched_format_or_repository_cannot_seal() -> Result { + let fixture = fixture(ObjectFormat::Sha1, 4).await?; + let budget = DiskBudget::new(128 << 20); + assert!(matches!( + directory(&fixture, budget.clone())?.seal(), + Err(MetadataError::Integrity) + )); + let source = segment(&fixture, budget.clone(), [2; 16])?; + for (repository, format) in [ + ([9; 16], ObjectFormat::Sha1), + (fixture.identity.repository, ObjectFormat::Sha256), + ] { + let mut writer = DirectoryBuilder::new( + fixture.root.path(), + budget.clone(), + repository, + [3; 16], + format, + limits(), + )?; + assert!(matches!( + writer.add_segment(&source), + Err(MetadataError::Integrity) + )); + assert!(matches!(writer.seal(), Err(MetadataError::Integrity))); + } + drop(source); + assert_eq!(budget.used(), 0); + Ok(()) +} + +#[tokio::test] +async fn relocation_cas_is_atomic_and_old_versions_cannot_restore_old_sources() -> Result { + let fixture = fixture(ObjectFormat::Sha256, 4).await?; + let budget = DiskBudget::new(128 << 20); + let old_source = segment(&fixture, budget.clone(), [1; 16])?; + let replacement = segment(&fixture, budget.clone(), [9; 16])?; + let mut writer = directory(&fixture, budget.clone())?; + writer.add_segment(&old_source)?; + let old = writer.seal()?; + let expected = old.entries_after(None)?; + let mut writer = directory(&fixture, budget.clone())?; + writer.add_run(&old)?; + let mut stale = expected.clone(); + stale.last_mut().ok_or("entry")?.source.digest[0] ^= 1; + assert!(matches!( + writer.relocate(&replacement, &stale), + Err(MetadataError::PlacementConflict) + )); + writer.relocate(&replacement, &expected)?; + assert!(matches!( + writer.relocate(&replacement, &expected), + Err(MetadataError::PlacementConflict) + )); + writer.add_segment(&old_source)?; + let updated = writer.seal()?; + assert_eq!( + updated.descriptor().inventory_digest, + old.descriptor().inventory_digest + ); + for entry in updated.entries_after(None)? { + assert_eq!(entry.location_version, 2); + assert_eq!( + entry.source, + SegmentKey { + operation: [9; 16], + digest: replacement.descriptor().digest + } + ); + } + let mut merged = directory(&fixture, budget.clone())?; + merged.add_run(&updated)?; + merged.add_run(&old)?; + let merged = merged.seal()?; + assert_eq!(merged.entries_after(None)?, updated.entries_after(None)?); + drop((merged, updated, old, replacement, old_source)); + assert_eq!(budget.used(), 0); + Ok(()) +} + +struct Loaded(Vec>); +impl snapshot::RunLoader for Loaded { + async fn load( + &self, + stored: StoredRun, + ) -> std::result::Result, MetadataError> { + self.0 + .iter() + .find(|run| run.descriptor() == stored.run) + .cloned() + .ok_or(MetadataError::Integrity) + } +} + +#[tokio::test] +async fn snapshot_bounds_selection_and_roundtrips_authenticated_root_bytes() -> Result { + use snapshot::{DirectorySnapshot, LEVEL_ZERO_ROOTS, MAX_LEVELS, MAX_SELECTED_RUNS}; + let fixture = fixture(ObjectFormat::Sha256, 4).await?; + let budget = DiskBudget::new(128 << 20); + let source = segment(&fixture, budget.clone(), [1; 16])?; + let store: Arc = Arc::new(InMemory::new()); + let artifacts = Arc::new(ArtifactStore::new( + Arc::clone(&store), + fixture.identity.repository, + )); + let index = index::RangeIndex::new(Arc::clone(&artifacts), fixture.identity.format); + let mut snapshot = + DirectorySnapshot::empty(fixture.identity.repository, fixture.identity.format); + let mut loaded = Loaded(Vec::new()); + let mut last = None; + for n in 0..=MAX_SELECTED_RUNS { + let mut writer = DirectoryBuilder::new( + fixture.root.path(), + budget.clone(), + fixture.identity.repository, + [n as u8 + 10; 16], + fixture.identity.format, + limits(), + )?; + writer.add_segment(&source)?; + let run = Arc::new(writer.seal()?); + let stored = Arc::clone(&run).upload(&artifacts).await?; + loaded.0.push(run); + let root = index.insert(None, [n as u8 + 70; 16], stored).await?; + if n < LEVEL_ZERO_ROOTS { + snapshot.append(&index, root).await?; + } else if n < MAX_SELECTED_RUNS { + snapshot.levels.push(Some(root)); + } else { + assert!(matches!( + snapshot.append(&index, root).await, + Err(index::IndexError::Limit) + )); + } + last = Some(root); + } + assert_eq!(snapshot.levels.len(), MAX_LEVELS); + let oid = *fixture.objects.keys().next().ok_or("oid")?; + assert_eq!( + snapshot.selected_runs(&index, oid).await?.len(), + MAX_SELECTED_RUNS + ); + assert_eq!( + snapshot + .lookup(&index, &loaded, oid) + .await? + .ok_or("entry")? + .header, + source.header(oid)?.ok_or("header")? + ); + let ids: Vec<_> = fixture.objects.keys().rev().copied().collect(); + let batch = snapshot.lookup_batch(&index, &loaded, &ids).await?; + for (oid, actual) in ids.iter().zip(batch) { + assert_eq!( + actual.ok_or("batch entry")?.header, + source.header(*oid)?.ok_or("header")? + ); + } + let stored = snapshot.upload(&artifacts, [90; 16]).await?; + let restored = DirectorySnapshot::download(&artifacts, stored).await?; + assert_eq!(restored, snapshot); + let bytes = snapshot.encode([90; 16])?; + let mut old_layout = bytes.clone(); + let domain = b"canopy.directory-root.v3\0"; + let at = old_layout + .windows(domain.len()) + .position(|bytes| bytes == domain) + .ok_or("domain")?; + for version in *b"12" { + old_layout[at + domain.len() - 2] = version; + assert!(matches!( + DirectorySnapshot::decode(&old_layout), + Err(index::IndexError::Integrity) + )); + } + for length in [0, 1, bytes.len() - 1] { + assert!(DirectorySnapshot::decode(&bytes[..length]).is_err()); + } + let mut trailing = bytes; + trailing.push(0); + assert!(DirectorySnapshot::decode(&trailing).is_err()); + let mut too_many = snapshot.clone(); + too_many.levels.push(None); + assert!(matches!(too_many.validate(), Err(index::IndexError::Limit))); + too_many = snapshot.clone(); + too_many.level_zero.push(last.ok_or("last")?); + assert!(matches!(too_many.validate(), Err(index::IndexError::Limit))); + let other = ArtifactStore::new(Arc::clone(&store), [99; 16]); + assert!(DirectorySnapshot::download(&other, stored).await.is_err()); + let mut wrong = stored; + wrong.format = ObjectFormat::Sha1; + assert!(matches!( + DirectorySnapshot::download(&artifacts, wrong).await, + Err(index::IndexError::Integrity) + )); + drop((snapshot, restored, loaded, source)); + assert_eq!(budget.used(), 0); + Ok(()) +} + +#[tokio::test] +async fn newer_placement_does_not_hide_conflicting_canonical_headers_in_other_levels() -> Result { + use snapshot::DirectorySnapshot; + let fixture = fixture(ObjectFormat::Sha1, 4).await?; + let budget = DiskBudget::new(128 << 20); + let original = segment(&fixture, budget.clone(), [1; 16])?; + let replacement = segment(&fixture, budget.clone(), [9; 16])?; + let mut writer = directory(&fixture, budget.clone())?; + writer.add_segment(&original)?; + let first = writer.seal()?; + let mut writer = directory(&fixture, budget.clone())?; + writer.add_run(&first)?; + writer.relocate(&replacement, &first.entries_after(None)?)?; + let newer = Arc::new(writer.seal()?); + let mut bad = builder(&fixture, budget.clone(), fixture.identity)?; + let mut objects = fixture.objects.values().cloned().collect::>(); + objects[0].0.digest[0] ^= 1; + fill(&mut bad, &objects)?; + let bad = bad.seal(&fixture.index)?; + let mut writer = directory(&fixture, budget.clone())?; + writer.add_segment(&bad)?; + let bad_run = Arc::new(writer.seal()?); + let store: Arc = Arc::new(InMemory::new()); + let artifacts = Arc::new(ArtifactStore::new(store, fixture.identity.repository)); + let index = index::RangeIndex::new(Arc::clone(&artifacts), fixture.identity.format); + let new_stored = Arc::clone(&newer).upload(&artifacts).await?; + let bad_stored = Arc::clone(&bad_run).upload(&artifacts).await?; + let mut snapshot = + DirectorySnapshot::empty(fixture.identity.repository, fixture.identity.format); + let new_root = index.insert(None, [72; 16], new_stored).await?; + snapshot.append(&index, new_root).await?; + snapshot + .levels + .push(Some(index.insert(None, [70; 16], bad_stored).await?)); + let loaded = Loaded(vec![newer, bad_run]); + let oid = objects[0].0.oid; + assert!(matches!( + snapshot.lookup(&index, &loaded, oid).await, + Err(index::IndexError::Metadata(MetadataError::IdentityConflict)) + )); + assert!(matches!( + snapshot.lookup_batch(&index, &loaded, &[oid, oid]).await, + Err(index::IndexError::Metadata(MetadataError::IdentityConflict)) + )); + let mut legitimate = + DirectorySnapshot::empty(fixture.identity.repository, fixture.identity.format); + legitimate.append(&index, new_root).await?; + let old = Arc::new(first); + let old_stored = Arc::clone(&old).upload(&artifacts).await?; + legitimate + .levels + .push(Some(index.insert(None, [71; 16], old_stored).await?)); + let loaded = Loaded(vec![Arc::clone(&loaded.0[0]), old]); + let chosen = legitimate + .lookup(&index, &loaded, oid) + .await? + .ok_or("entry")?; + assert_eq!(chosen.location_version, 2); + assert_eq!(chosen.source.operation, [9; 16]); + Ok(()) +} + +// Deliberately inconsistent authenticated directory fixture. Keep arbitrary +// entry insertion inside directory tests rather than exposing it to services. +pub(in crate::packs) fn inconsistent_run( + root: &Path, + repository: [u8; 16], + operation: [u8; 16], + format: ObjectFormat, + entries: &[DirectoryEntry], +) -> std::result::Result { + let mut writer = DirectoryBuilder::new( + root, + DiskBudget::new(128 << 20), + repository, + operation, + format, + limits(), + )?; + writer.put_entries(entries)?; + writer.seal() +} diff --git a/crates/canopy-server/src/packs/directory/tests/compaction_inventory.rs b/crates/canopy-server/src/packs/directory/tests/compaction_inventory.rs new file mode 100644 index 0000000..4bba18d --- /dev/null +++ b/crates/canopy-server/src/packs/directory/tests/compaction_inventory.rs @@ -0,0 +1,35 @@ +use super::*; + +#[tokio::test] +async fn compaction_copy_checks_each_full_input_inventory_and_poisons_partial_merges() -> Result { + let fixture = fixture(ObjectFormat::Sha256, 40).await?; + let budget = DiskBudget::new(128 << 20); + let source = segment(&fixture, budget.clone(), [1; 16])?; + for fault in 0..3 { + let mut writer = directory(&fixture, budget.clone())?; + writer.add_segment(&source)?; + let mut run = writer.seal()?; + match fault { + 0 => run.descriptor.object_count += 1, + 1 => run.descriptor.inventory_digest[0] ^= 1, + _ => run.descriptor.first_oid = ObjectId::Sha256([1; 32]), + } + assert!(matches!( + run.verify_inventory(), + Err(MetadataError::Integrity) + )); + let mut merged = directory(&fixture, budget.clone())?; + assert!(matches!( + merged.add_run(&run), + Err(MetadataError::Integrity) + )); + assert!(matches!( + merged.add_run(&run), + Err(MetadataError::Integrity) + )); + assert!(matches!(merged.seal(), Err(MetadataError::Integrity))); + } + drop(source); + assert_eq!(budget.used(), 0); + Ok(()) +} diff --git a/crates/canopy-server/src/packs/directory/tests/coverage.rs b/crates/canopy-server/src/packs/directory/tests/coverage.rs new file mode 100644 index 0000000..e65bb0c --- /dev/null +++ b/crates/canopy-server/src/packs/directory/tests/coverage.rs @@ -0,0 +1,152 @@ +use super::*; +use crate::packs::catalog::{CatalogFileLimits, CatalogFiles}; +use crate::packs::directory::{ + index::IndexError, + snapshot::{DirectorySnapshot, RunLoader}, +}; + +#[tokio::test] +async fn split_coverage_preserves_exact_inventory_and_rejects_forged_parent_or_parts() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = fixture(format, 40).await?; + let budget = DiskBudget::new(128 << 20); + let source = segment(&fixture, budget.clone(), [1; 16])?; + let mut builder = directory(&fixture, budget.clone())?; + builder.add_segment(&source)?; + let run = builder.seal()?; + let ids = fixture.objects.keys().copied().collect::>(); + let (left, right, edges) = run.split_coverage(run.descriptor().coverage(), ids[15])?; + let left = left.ok_or("left")?; + let right = right.ok_or("right")?; + assert_eq!( + left.object_count + right.object_count, + run.descriptor().object_count + ); + assert_eq!(left.first_oid, ids[0]); + assert_eq!(left.last_oid, ids[15]); + assert_eq!(right.first_oid, ids[16]); + assert_eq!(right.last_oid, *ids.last().ok_or("last")?); + assert_eq!(edges, run.verify_coverage(left)?); + run.verify_coverage(right)?; + let mut merged = directory(&fixture, budget.clone())?; + merged.add_coverage(&run, right)?; + merged.add_coverage(&run, left)?; + assert_eq!( + merged.seal()?.descriptor().inventory_digest, + run.descriptor().inventory_digest + ); + let zero: ObjectId = vec![0; format.bytes()].try_into()?; + assert_eq!( + run.split_coverage(run.descriptor().coverage(), zero)?.0, + None + ); + assert_eq!( + run.split_coverage(run.descriptor().coverage(), zero)?.1, + Some(run.descriptor().coverage()) + ); + assert_eq!(run.split_coverage(left, left.last_oid)?.1, None); + for fault in 0..3 { + let mut bad = left; + match fault { + 0 => bad.inventory_digest[0] ^= 1, + 1 => bad.object_count += 1, + _ => bad.first_oid = ids[1], + } + assert!(run.split_coverage(bad, ids[5]).is_err()); + let mut poisoned = directory(&fixture, budget.clone())?; + assert!(poisoned.add_coverage(&run, bad).is_err()); + assert!(poisoned.seal().is_err()); + } + let mut fake_full = run.descriptor().coverage(); + fake_full.object_count -= 1; + assert!(fake_full.validate(run.descriptor()).is_err()); + drop(run); + drop(source); + assert_eq!(budget.used(), 0); + } + Ok(()) +} + +#[tokio::test] +async fn logical_ranges_share_one_authenticated_file_and_bound_batch_lookup_and_path_updates() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = fixture(format, 40).await?; + let budget = DiskBudget::new(128 << 20); + let source = segment(&fixture, budget.clone(), [1; 16])?; + let mut builder = directory(&fixture, budget.clone())?; + builder.add_segment(&source)?; + let run = Arc::new(builder.seal()?); + let ids = fixture.objects.keys().copied().collect::>(); + let (left, right, _) = run.split_coverage(run.descriptor().coverage(), ids[15])?; + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new(provider, fixture.identity.repository)); + let full = run.upload(&store).await?; + let left = StoredRun { + coverage: left.ok_or("left")?, + ..full + }; + let right = StoredRun { + coverage: right.ok_or("right")?, + ..full + }; + let cache_budget = DiskBudget::new(128 << 20); + let files = CatalogFiles::new( + fixture.root.path(), + cache_budget.clone(), + Arc::clone(&store), + format, + CatalogFileLimits::default(), + )?; + let original = files.load(full).await?; + let a = files.load(left).await?; + let b = files.load(right).await?; + assert!(Arc::ptr_eq(&original, &a) && Arc::ptr_eq(&a, &b)); + assert_eq!(files.stats()?.downloaded_files, 1); + let index = index::RangeIndex::new(store, format); + let first = index.insert(None, [50; 16], left).await?; + let both = index.insert(Some(first), [50; 16], right).await?; + assert_eq!(both.record_count, 2); + assert_eq!(both.object_count, full.coverage.object_count); + assert_eq!(index.find(Some(both), ids[0]).await?, Some(left)); + assert_eq!(index.find(Some(both), ids[20]).await?, Some(right)); + assert!(matches!( + index.remove(Some(both), [51; 16], full).await, + Err(IndexError::Stale) + )); + assert!(matches!( + index.insert(Some(both), [51; 16], full).await, + Err(IndexError::RangeOverlap) + )); + let mut snapshot = DirectorySnapshot::empty(fixture.identity.repository, format); + snapshot.append(&index, both).await?; + let entries = snapshot.lookup_batch(&index, &files, &ids).await?; + for (oid, entry) in ids.iter().zip(entries) { + assert_eq!( + entry.ok_or("entry")?.header, + source.header(*oid)?.ok_or("header")? + ); + } + let mut narrow = DirectorySnapshot::empty(fixture.identity.repository, format); + narrow.append(&index, first).await?; + assert!(narrow.lookup(&index, &files, ids[20]).await?.is_none()); + let remaining = index + .remove(Some(both), [51; 16], left) + .await? + .ok_or("remaining")?; + assert_eq!(remaining.object_count, right.coverage.object_count); + assert!(index.find(Some(remaining), ids[0]).await?.is_none()); + assert_eq!(index.find(Some(both), ids[0]).await?, Some(left)); + let mut forged = right; + forged.artifact.manifest_digest[0] ^= 1; + assert!(files.load(forged).await.is_err()); + drop(original); + drop(a); + drop(b); + drop(files); + assert_eq!(cache_budget.used(), 0); + drop(source); + assert_eq!(budget.used(), 0); + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/directory/tests/partition.rs b/crates/canopy-server/src/packs/directory/tests/partition.rs new file mode 100644 index 0000000..d1051ba --- /dev/null +++ b/crates/canopy-server/src/packs/directory/tests/partition.rs @@ -0,0 +1,208 @@ +use super::*; + +fn run_limits() -> MetadataLimits { + MetadataLimits { + max_file_bytes: 16 << 10, + cache_kib: 16, + } +} + +#[tokio::test] +async fn streaming_partition_preserves_sources_versions_and_one_candidate_per_root() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = fixture(format, 1600).await?; + let budget = DiskBudget::new(128 << 20); + let source = segment(&fixture, budget.clone(), [1; 16])?; + let mut writer = directory(&fixture, budget.clone())?; + writer.add_segment(&source)?; + let original = writer.seal()?; + let replacement = segment(&fixture, budget.clone(), [9; 16])?; + let mut writer = directory(&fixture, budget.clone())?; + writer.add_run(&original)?; + writer.relocate(&replacement, &original.entries_after(None)?)?; + let input = Arc::new(writer.seal()?); + drop((original, replacement)); + assert_eq!(input.entries_after(None)?[0].location_version, 2); + let input_size = input.descriptor().size; + let mut stream = + DirectoryPartitioner::new(Arc::clone(&input), budget.clone(), run_limits())?; + let artifacts = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.identity.repository, + )); + let index = index::RangeIndex::new(Arc::clone(&artifacts), format); + let mut root = None; + let mut loaded = Loaded(Vec::new()); + let mut count = 0_u64; + let mut previous = None; + let mut bytes = 0; + while let Some(run) = stream.next_run()? { + let descriptor = run.descriptor(); + assert!(descriptor.size <= run_limits().max_file_bytes); + assert!(previous.is_none_or(|last| last < descriptor.first_oid)); + previous = Some(descriptor.last_oid); + let mut after = None; + loop { + let entries = run.entries_after(after)?; + if entries.is_empty() { + break; + } + for entry in &entries { + assert_eq!(Some(*entry), input.find(entry.header.object.oid)?); + count += 1; + } + after = entries.last().map(|entry| entry.header.object.oid); + } + let stored = Arc::clone(&run).upload(&artifacts).await?; + root = Some(index.insert(root, [80; 16], stored).await?); + bytes += descriptor.size; + loaded.0.push(run); + // Completed files shrink to their exact charge; no abandoned output + // reservation or rollback journal accumulates between calls. + assert_eq!(budget.used(), source.descriptor().size + input_size + bytes); + } + assert!(stream.next_run()?.is_none()); + assert_eq!(count, input.descriptor().object_count); + let root = root.ok_or("root")?; + assert!(root.record_count > snapshot::LEVEL_ZERO_ROOTS as u64); + assert_eq!(root.object_count, count); + let mut snapshot = snapshot::DirectorySnapshot::empty(fixture.identity.repository, format); + snapshot.append(&index, root).await?; + snapshot.append(&index, root).await?; + assert_eq!(snapshot.level_zero.len(), 1); + for ids in fixture + .objects + .keys() + .copied() + .collect::>() + .chunks(PAGE_OBJECTS) + { + for oid in ids { + assert_eq!(snapshot.selected_runs(&index, *oid).await?.len(), 1); + } + let results = snapshot.lookup_batch(&index, &loaded, ids).await?; + for (oid, result) in ids.iter().zip(results) { + assert_eq!(result, input.find(*oid)?); + } + } + drop((stream, input, source, loaded)); + assert_eq!(budget.used(), 0); + } + Ok(()) +} + +#[tokio::test] +async fn partition_failure_poisons_stream_and_small_runs_reuse_the_original_file() -> Result { + let fixture = fixture(ObjectFormat::Sha256, 80).await?; + let budget = DiskBudget::new(128 << 20); + let source = segment(&fixture, budget.clone(), [1; 16])?; + let mut writer = directory(&fixture, budget.clone())?; + writer.add_segment(&source)?; + let input = Arc::new(writer.seal()?); + assert!(input.descriptor().size > run_limits().max_file_bytes); + let output_budget = DiskBudget::new(1); + let mut stream = + DirectoryPartitioner::new(Arc::clone(&input), output_budget.clone(), run_limits())?; + assert!(matches!(stream.next_run(), Err(MetadataError::Budget(_)))); + assert!(matches!(stream.next_run(), Err(MetadataError::Integrity))); + assert_eq!(output_budget.used(), 0); + let mut small = DirectoryPartitioner::new(Arc::clone(&input), output_budget, limits())?; + assert!(Arc::ptr_eq(&input, &small.next_run()?.ok_or("small run")?)); + assert!(small.next_run()?.is_none()); + drop((stream, small, input, source)); + assert_eq!(budget.used(), 0); + Ok(()) +} + +#[tokio::test] +async fn partition_exhaustion_rejects_an_incomplete_input_inventory() -> Result { + let fixture = fixture(ObjectFormat::Sha1, 80).await?; + let budget = DiskBudget::new(128 << 20); + let source = segment(&fixture, budget.clone(), [1; 16])?; + for mismatch in 0..3 { + let mut writer = directory(&fixture, budget.clone())?; + writer.add_segment(&source)?; + let mut input = writer.seal()?; + // Trusted memory fault injection: authenticated physical descriptors do + // not by themselves prove complete canonical output coverage. + match mismatch { + 0 => input.descriptor.object_count += 1, + 1 => input.descriptor.inventory_digest[0] ^= 1, + _ => input.descriptor.last_oid = ObjectId::Sha1([255; 20]), + } + let mut stream = DirectoryPartitioner::new(Arc::new(input), budget.clone(), run_limits())?; + loop { + match stream.next_run() { + Ok(Some(run)) => drop(run), + Ok(None) => return Err("incomplete inventory accepted".into()), + Err(MetadataError::Integrity) => break, + Err(error) => return Err(error.into()), + } + } + assert!(matches!(stream.next_run(), Err(MetadataError::Integrity))); + } + drop(source); + assert_eq!(budget.used(), 0); + Ok(()) +} + +#[test] +fn canceled_partition_observer_keeps_workspace_and_disk_charge_until_worker_drains() -> Result { + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .max_blocking_threads(1) + .build()?; + runtime.block_on(async { + let fixture = fixture(ObjectFormat::Sha256, 80).await?; + let budget = DiskBudget::new(128 << 20); + let source = segment(&fixture, budget.clone(), [1; 16])?; + let root = tempfile::TempDir::new()?; + let workspace = Arc::new(tempfile::TempDir::new_in(root.path())?); + let path = workspace.path().to_owned(); + let mut writer = DirectoryBuilder::new( + workspace.path(), + budget.clone(), + fixture.identity.repository, + [80; 16], + fixture.identity.format, + limits(), + )?; + writer.retain_workspace(workspace); + writer.add_segment(&source)?; + let input = Arc::new(writer.seal()?); + let input_size = input.descriptor().size; + drop(source); + let mut stream = DirectoryPartitioner::new(input, budget.clone(), run_limits())?; + let (release_tx, release_rx) = std::sync::mpsc::channel(); + let started = Arc::new(tokio::sync::Notify::new()); + let notify = Arc::clone(&started); + let blocked = tokio::task::spawn_blocking(move || { + notify.notify_one(); + release_rx.recv().unwrap(); + }); + started.notified().await; + let queued = Arc::new(tokio::sync::Notify::new()); + let notify = Arc::clone(&queued); + let observer = tokio::spawn(async move { + let task = tokio::task::spawn_blocking(move || stream.next_run()); + notify.notify_one(); + task.await + }); + queued.notified().await; + observer.abort(); + assert!(matches!(observer.await, Err(error) if error.is_cancelled())); + assert!(path.exists()); + assert_eq!(budget.used(), input_size); + release_tx.send(())?; + blocked.await?; + tokio::time::timeout(std::time::Duration::from_secs(5), async { + while budget.used() != 0 { + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + }) + .await?; + assert!(!path.exists()); + assert_eq!(std::fs::read_dir(root.path())?.count(), 0); + Ok::<_, Box>(()) + }) +} diff --git a/crates/canopy-server/src/packs/directory/tests/partition_lifetime.rs b/crates/canopy-server/src/packs/directory/tests/partition_lifetime.rs new file mode 100644 index 0000000..edc84cf --- /dev/null +++ b/crates/canopy-server/src/packs/directory/tests/partition_lifetime.rs @@ -0,0 +1,42 @@ +use super::*; + +#[tokio::test] +async fn partition_output_retains_workspace_after_input_and_stream_drop() -> Result { + let fixture = fixture(ObjectFormat::Sha256, 80).await?; + let budget = DiskBudget::new(128 << 20); + let source = segment(&fixture, budget.clone(), [1; 16])?; + let root = tempfile::TempDir::new()?; + let workspace = Arc::new(tempfile::TempDir::new_in(root.path())?); + let path = workspace.path().to_owned(); + let mut writer = DirectoryBuilder::new( + workspace.path(), + budget.clone(), + fixture.identity.repository, + [80; 16], + fixture.identity.format, + limits(), + )?; + writer.retain_workspace(workspace); + writer.add_segment(&source)?; + let input = Arc::new(writer.seal()?); + drop(source); + let mut stream = DirectoryPartitioner::new( + input, + budget.clone(), + MetadataLimits { + max_file_bytes: 16 << 10, + cache_kib: 16, + }, + )?; + let output = stream.next_run()?.ok_or("output")?; + let entry = output.entries_after(None)?[0]; + drop(stream); + assert!(path.exists() && output.path().exists()); + assert_eq!(budget.used(), output.descriptor().size); + assert_eq!(output.find(entry.header.object.oid)?, Some(entry)); + drop(output); + assert_eq!(budget.used(), 0); + assert!(!path.exists()); + assert_eq!(std::fs::read_dir(root.path())?.count(), 0); + Ok(()) +} diff --git a/crates/canopy-server/src/packs/directory/writer.rs b/crates/canopy-server/src/packs/directory/writer.rs new file mode 100644 index 0000000..13124b6 --- /dev/null +++ b/crates/canopy-server/src/packs/directory/writer.rs @@ -0,0 +1,353 @@ +use super::*; + +/// Disk-backed canonical merge. Sources are sealed shards/runs, not arbitrary +/// user headers. A failed source copy poisons the builder until it is dropped. +pub struct DirectoryBuilder { + connection: Connection, + admitted: AdmittedFile, + repository: [u8; 16], + operation: [u8; 16], + format: ObjectFormat, + limits: MetadataLimits, + failed: bool, +} +impl DirectoryBuilder { + pub(in crate::packs) fn retain_workspace(&mut self, workspace: Arc) { + self.admitted.retain_workspace(workspace); + } + pub fn new( + root: &Path, + budget: DiskBudget, + repository: [u8; 16], + operation: [u8; 16], + format: ObjectFormat, + limits: MetadataLimits, + ) -> Result { + if limits.cache_kib == 0 + || limits.cache_kib > i32::MAX as u32 + || limits.max_file_bytes < 16 << 10 + || limits.max_file_bytes > canopy_object_storage::external::MAX_ARTIFACT_BYTES + || !limits.max_file_bytes.is_multiple_of(4096) + { + return Err(MetadataError::Limit); + } + let reservation = metadata::growth::reserve(&budget, limits.max_file_bytes)?; + let file = tempfile::Builder::new() + .prefix("canopy-directory-") + .tempfile_in(root)?; + let mut admitted = AdmittedFile::new(file, reservation); + let mut connection = Connection::open(admitted.file().path())?; + connection.execute_batch("PRAGMA page_size=4096; PRAGMA journal_mode=DELETE; PRAGMA synchronous=FULL; PRAGMA foreign_keys=ON; PRAGMA trusted_schema=OFF; PRAGMA mmap_size=0;")?; + connection.pragma_update(None, "cache_size", -(limits.cache_kib as i64))?; + metadata::growth::configure(&connection, &mut admitted)?; + metadata::growth::transaction( + &mut connection, + &mut admitted, + limits.max_file_bytes, + |tx| { + tx.execute_batch(SCHEMA)?; + Ok::<_, MetadataError>(()) + }, + )?; + Ok(Self { + connection, + admitted, + repository, + operation, + format, + limits, + failed: false, + }) + } + pub fn add_segment(&mut self, segment: &MetadataSegment) -> Result<(), MetadataError> { + let descriptor = segment.descriptor(); + if self.failed + || descriptor.identity.repository != self.repository + || descriptor.identity.format != self.format + { + return Err(MetadataError::Integrity); + } + self.failed = true; + let source = SegmentKey { + operation: descriptor.identity.operation, + digest: descriptor.digest, + }; + let mut after = None; + let mut copied = 0_u64; + loop { + let headers = segment.headers_after(after)?; + if headers.is_empty() { + break; + } + copied = copied + .checked_add(headers.len() as u64) + .ok_or(MetadataError::Limit)?; + after = headers.last().map(|header| header.object.oid); + self.put_entries( + &headers + .into_iter() + .map(|header| DirectoryEntry { + header, + source, + location_version: 1, + }) + .collect::>(), + )?; + } + if copied != u64::from(descriptor.identity.object_count) { + return Err(MetadataError::Integrity); + } + self.failed = false; + Ok(()) + } + /// Copy a certified run during directory compaction. Duplicate identities + /// preserve their full canonical/graph inventory and choose a stable source. + pub fn add_run(&mut self, run: &DirectoryRun) -> Result<(), MetadataError> { + let descriptor = run.descriptor(); + if self.failed + || descriptor.repository != self.repository + || descriptor.format != self.format + { + return Err(MetadataError::Integrity); + } + self.failed = true; + run.copy_coverage(descriptor.coverage(), |entries| self.put_entries(entries))?; + self.failed = false; + Ok(()) + } + /// Merge only the catalog's certified logical coverage of a physical file. + pub(in crate::packs) fn add_coverage( + &mut self, + run: &DirectoryRun, + coverage: RunCoverage, + ) -> Result<(), MetadataError> { + if self.failed + || run.descriptor.repository != self.repository + || run.descriptor.format != self.format + { + return Err(MetadataError::Integrity); + } + self.failed = true; + run.copy_coverage(coverage, |entries| self.put_entries(entries))?; + self.failed = false; + Ok(()) + } + /// Switch a bounded batch in a private compaction spool. The caller copies + /// certified input ranges first. Publication still CASes that input root; + /// this local CAS cannot substitute for the admitted Cell owner fence. + pub fn relocate( + &mut self, + replacement: &MetadataSegment, + expected: &[DirectoryEntry], + ) -> Result<(), MetadataError> { + let descriptor = replacement.descriptor(); + if self.failed + || descriptor.identity.repository != self.repository + || descriptor.identity.format != self.format + { + return Err(MetadataError::Integrity); + } + if expected.is_empty() || expected.len() > PAGE_OBJECTS { + return Err(MetadataError::Limit); + } + let source = SegmentKey { + operation: descriptor.identity.operation, + digest: descriptor.digest, + }; + let mut versions = Vec::with_capacity(expected.len()); + for entry in expected { + if entry.header.object.oid.format() != self.format || entry.location_version == 0 { + return Err(MetadataError::Integrity); + } + if entry.source == source { + return Err(MetadataError::PlacementConflict); + } + if replacement.header(entry.header.object.oid)? != Some(entry.header) { + return Err(MetadataError::IdentityConflict); + } + versions.push( + entry + .location_version + .checked_add(1) + .filter(|version| *version <= i64::MAX as u64) + .ok_or(MetadataError::Limit)?, + ); + } + metadata::growth::transaction( + &mut self.connection, + &mut self.admitted, + self.limits.max_file_bytes, + |transaction| { + { + let mut existing = transaction.prepare_cached("SELECT oid,kind,size,digest,edge_count,edge_digest,source_operation,source_digest,location_version FROM objects WHERE oid=?1")?; + let mut update = transaction.prepare_cached("UPDATE objects SET source_operation=?2,source_digest=?3,location_version=?4 WHERE oid=?1")?; + for (entry, version) in expected.iter().zip(&versions) { + let current = existing + .query_row([entry.header.object.oid.as_ref()], super::entry) + .optional()? + .ok_or(MetadataError::PlacementConflict)?; + if current.header != entry.header { + return Err(MetadataError::IdentityConflict); + } + if current != *entry { + return Err(MetadataError::PlacementConflict); + } + update.execute(params![ + entry.header.object.oid.as_ref(), + source.operation.as_slice(), + source.digest.as_slice(), + *version as i64 + ])?; + } + } + Ok(()) + }, + ) + } + pub(super) fn put_entries(&mut self, entries: &[DirectoryEntry]) -> Result<(), MetadataError> { + if entries.is_empty() || entries.len() > PAGE_OBJECTS { + return Err(MetadataError::Limit); + } + if entries.iter().any(|entry| { + entry.header.object.oid.format() != self.format + || entry.header.object.oid.is_zero() + || entry.header.object.size > i64::MAX as u64 + || entry.header.edge_count > i64::MAX as u64 + || entry.location_version == 0 + || entry.location_version > i64::MAX as u64 + }) { + return Err(MetadataError::Integrity); + } + metadata::growth::transaction( + &mut self.connection, + &mut self.admitted, + self.limits.max_file_bytes, + |transaction| { + { + let mut insert = transaction.prepare_cached( + "INSERT INTO objects VALUES (?1,?2,?3,?4,?5,?6,?7,?8,?9) ON CONFLICT DO NOTHING", + )?; + let mut existing = transaction.prepare_cached("SELECT oid,kind,size,digest,edge_count,edge_digest,source_operation,source_digest,location_version FROM objects WHERE oid=?1")?; + let mut relocate = transaction.prepare_cached( + "UPDATE objects SET source_operation=?2,source_digest=?3,location_version=?4 WHERE oid=?1", + )?; + for entry in entries { + let header = entry.header; + insert.execute(params![ + header.object.oid.as_ref(), + header.object.kind.git_name(), + header.object.size as i64, + header.object.digest.as_slice(), + header.edge_count as i64, + header.edge_digest.as_slice(), + entry.source.operation.as_slice(), + entry.source.digest.as_slice(), + entry.location_version as i64 + ])?; + let stored = + existing.query_row([header.object.oid.as_ref()], super::entry)?; + if stored.header != header { + return Err(MetadataError::IdentityConflict); + } + if entry.location_version > stored.location_version + || (entry.location_version == stored.location_version + && entry.source < stored.source) + { + relocate.execute(params![ + header.object.oid.as_ref(), + entry.source.operation.as_slice(), + entry.source.digest.as_slice(), + entry.location_version as i64 + ])?; + } + } + } + Ok(()) + }, + ) + } + pub fn seal(self) -> Result { + self.seal_with_edges().map(|(run, _)| run) + } + pub(in crate::packs) fn seal_with_edges( + mut self, + ) -> Result<(DirectoryRun, u64), MetadataError> { + if self.failed { + return Err(MetadataError::Integrity); + } + let mut inventory = inventory_seed(self.format); + let mut after = Vec::new(); + let mut count = 0_u64; + let mut edges = 0_u64; + let mut first = None; + let mut last = None; + loop { + let entries = { + let mut statement = self.connection.prepare_cached("SELECT oid,kind,size,digest,edge_count,edge_digest,source_operation,source_digest,location_version FROM objects WHERE oid>?1 ORDER BY oid LIMIT ?2")?; + statement + .query_map(params![after, PAGE_OBJECTS as i64], entry)? + .collect::>>()? + }; + if entries.is_empty() { + break; + } + for entry in entries { + inventory = fold_header(inventory, count, entry.header); + count = count.checked_add(1).ok_or(MetadataError::Limit)?; + edges = edges + .checked_add(entry.header.edge_count) + .ok_or(MetadataError::Limit)?; + first.get_or_insert(entry.header.object.oid); + last = Some(entry.header.object.oid); + after = entry.header.object.oid.to_vec(); + } + } + let first_oid = first.ok_or(MetadataError::Integrity)?; + let last_oid = last.ok_or(MetadataError::Integrity)?; + metadata::growth::transaction( + &mut self.connection, + &mut self.admitted, + self.limits.max_file_bytes, + |tx| { + tx.execute( + "INSERT INTO directory_identity VALUES (1,?1,?2,?3,?4,?5,?6,?7)", + params![ + self.repository.as_slice(), + self.operation.as_slice(), + self.format.as_str(), + i64::try_from(count).map_err(|_| MetadataError::Limit)?, + first_oid.as_ref(), + last_oid.as_ref(), + inventory.as_slice() + ], + )?; + Ok::<_, MetadataError>(()) + }, + )?; + self.connection + .close() + .map_err(|(_, error)| MetadataError::Sql(error))?; + self.admitted.clean_journal()?; + self.admitted.file().as_file().sync_all()?; + let size = self.admitted.file().as_file().metadata()?.len(); + if size > self.limits.max_file_bytes { + return Err(MetadataError::Limit); + } + let descriptor = RunDescriptor { + repository: self.repository, + operation: self.operation, + format: self.format, + object_count: count, + first_oid, + last_oid, + inventory_digest: inventory, + size, + digest: file_digest(self.admitted.file().path(), size)?, + }; + self.admitted.reservation().resize(size)?; + Ok(( + DirectoryRun::open_admitted(self.admitted, descriptor, self.limits.cache_kib)?, + edges, + )) + } +} diff --git a/crates/canopy-server/src/packs/input_artifact.rs b/crates/canopy-server/src/packs/input_artifact.rs new file mode 100644 index 0000000..6e4e8ea --- /dev/null +++ b/crates/canopy-server/src/packs/input_artifact.rs @@ -0,0 +1,100 @@ +//! Shared bounded metadata roots for retained input/result custody. +use super::directory::index::codec::{artifact, fixed, read_artifact}; +use canopy_object_storage::artifact::{ + ArtifactDescriptor, ArtifactKey, ArtifactKind, ArtifactStore, +}; +use cellule_runtime::codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue}; + +pub(crate) const INPUT_ROOT_BYTES: u32 = 128 << 10; +pub(crate) const MAX_INPUT_ROOT_BYTES: u32 = 256 << 10; +#[derive(Debug, thiserror::Error)] +pub enum InputRootError { + #[error("retained input root codec failed")] + Codec(#[from] CodecError), + #[error("retained input root artifact failed")] + Artifact(#[from] canopy_object_storage::artifact::ArtifactError), +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(crate) struct StoredInputRoot { + pub operation: [u8; 16], + pub artifact: ArtifactDescriptor, +} +impl StoredInputRoot { + pub fn validate(self, limit: u32) -> Result<(), CodecError> { + super::publication::codec::artifact_valid(self.operation)?; + if limit == 0 + || limit > MAX_INPUT_ROOT_BYTES + || self.artifact.size == 0 + || self.artifact.size > u64::from(limit) + || self.artifact.manifest_digest == [0; 32] + { + return Err(CodecError::Invalid("retained input root")); + } + Ok(()) + } + fn key(self) -> ArtifactKey { + ArtifactKey { + operation: self.operation, + binding_digest: self.artifact.digest, + kind: ArtifactKind::InputRoot, + } + } + pub async fn upload( + store: &ArtifactStore, + operation: [u8; 16], + value: &T, + limit: u32, + ) -> Result { + super::publication::codec::artifact_valid(operation)?; + if limit == 0 || limit > MAX_INPUT_ROOT_BYTES { + return Err(CodecError::Limit.into()); + } + let mut e = BoundedEncoder::new(limit)?; + value.encode(&mut e)?; + let bytes = e.finish(); + let digest = *blake3::hash(&bytes).as_bytes(); + let key = ArtifactKey { + operation, + binding_digest: digest, + kind: ArtifactKind::InputRoot, + }; + let artifact = store + .put(key, bytes.len() as u64, digest, &mut bytes.as_slice()) + .await?; + Ok(Self { + operation, + artifact, + }) + } + pub async fn read( + self, + store: &ArtifactStore, + limit: u32, + ) -> Result { + self.validate(limit)?; + let mut reader = store.read(self.key(), self.artifact).await?; + let mut bytes = Vec::with_capacity(self.artifact.size as usize); + while let Some(part) = reader.next().await? { + bytes.extend_from_slice(&part); + } + let mut d = BoundedDecoder::new(&bytes, limit)?; + let value = T::decode(&mut d)?; + d.finish()?; + Ok(value) + } +} +impl WireValue for StoredInputRoot { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.validate(MAX_INPUT_ROOT_BYTES)?; + e.write_bytes(&self.operation)?; + artifact(e, self.artifact) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + operation: fixed(d)?, + artifact: read_artifact(d)?, + }; + value.validate(MAX_INPUT_ROOT_BYTES)?; + Ok(value) + } +} diff --git a/crates/canopy-server/src/packs/metadata/growth.rs b/crates/canopy-server/src/packs/metadata/growth.rs new file mode 100644 index 0000000..f5c05ab --- /dev/null +++ b/crates/canopy-server/src/packs/metadata/growth.rs @@ -0,0 +1,245 @@ +//! Admission for growing private SQLite files. Immutable artifacts keep their +//! existing format; only disposable construction-time capacity changes. +use super::{AdmittedFile, DiskBudget, DiskReservation, MetadataError}; +use rusqlite::{Connection, Transaction}; +use std::error::Error; + +pub(in crate::packs) const INITIAL_BYTES: u64 = 64 << 10; +const PAGE_BYTES: u64 = 4096; +const SCRATCH_FACTOR: u64 = 3; + +pub(in crate::packs) fn reserve( + budget: &DiskBudget, + maximum: u64, +) -> Result { + Ok(budget.try_reserve(maximum.min(INITIAL_BYTES) * SCRATCH_FACTOR)?) +} + +pub(in crate::packs) fn configure( + connection: &Connection, + file: &mut AdmittedFile, +) -> Result<(), MetadataError> { + let pages = file.reservation().bytes() / SCRATCH_FACTOR / PAGE_BYTES; + connection.pragma_update(None, "max_page_count", pages)?; + let actual: u64 = connection.pragma_query_value(None, "max_page_count", |r| r.get(0))?; + if actual != pages { + return Err(MetadataError::Integrity); + } + Ok(()) +} + +/// The body must have no externally visible effects. Return cursors, digests, +/// and counters as the transaction's result and adopt them only after success. +/// A retry replays the complete body after SQLite has rolled it back. Resources +/// backing replayable inputs must remain owned throughout all attempts. +pub(in crate::packs) fn transaction( + connection: &mut Connection, + file: &mut AdmittedFile, + maximum: u64, + mut body: impl FnMut(&Transaction<'_>) -> Result, +) -> Result +where + E: Error + From + 'static, +{ + loop { + let result = (|| { + let tx = connection.transaction().map_err(MetadataError::from)?; + let value = body(&tx)?; + tx.commit().map_err(MetadataError::from)?; + Ok(value) + })(); + let error = match result { + Ok(value) => return Ok(value), + Err(error) => error, + }; + if !disk_full(&error) { + return Err(error); + } + // Drop/automatic rollback may have failed. Never replay a body while a + // previous transaction remains live, even if more disk becomes free. + if !connection.is_autocommit() { + return Err(MetadataError::Integrity.into()); + } + let current: u64 = connection + .pragma_query_value(None, "max_page_count", |r| r.get(0)) + .map_err(MetadataError::from)?; + let bytes = current + .checked_mul(PAGE_BYTES) + .ok_or(MetadataError::Limit)?; + if bytes >= maximum { + return Err(MetadataError::Limit.into()); + } + let next = bytes + .checked_mul(2) + .ok_or(MetadataError::Limit)? + .min(maximum); + // Admit database + rollback journal + conservative SQLite overhead + // before allowing SQLite to allocate any additional pages. A failed + // admission leaves both the old cap and its reservation unchanged. + file.reservation() + .resize( + next.checked_mul(SCRATCH_FACTOR) + .ok_or(MetadataError::Limit)?, + ) + .map_err(MetadataError::from)?; + connection + .pragma_update(None, "max_page_count", next / PAGE_BYTES) + .map_err(MetadataError::from)?; + let actual: u64 = connection + .pragma_query_value(None, "max_page_count", |r| r.get(0)) + .map_err(MetadataError::from)?; + if actual != next / PAGE_BYTES { + return Err(MetadataError::Integrity.into()); + } + } +} + +fn disk_full(error: &(dyn Error + 'static)) -> bool { + let mut current = Some(error); + while let Some(error) = current { + if let Some(sql) = error.downcast_ref::() { + return sql.sqlite_error_code() == Some(rusqlite::ErrorCode::DiskFull); + } + current = error.source(); + } + false +} + +#[cfg(test)] +mod tests { + use super::*; + type Result = std::result::Result>; + + fn scratch(budget: &DiskBudget, maximum: u64) -> Result<(Connection, AdmittedFile)> { + let mut file = + AdmittedFile::new(tempfile::NamedTempFile::new()?, reserve(budget, maximum)?); + let mut db = Connection::open(file.file().path())?; + db.execute_batch("PRAGMA page_size=4096; PRAGMA journal_mode=DELETE; PRAGMA mmap_size=0;")?; + configure(&db, &mut file)?; + transaction(&mut db, &mut file, maximum, |tx| { + tx.execute_batch("CREATE TABLE records(id INTEGER PRIMARY KEY, body BLOB NOT NULL)")?; + Ok::<_, MetadataError>(()) + })?; + Ok((db, file)) + } + + #[test] + fn growth_replays_only_rolled_back_rows_and_reserves_before_pages() -> Result { + let maximum = 1 << 20; + let budget = DiskBudget::new(maximum * SCRATCH_FACTOR); + let (mut db, mut file) = scratch(&budget, maximum)?; + assert_eq!(budget.used(), INITIAL_BYTES * SCRATCH_FACTOR); + let path = file.file().path().to_owned(); + let mut attempts = 0; + let count = transaction(&mut db, &mut file, maximum, |tx| { + attempts += 1; + // A capacity failure after a successful prefix must not survive. + assert_eq!( + tx.query_row("SELECT count(*) FROM records", [], |r| r.get::<_, u64>(0))?, + 0 + ); + for id in 0..100 { + tx.execute("INSERT INTO records VALUES(?1,zeroblob(4096))", [id])?; + let mut journal = path.as_os_str().to_owned(); + journal.push("-journal"); + let actual = std::fs::metadata(&path)?.len() + + std::fs::metadata(std::path::Path::new(&journal)) + .map(|m| m.len()) + .unwrap_or(0); + assert!(actual <= budget.used()); + } + Ok::<_, MetadataError>(100) + })?; + assert!(attempts >= 3); + assert_eq!(count, 100); + assert_eq!( + db.query_row("SELECT count(*) FROM records", [], |r| r.get::<_, u64>(0))?, + 100 + ); + let cap: u64 = db.pragma_query_value(None, "max_page_count", |r| r.get(0))?; + assert_eq!(budget.used(), cap * PAGE_BYTES * SCRATCH_FACTOR); + assert!(budget.used() < maximum * SCRATCH_FACTOR); + drop(db); + drop(file); + assert_eq!(budget.used(), 0); + assert!(!path.exists()); + Ok(()) + } + + #[test] + fn denied_growth_keeps_cap_credit_and_prior_committed_rows() -> Result { + let maximum = 1 << 20; + let budget = DiskBudget::new(INITIAL_BYTES * SCRATCH_FACTOR); + let (mut db, mut file) = scratch(&budget, maximum)?; + db.execute("INSERT INTO records VALUES(-1,x'01')", [])?; + let error = transaction(&mut db, &mut file, maximum, |tx| { + for id in 0..100 { + tx.execute("INSERT INTO records VALUES(?1,zeroblob(4096))", [id])?; + } + Ok::<_, MetadataError>(()) + }) + .unwrap_err(); + assert!(matches!(error, MetadataError::Budget(_))); + assert!(db.is_autocommit()); + assert_eq!(budget.used(), INITIAL_BYTES * SCRATCH_FACTOR); + assert_eq!( + db.pragma_query_value(None, "max_page_count", |r| r.get::<_, u64>(0))?, + INITIAL_BYTES / PAGE_BYTES + ); + assert_eq!( + db.query_row("SELECT count(*) FROM records WHERE id=-1", [], |r| r + .get::<_, u64>(0))?, + 1 + ); + assert_eq!( + db.query_row("SELECT count(*) FROM records WHERE id>=0", [], |r| r + .get::<_, u64>(0))?, + 0 + ); + drop(db); + drop(file); + assert_eq!(budget.used(), 0); + Ok(()) + } + + #[test] + fn configured_ceiling_and_domain_errors_never_escape_or_retry() -> Result { + // A non-power-of-two ceiling proves the last growth is clipped exactly. + let maximum = 96 << 10; + let budget = DiskBudget::new(maximum * SCRATCH_FACTOR); + let (mut db, mut file) = scratch(&budget, maximum)?; + let mut attempts = 0; + let error = transaction(&mut db, &mut file, maximum, |tx| { + attempts += 1; + for id in 0..100 { + tx.execute("INSERT INTO records VALUES(?1,zeroblob(4096))", [id])?; + } + Ok::<_, MetadataError>(()) + }) + .unwrap_err(); + assert!(matches!(error, MetadataError::Limit)); + assert_eq!(attempts, 2); + assert_eq!(budget.used(), maximum * SCRATCH_FACTOR); + assert_eq!( + db.query_row("SELECT count(*) FROM records", [], |r| r.get::<_, u64>(0))?, + 0 + ); + attempts = 0; + let error = transaction(&mut db, &mut file, maximum, |tx| { + attempts += 1; + tx.execute("INSERT INTO records VALUES(1,x'01')", [])?; + Err::<(), _>(MetadataError::IdentityConflict) + }) + .unwrap_err(); + assert!(matches!(error, MetadataError::IdentityConflict)); + assert_eq!(attempts, 1); + assert_eq!( + db.query_row("SELECT count(*) FROM records", [], |r| r.get::<_, u64>(0))?, + 0 + ); + drop(db); + drop(file); + assert_eq!(budget.used(), 0); + Ok(()) + } +} diff --git a/crates/canopy-server/src/packs/metadata/mod.rs b/crates/canopy-server/src/packs/metadata/mod.rs new file mode 100644 index 0000000..345d9ac --- /dev/null +++ b/crates/canopy-server/src/packs/metadata/mod.rs @@ -0,0 +1,461 @@ +//! Checked, immutable SQLite metadata shards with bounded buffers/cache. +//! +//! A shard covers an exact contiguous ordinal slice of a checked native index. +//! Shards can therefore split very large physical packs without one unbounded +//! metadata file. Catalog verification must cover every ordinal exactly once. + +use crate::{ObjectFormat, ObjectId, ObjectKind, git_format::pack_index::PackIndex}; +use cellule_ltx::{DiskBudget, DiskReservation, LtxError}; +use rusqlite::{Connection, OpenFlags, OptionalExtension, params}; +use std::{ + fs::File, + io::{self, Read}, + path::Path, + sync::Mutex, +}; + +pub(in crate::packs) mod growth; +mod writer; +pub use writer::MetadataBuilder; +pub(crate) mod transport; +pub use transport::StoredSegment; + +pub const PAGE_OBJECTS: usize = 512; +const APPLICATION_ID: u32 = 1_128_353_357; +const SCHEMA: &str = include_str!("schema.sql"); + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct CanonicalObject { + pub oid: ObjectId, + pub kind: ObjectKind, + pub size: u64, + pub digest: [u8; 32], +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ObjectHeader { + pub object: CanonicalObject, + pub edge_count: u64, + pub edge_digest: [u8; 32], +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct TypedEdge { + pub child: ObjectId, + pub expected_kind: ObjectKind, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct SegmentIdentity { + pub repository: [u8; 16], + pub operation: [u8; 16], + pub format: ObjectFormat, + pub pack_digest: [u8; 32], + pub git_checksum: ObjectId, + pub first_ordinal: u32, + pub object_count: u32, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct SegmentDescriptor { + pub identity: SegmentIdentity, + pub edge_count: u64, + pub inventory_digest: [u8; 32], + pub first_oid: ObjectId, + pub last_oid: ObjectId, + pub size: u64, + pub digest: [u8; 32], +} + +#[derive(Clone, Copy, Debug)] +pub struct MetadataLimits { + /// Main database ceiling. Builders admit 64 KiB initially (or this ceiling + /// if smaller), then double before raising SQLite's cap; each cap reserves + /// 3x for database, rollback journal, and overhead. Sealing retains only the + /// exact immutable file size. + pub max_file_bytes: u64, + pub cache_kib: u32, +} +impl Default for MetadataLimits { + fn default() -> Self { + Self { + max_file_bytes: 256 << 20, + cache_kib: 8 << 10, + } + } +} + +#[derive(Debug, thiserror::Error)] +pub enum MetadataError { + #[error("metadata I/O failed")] + Io(#[from] io::Error), + #[error("metadata SQLite operation failed")] + Sql(#[source] rusqlite::Error), + #[error("metadata disk admission failed")] + Budget(#[from] LtxError), + #[error("metadata identity, graph or artifact is invalid")] + Integrity, + #[error("the same Git object ID has conflicting canonical metadata")] + IdentityConflict, + #[error("the expected object placement is absent or changed")] + PlacementConflict, + #[error("metadata input exceeds its bounded batch or shard limit")] + Limit, + #[error("metadata task failed")] + Task(#[from] tokio::task::JoinError), + #[error("metadata artifact transfer failed")] + Artifact(#[from] canopy_object_storage::artifact::ArtifactError), +} +impl From for MetadataError { + fn from(error: rusqlite::Error) -> Self { + // Preserve the SQLite cause so admitted builders can distinguish a + // rolled-back capacity failure from validation/batch limits. + Self::Sql(error) + } +} + +pub(super) fn kind_code(kind: ObjectKind) -> u8 { + match kind { + ObjectKind::Blob => 1, + ObjectKind::Tree => 2, + ObjectKind::Commit => 3, + ObjectKind::Tag => 4, + } +} +pub(super) fn kind(name: &str) -> rusqlite::Result { + match name { + "blob" => Ok(ObjectKind::Blob), + "tree" => Ok(ObjectKind::Tree), + "commit" => Ok(ObjectKind::Commit), + "tag" => Ok(ObjectKind::Tag), + _ => Err(rusqlite::Error::InvalidQuery), + } +} +pub(super) fn oid(bytes: Vec) -> rusqlite::Result { + bytes.try_into().map_err(|_| rusqlite::Error::InvalidQuery) +} +pub(super) fn digest(bytes: Vec) -> rusqlite::Result<[u8; 32]> { + bytes.try_into().map_err(|_| rusqlite::Error::InvalidQuery) +} +pub(super) fn unsigned(value: i64) -> rusqlite::Result { + value.try_into().map_err(|_| rusqlite::Error::InvalidQuery) +} +pub(super) fn canonical(row: &rusqlite::Row<'_>) -> rusqlite::Result { + Ok(CanonicalObject { + oid: oid(row.get(0)?)?, + kind: kind(&row.get::<_, String>(1)?)?, + size: unsigned(row.get(2)?)?, + digest: digest(row.get(3)?)?, + }) +} +pub(super) fn header(row: &rusqlite::Row<'_>) -> rusqlite::Result { + Ok(ObjectHeader { + object: canonical(row)?, + edge_count: unsigned(row.get(4)?)?, + edge_digest: digest(row.get(5)?)?, + }) +} + +pub(super) fn edge_seed(parent: ObjectId) -> [u8; 32] { + let mut hash = blake3::Hasher::new(); + hash.update(b"canopy.edges.v1\0"); + hash.update(&[parent.format().bytes() as u8]); + hash.update(&parent); + *hash.finalize().as_bytes() +} +pub(super) fn fold(previous: [u8; 32], ordinal: u64, record: &[u8]) -> [u8; 32] { + let mut hash = blake3::Hasher::new(); + hash.update(&previous); + hash.update(&ordinal.to_le_bytes()); + hash.update(record); + *hash.finalize().as_bytes() +} +pub(super) fn inventory_seed(identity: SegmentIdentity) -> [u8; 32] { + let mut hash = blake3::Hasher::new(); + hash.update(b"canopy.segment.v1\0"); + hash.update(&[identity.format.bytes() as u8]); + hash.update(&identity.pack_digest); + hash.update(&u64::from(identity.first_ordinal).to_le_bytes()); + hash.update(&u64::from(identity.object_count).to_le_bytes()); + *hash.finalize().as_bytes() +} +pub(super) fn fold_header(previous: [u8; 32], ordinal: u64, header: ObjectHeader) -> [u8; 32] { + let mut record = [0; 113]; + let width = header.object.oid.len(); + record[..width].copy_from_slice(&header.object.oid); + record[width] = kind_code(header.object.kind); + record[width + 1..width + 9].copy_from_slice(&header.object.size.to_le_bytes()); + record[width + 9..width + 41].copy_from_slice(&header.object.digest); + record[width + 41..width + 49].copy_from_slice(&header.edge_count.to_le_bytes()); + record[width + 49..width + 81].copy_from_slice(&header.edge_digest); + fold(previous, ordinal, &record[..width + 81]) +} + +fn validate_identity(identity: SegmentIdentity) -> Result<(), MetadataError> { + if identity.git_checksum.format() != identity.format + || identity.git_checksum.is_zero() + || identity.object_count == 0 + || identity + .first_ordinal + .checked_add(identity.object_count) + .is_none() + { + return Err(MetadataError::Integrity); + } + Ok(()) +} +pub(crate) fn file_digest(path: &Path, expected_size: u64) -> Result<[u8; 32], MetadataError> { + file_digest_with(path, expected_size, || Ok(())) +} +pub(super) fn file_digest_with>( + path: &Path, + expected_size: u64, + mut checkpoint: impl FnMut() -> Result<(), E>, +) -> Result<[u8; 32], E> { + checkpoint()?; + let mut file = File::open(path).map_err(MetadataError::from)?; + if file.metadata().map_err(MetadataError::from)?.len() != expected_size { + return Err(MetadataError::Integrity.into()); + } + let mut hash = blake3::Hasher::new(); + let mut buffer = [0; 64 << 10]; + let mut size = 0_u64; + loop { + checkpoint()?; + let count = file.read(&mut buffer).map_err(MetadataError::from)?; + if count == 0 { + break; + } + size = size.checked_add(count as u64).ok_or(MetadataError::Limit)?; + if size > expected_size { + return Err(MetadataError::Integrity.into()); + } + hash.update(&buffer[..count]); + } + if size != expected_size { + return Err(MetadataError::Integrity.into()); + } + Ok(*hash.finalize().as_bytes()) +} + +/// Owns the verified local file and disk admission. Readers share bounded SQLite +/// lookup, never an in-memory copy of all rows. The file must stay immutable. +pub struct MetadataSegment { + connection: Mutex, + admitted: AdmittedFile, + descriptor: SegmentDescriptor, +} +// Reader slots and the private workspace follow the file through queued jobs. +pub(super) struct ReaderAdmission { + pub(super) _root: std::sync::Arc, + pub(super) _slot: tokio::sync::OwnedSemaphorePermit, +} + +// Keep file cleanup ahead of budget release on validation errors too. +pub(super) struct AdmittedFile { + file: Option, + reader: Option, + workspace: Option>, + // Rust drops fields in declaration order after Drop. Keep workspace/reader + // cleanup ahead of admission release even after successful file.close(). + reservation: Option, +} +impl AdmittedFile { + pub(super) fn new(file: tempfile::NamedTempFile, reservation: DiskReservation) -> Self { + Self { + file: Some(file), + reservation: Some(reservation), + reader: None, + workspace: None, + } + } + pub(super) fn retain_workspace(&mut self, workspace: std::sync::Arc) { + self.workspace = Some(workspace); + } + pub(super) fn workspace(&self) -> Option> { + self.workspace.clone() + } + pub(super) fn with_reader(mut self, reader: Option) -> Self { + self.reader = reader; + self + } + pub(super) fn file(&self) -> &tempfile::NamedTempFile { + self.file.as_ref().expect("admitted file is owned") + } + pub(super) fn file_mut(&mut self) -> &mut tempfile::NamedTempFile { + self.file.as_mut().expect("admitted file is owned") + } + pub(super) fn reservation(&mut self) -> &mut DiskReservation { + self.reservation.as_mut().expect("admission is owned") + } + pub(super) fn clean_journal(&self) -> io::Result<()> { + let mut path = self.file().path().as_os_str().to_owned(); + path.push("-journal"); + match std::fs::remove_file(Path::new(&path)) { + Ok(()) => Ok(()), + Err(error) if error.kind() == io::ErrorKind::NotFound => Ok(()), + Err(error) => Err(error), + } + } +} +impl Drop for AdmittedFile { + fn drop(&mut self) { + let journal = self.clean_journal(); + let Some(file) = self.file.take() else { + return; + }; + let main = file.close(); + if let Err(error) = journal.and(main) { + tracing::warn!(%error, "metadata cleanup failed; retaining disk admission for workspace recovery"); + if let Some(reservation) = self.reservation.take() { + std::mem::forget(reservation); + } + } + } +} +impl MetadataSegment { + /// The caller owns/admitted the downloaded file and supplies a descriptor + /// from a certified catalog. Bytes are hashed before SQLite opens them. + pub fn open( + file: tempfile::NamedTempFile, + reservation: DiskReservation, + descriptor: SegmentDescriptor, + cache_kib: u32, + ) -> Result { + Self::open_admitted(AdmittedFile::new(file, reservation), descriptor, cache_kib) + } + fn open_admitted( + mut admitted: AdmittedFile, + descriptor: SegmentDescriptor, + cache_kib: u32, + ) -> Result { + validate_identity(descriptor.identity)?; + if cache_kib == 0 + || cache_kib > i32::MAX as u32 + || descriptor.size > admitted.reservation().bytes() + || descriptor.first_oid.format() != descriptor.identity.format + || descriptor.last_oid.format() != descriptor.identity.format + || descriptor.first_oid.is_zero() + || descriptor.first_oid > descriptor.last_oid + || file_digest(admitted.file().path(), descriptor.size)? != descriptor.digest + { + return Err(MetadataError::Integrity); + } + let connection = Connection::open_with_flags( + admitted.file().path(), + OpenFlags::SQLITE_OPEN_READ_ONLY | OpenFlags::SQLITE_OPEN_NO_MUTEX, + )?; + connection.execute_batch( + "PRAGMA query_only=ON; PRAGMA trusted_schema=OFF; PRAGMA mmap_size=0;", + )?; + connection.pragma_update(None, "cache_size", -(cache_kib as i64))?; + let app: u32 = connection.pragma_query_value(None, "application_id", |row| row.get(0))?; + let version: u32 = connection.pragma_query_value(None, "user_version", |row| row.get(0))?; + if app != APPLICATION_ID || version != 1 { + return Err(MetadataError::Integrity); + } + let stored = connection.query_row("SELECT repository_id, operation_id, object_format, pack_digest, git_checksum, first_ordinal, object_count, edge_count, inventory_digest FROM segment_identity WHERE singleton = 1", [], |row| { + Ok((row.get::<_, Vec>(0)?,row.get::<_, Vec>(1)?,row.get::<_, String>(2)?,row.get::<_, Vec>(3)?,row.get::<_, Vec>(4)?,row.get::<_, u32>(5)?,row.get::<_, u32>(6)?,row.get::<_, u64>(7)?,row.get::<_, Vec>(8)?)) + })?; + let identity = descriptor.identity; + if stored + != ( + identity.repository.to_vec(), + identity.operation.to_vec(), + identity.format.as_str().into(), + identity.pack_digest.to_vec(), + identity.git_checksum.to_vec(), + identity.first_ordinal, + identity.object_count, + descriptor.edge_count, + descriptor.inventory_digest.to_vec(), + ) + { + return Err(MetadataError::Integrity); + } + Ok(Self { + admitted, + descriptor, + connection: Mutex::new(connection), + }) + } + pub fn descriptor(&self) -> SegmentDescriptor { + self.descriptor + } + pub fn path(&self) -> &Path { + self.admitted.file().path() + } + fn connection(&self) -> Result, MetadataError> { + self.connection.lock().map_err(|_| MetadataError::Integrity) + } + pub fn header(&self, oid: ObjectId) -> Result, MetadataError> { + Ok(self.headers(&[oid])?.pop().flatten()) + } + pub fn headers(&self, ids: &[ObjectId]) -> Result>, MetadataError> { + if ids.len() > PAGE_OBJECTS { + return Err(MetadataError::Limit); + } + let connection = self.connection()?; + let mut statement = connection.prepare_cached( + "SELECT oid, kind, size, digest, edge_count, edge_digest FROM objects WHERE oid = ?1", + )?; + ids.iter() + .map(|oid| { + if oid.format() != self.descriptor.identity.format { + return Ok(None); + } + Ok(statement.query_row([oid.as_ref()], header).optional()?) + }) + .collect() + } + pub fn headers_after( + &self, + after: Option, + ) -> Result, MetadataError> { + if after.is_some_and(|oid| oid.format() != self.descriptor.identity.format) { + return Err(MetadataError::Integrity); + } + let connection = self.connection()?; + let mut statement = connection.prepare_cached("SELECT oid, kind, size, digest, edge_count, edge_digest FROM objects WHERE oid > ?1 ORDER BY oid LIMIT ?2")?; + Ok(statement + .query_map( + params![ + after.as_ref().map_or(&[][..], AsRef::<[u8]>::as_ref), + PAGE_OBJECTS as i64 + ], + header, + )? + .collect::>()?) + } + pub fn edges_after( + &self, + parent: ObjectId, + after: Option, + ) -> Result, MetadataError> { + if parent.format() != self.descriptor.identity.format + || after.is_some_and(|oid| oid.format() != parent.format()) + { + return Err(MetadataError::Integrity); + } + let connection = self.connection()?; + let mut statement = connection.prepare_cached("SELECT child, expected_kind FROM object_edges WHERE parent = ?1 AND child > ?2 ORDER BY child LIMIT ?3")?; + Ok(statement + .query_map( + params![ + parent.as_ref(), + after.as_ref().map_or(&[][..], AsRef::<[u8]>::as_ref), + PAGE_OBJECTS as i64 + ], + |row| { + Ok(TypedEdge { + child: oid(row.get(0)?)?, + expected_kind: kind(&row.get::<_, String>(1)?)?, + }) + }, + )? + .collect::>()?) + } +} + +#[cfg(test)] +pub(super) mod tests; diff --git a/crates/canopy-server/src/packs/metadata/schema.sql b/crates/canopy-server/src/packs/metadata/schema.sql new file mode 100644 index 0000000..3a78459 --- /dev/null +++ b/crates/canopy-server/src/packs/metadata/schema.sql @@ -0,0 +1,39 @@ +PRAGMA application_id = 1128353357; +PRAGMA user_version = 1; + +CREATE TABLE segment_identity ( + singleton INTEGER PRIMARY KEY CHECK(singleton = 1), + repository_id BLOB NOT NULL CHECK(length(repository_id) = 16), + operation_id BLOB NOT NULL CHECK(length(operation_id) = 16), + object_format TEXT NOT NULL CHECK(object_format IN ('sha1', 'sha256')), + pack_digest BLOB NOT NULL CHECK(length(pack_digest) = 32), + git_checksum BLOB NOT NULL, + first_ordinal INTEGER NOT NULL CHECK(first_ordinal BETWEEN 0 AND 4294967295), + object_count INTEGER NOT NULL CHECK(object_count BETWEEN 1 AND 4294967295), + edge_count INTEGER NOT NULL CHECK(edge_count >= 0), + inventory_digest BLOB NOT NULL CHECK(length(inventory_digest) = 32), + CHECK(length(git_checksum) = CASE object_format WHEN 'sha1' THEN 20 ELSE 32 END), + CHECK(first_ordinal + object_count <= 4294967295) +) WITHOUT ROWID; + +-- Canonical identity fields and typed graph meaning reuse the Repository Cell +-- model. There are no SQL object bodies or preferred-location updates here. +CREATE TABLE objects ( + oid BLOB PRIMARY KEY CHECK(length(oid) IN (20, 32)), + kind TEXT NOT NULL CHECK(kind IN ('blob', 'tree', 'commit', 'tag')), + size INTEGER NOT NULL CHECK(size >= 0), + digest BLOB NOT NULL CHECK(length(digest) = 32), + edge_count INTEGER CHECK(edge_count >= 0), + edge_digest BLOB CHECK(length(edge_digest) = 32), + CHECK((edge_count IS NULL) = (edge_digest IS NULL)), + CHECK(kind != 'blob' OR edge_count IS NULL OR edge_count = 0) +) WITHOUT ROWID; + +CREATE TABLE object_edges ( + parent BLOB NOT NULL REFERENCES objects(oid), + child BLOB NOT NULL CHECK(length(child) IN (20, 32)), + expected_kind TEXT NOT NULL CHECK(expected_kind IN ('blob', 'tree', 'commit', 'tag')), + PRIMARY KEY(parent, child) +) WITHOUT ROWID; +-- A child may be in a dependency generation, rather than this physical pack. +-- The catalog verifier must certify that dependency before publication. diff --git a/crates/canopy-server/src/packs/metadata/tests.rs b/crates/canopy-server/src/packs/metadata/tests.rs new file mode 100644 index 0000000..da9da7c --- /dev/null +++ b/crates/canopy-server/src/packs/metadata/tests.rs @@ -0,0 +1,662 @@ +use super::*; +use crate::git_objects::GitObjects; +use std::{collections::BTreeMap, process::Stdio}; +use tokio::{io::AsyncWriteExt, process::Command}; + +type Result = std::result::Result>; + +async fn git(path: &Path, args: &[&str], input: Option>) -> Result> { + let mut command = Command::new("git"); + command + .env_clear() + .env("PATH", std::env::var_os("PATH").ok_or("PATH")?) + .env("HOME", path) + .env("GIT_CONFIG_NOSYSTEM", "1") + .env( + "GIT_CONFIG_GLOBAL", + if cfg!(windows) { "NUL" } else { "/dev/null" }, + ) + .env("LC_ALL", "C") + .arg("-C") + .arg(path) + .args(args) + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()); + let mut child = command.spawn()?; + if let Some(input) = input { + child.stdin.take().ok_or("stdin")?.write_all(&input).await?; + } else { + drop(child.stdin.take()); + } + let output = child.wait_with_output().await?; + if !output.status.success() { + return Err(String::from_utf8_lossy(&output.stderr).into_owned().into()); + } + Ok(output.stdout) +} + +pub(in crate::packs) struct Fixture { + pub(in crate::packs) root: tempfile::TempDir, + pub(in crate::packs) index: PackIndex, + pub(in crate::packs) identity: SegmentIdentity, + pub(in crate::packs) objects: BTreeMap)>, +} +pub(in crate::packs) async fn fixture(format: ObjectFormat, blobs: usize) -> Result { + let root = tempfile::TempDir::new()?; + git( + root.path(), + &[ + "init", + "--bare", + &format!("--object-format={}", format.as_str()), + ], + None, + ) + .await?; + // fast-import avoids one process per fixture object. The input is a test + // fixture; production verification streams native bodies into bounded SQL. + let mut input = b"commit refs/heads/main\ncommitter Metadata Test 1 +0000\ndata 7\nfixture\n".to_vec(); + for n in 0..blobs { + let body = format!("fixture body {n}\n"); + input.extend_from_slice( + format!("M 100644 inline file-{n}\ndata {}\n{body}", body.len()).as_bytes(), + ); + } + input.extend_from_slice(b"\n"); + git(root.path(), &["fast-import", "--quiet"], Some(input)).await?; + git( + root.path(), + &[ + "-c", + "user.name=Test", + "-c", + "user.email=test@example.invalid", + "tag", + "-a", + "metadata", + "-m", + "annotation", + "refs/heads/main", + ], + None, + ) + .await?; + git(root.path(), &["repack", "-ad"], None).await?; + let index_path = std::fs::read_dir(root.path().join("objects/pack"))? + .filter_map(|entry| entry.ok().map(|entry| entry.path())) + .find(|path| path.extension().is_some_and(|ext| ext == "idx")) + .ok_or("index")?; + let index = PackIndex::open(&index_path, format)?; + let pack = std::fs::read(index_path.with_extension("pack"))?; + let identity = SegmentIdentity { + repository: [1; 16], + operation: [2; 16], + format, + pack_digest: *blake3::hash(&pack).as_bytes(), + git_checksum: index.pack_checksum(), + first_ordinal: 0, + object_count: index.len(), + }; + let ids = index.ids().collect::>>()?; + let mut reader = GitObjects::packed( + root.path(), + ids, + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )?; + let mut objects = BTreeMap::new(); + while let Some(oid) = reader.next().await? { + let (kind, body) = reader.read(oid).await?.body().await?; + let edges = crate::graph::edges(format, kind, &body) + .ok_or("graph")? + .into_iter() + .map(|(child, expected_kind)| { + Ok(TypedEdge { + child, + expected_kind: expected_kind.ok_or("typed edge")?, + }) + }) + .collect::>>()?; + objects.insert( + oid, + ( + CanonicalObject { + oid, + kind, + size: body.len() as u64, + digest: *blake3::hash(&body).as_bytes(), + }, + edges, + ), + ); + } + reader.finish().await?; + Ok(Fixture { + root, + index, + identity, + objects, + }) +} +pub(in crate::packs) fn limits() -> MetadataLimits { + MetadataLimits { + max_file_bytes: 16 << 20, + cache_kib: 64, + } +} +pub(in crate::packs) fn builder( + fixture: &Fixture, + budget: DiskBudget, + identity: SegmentIdentity, +) -> Result { + Ok(MetadataBuilder::new( + fixture.root.path(), + budget, + identity, + limits(), + )?) +} +pub(in crate::packs) fn fill( + builder: &mut MetadataBuilder, + objects: &[(CanonicalObject, Vec)], +) -> Result { + for objects in objects.chunks(PAGE_OBJECTS) { + builder.put_objects( + &objects + .iter() + .map(|(object, _)| *object) + .collect::>(), + )?; + } + for (object, edges) in objects { + for edges in edges.chunks(PAGE_OBJECTS) { + builder.put_edges(object.oid, edges)?; + } + } + Ok(()) +} + +pub(in crate::packs) async fn prepared_segment() -> Result<( + tempfile::TempDir, + std::sync::Arc, + DiskBudget, +)> { + let fixture = fixture(ObjectFormat::Sha256, 4).await?; + let budget = DiskBudget::new(128 << 20); + let mut writer = builder(&fixture, budget.clone(), fixture.identity)?; + fill( + &mut writer, + &fixture.objects.values().cloned().collect::>(), + )?; + let segment = std::sync::Arc::new(writer.seal(&fixture.index)?); + Ok((fixture.root, segment, budget)) +} + +#[tokio::test] +async fn native_objects_roundtrip_with_deterministic_inventory_and_duplicate_replay() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = fixture(format, 4).await?; + let budget = DiskBudget::new(128 << 20); + let objects = fixture.objects.values().cloned().collect::>(); + let mut first = builder(&fixture, budget.clone(), fixture.identity)?; + fill(&mut first, &objects)?; + fill(&mut first, &objects)?; + let first = first.seal(&fixture.index)?; + let mut second = builder(&fixture, budget.clone(), fixture.identity)?; + fill(&mut second, &objects.into_iter().rev().collect::>())?; + let second = second.seal(&fixture.index)?; + assert_eq!( + first.descriptor().inventory_digest, + second.descriptor().inventory_digest + ); + assert_eq!( + first.descriptor().identity.object_count, + fixture.objects.len() as u32 + ); + for (oid, (object, expected_edges)) in &fixture.objects { + let header = first.header(*oid)?.ok_or("header")?; + assert_eq!(header.object, *object); + assert_eq!(header.edge_count, expected_edges.len() as u64); + assert_eq!(first.edges_after(*oid, None)?, *expected_edges); + } + assert!( + first + .header(if format == ObjectFormat::Sha1 { + ObjectId::Sha256([3; 32]) + } else { + ObjectId::Sha1([3; 20]) + })? + .is_none() + ); + drop(first); + drop(second); + assert_eq!(budget.used(), 0); + } + Ok(()) +} + +#[tokio::test] +async fn shard_page_limit_rolls_back_then_allows_a_valid_smaller_inventory() -> Result { + let fixture = fixture(ObjectFormat::Sha256, 4).await?; + let budget = DiskBudget::new(1 << 20); + let mut writer = MetadataBuilder::new( + fixture.root.path(), + budget.clone(), + fixture.identity, + MetadataLimits { + max_file_bytes: 32 << 10, + cache_kib: 64, + }, + )?; + let fake = (1_u32..=512) + .map(|n| { + let mut oid = [0; 32]; + oid[..4].copy_from_slice(&n.to_be_bytes()); + CanonicalObject { + oid: ObjectId::Sha256(oid), + kind: ObjectKind::Blob, + size: 0, + digest: *blake3::hash(b"").as_bytes(), + } + }) + .collect::>(); + assert!(matches!( + writer.put_objects(&fake), + Err(MetadataError::Limit) + )); + fill( + &mut writer, + &fixture.objects.values().cloned().collect::>(), + )?; + let segment = writer.seal(&fixture.index)?; + assert!(segment.descriptor().size <= 32 << 10); + assert_eq!(budget.used(), segment.descriptor().size); + drop(segment); + assert_eq!(budget.used(), 0); + Ok(()) +} + +#[cfg(unix)] +#[tokio::test] +async fn failed_file_cleanup_retains_disk_admission_for_workspace_recovery() -> Result { + let (root, segment, budget) = prepared_segment().await?; + let charged = budget.used(); + let old = segment.path().to_owned(); + let retained = root.path().join("retained-metadata"); + std::fs::rename(&old, &retained)?; + std::fs::create_dir(&old)?; + drop(segment); + assert!(retained.exists()); + assert_eq!(budget.used(), charged); + std::fs::remove_dir(&old)?; + std::fs::remove_file(&retained)?; + // A failed cleanup conservatively retains its original reservation. A + // fresh workspace startup reclaims files before allocating a fresh budget. + assert_eq!(budget.used(), charged); + Ok(()) +} + +#[tokio::test] +async fn sealed_segments_roundtrip_through_authenticated_storage_and_reject_bad_bindings() -> Result +{ + use canopy_object_storage::artifact::{ArtifactKind, ArtifactStore}; + use object_store::{ObjectStore, ObjectStoreExt, memory::InMemory}; + use std::sync::Arc; + + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = fixture(format, 20).await?; + let budget = DiskBudget::new(128 << 20); + let mut writer = builder(&fixture, budget.clone(), fixture.identity)?; + fill( + &mut writer, + &fixture.objects.values().cloned().collect::>(), + )?; + let segment = Arc::new(writer.seal(&fixture.index)?); + let store: Arc = Arc::new(InMemory::new()); + let artifacts = ArtifactStore::new(Arc::clone(&store), fixture.identity.repository); + let (first, second) = tokio::join!( + Arc::clone(&segment).upload(&artifacts), + Arc::clone(&segment).upload(&artifacts) + ); + let stored = first?; + assert_eq!(stored, second?); + let download_root = tempfile::TempDir::new()?; + let downloaded = MetadataSegment::download( + download_root.path(), + budget.clone(), + &artifacts, + stored, + limits(), + ) + .await?; + assert_eq!(downloaded.descriptor(), segment.descriptor()); + for oid in fixture.objects.keys() { + assert_eq!(downloaded.header(*oid)?, segment.header(*oid)?); + assert_eq!( + downloaded.edges_after(*oid, None)?, + segment.edges_after(*oid, None)? + ); + } + drop(downloaded); + assert_eq!(budget.used(), stored.segment.size); + + let other = ArtifactStore::new(Arc::clone(&store), [99; 16]); + assert!(matches!( + Arc::clone(&segment).upload(&other).await, + Err(MetadataError::Integrity) + )); + assert!(matches!( + MetadataSegment::download( + download_root.path(), + budget.clone(), + &other, + stored, + limits() + ) + .await, + Err(MetadataError::Integrity) + )); + let mut wrong = stored; + wrong.artifact.size += 1; + assert!(matches!( + MetadataSegment::download( + download_root.path(), + budget.clone(), + &artifacts, + wrong, + limits() + ) + .await, + Err(MetadataError::Integrity) + )); + wrong = stored; + wrong.artifact.digest[0] ^= 1; + assert!(matches!( + MetadataSegment::download( + download_root.path(), + budget.clone(), + &artifacts, + wrong, + limits() + ) + .await, + Err(MetadataError::Integrity) + )); + assert!(matches!( + MetadataSegment::download( + download_root.path(), + DiskBudget::new(stored.segment.size - 1), + &artifacts, + stored, + limits() + ) + .await, + Err(MetadataError::Budget(_)) + )); + + let path = artifacts.path( + canopy_object_storage::artifact::ArtifactKey { + operation: fixture.identity.operation, + binding_digest: fixture.identity.pack_digest, + kind: ArtifactKind::Metadata, + }, + stored.artifact.digest, + )?; + assert!(path.as_ref().contains("/metadata/")); + let part_path = canopy_object_storage::external::part(&path, 0); + store + .put( + &part_path, + bytes::Bytes::from(vec![0; stored.segment.size as usize]).into(), + ) + .await?; + assert!( + MetadataSegment::download( + download_root.path(), + budget.clone(), + &artifacts, + stored, + limits() + ) + .await + .is_err() + ); + assert_eq!(budget.used(), stored.segment.size); + assert_eq!(std::fs::read_dir(download_root.path())?.count(), 0); + drop(segment); + assert_eq!(budget.used(), 0); + } + Ok(()) +} + +#[tokio::test] +async fn header_and_typed_edge_conflicts_roll_back_entire_batches() -> Result { + let fixture = fixture(ObjectFormat::Sha1, 4).await?; + let budget = DiskBudget::new(128 << 20); + let mut writer = builder(&fixture, budget.clone(), fixture.identity)?; + fill( + &mut writer, + &fixture.objects.values().cloned().collect::>(), + )?; + let original = fixture + .objects + .values() + .find(|(object, _)| object.kind == ObjectKind::Blob) + .ok_or("blob")? + .0; + let mut extra = original; + extra.oid = ObjectId::Sha1([37; 20]); + assert!(!fixture.objects.contains_key(&extra.oid)); + let mut conflict = original; + conflict.digest = [38; 32]; + assert!(matches!( + writer.put_objects(&[extra, conflict]), + Err(MetadataError::IdentityConflict) + )); + let (tree, edges) = fixture + .objects + .values() + .find(|(object, _)| object.kind == ObjectKind::Tree) + .ok_or("tree")?; + assert!(matches!( + writer.put_edges( + tree.oid, + &[ + TypedEdge { + child: extra.oid, + expected_kind: ObjectKind::Blob + }, + TypedEdge { + child: edges[0].child, + expected_kind: ObjectKind::Tree + } + ] + ), + Err(MetadataError::IdentityConflict) + )); + let segment = writer.seal(&fixture.index)?; + assert!(segment.header(extra.oid)?.is_none()); + assert_eq!(segment.edges_after(tree.oid, None)?, *edges); + drop(segment); + assert_eq!(budget.used(), 0); + Ok(()) +} + +#[tokio::test] +async fn exact_native_ordinal_shards_cover_a_pack_without_rescanning_prefixes() -> Result { + let fixture = fixture(ObjectFormat::Sha256, 9).await?; + let budget = DiskBudget::new(128 << 20); + let split = fixture.index.len() / 2; + let objects = fixture.objects.values().cloned().collect::>(); + let mut descriptors = Vec::new(); + for (first, count) in [(0, split), (split, fixture.index.len() - split)] { + let identity = SegmentIdentity { + first_ordinal: first, + object_count: count, + ..fixture.identity + }; + let mut writer = builder(&fixture, budget.clone(), identity)?; + fill( + &mut writer, + &objects[first as usize..(first + count) as usize], + )?; + let segment = writer.seal(&fixture.index)?; + descriptors.push(segment.descriptor()); + assert_eq!(segment.headers_after(None)?.len(), count as usize); + } + assert!(descriptors[0].last_oid < descriptors[1].first_oid); + assert_eq!( + descriptors[0].identity.object_count + descriptors[1].identity.object_count, + fixture.index.len() + ); + assert_eq!(budget.used(), 0); + Ok(()) +} + +#[tokio::test] +async fn wrong_index_inventory_and_missing_structural_edges_cannot_seal() -> Result { + let fixture = fixture(ObjectFormat::Sha1, 2).await?; + let budget = DiskBudget::new(128 << 20); + let mut writer = builder(&fixture, budget.clone(), fixture.identity)?; + let objects = fixture + .objects + .values() + .map(|(object, _)| *object) + .collect::>(); + writer.put_objects(&objects)?; + assert!(matches!( + writer.seal(&fixture.index), + Err(MetadataError::Integrity) + )); + let identity = SegmentIdentity { + first_ordinal: 1, + ..fixture.identity + }; + let mut writer = builder(&fixture, budget.clone(), identity)?; + fill( + &mut writer, + &fixture.objects.values().cloned().collect::>(), + )?; + assert!(matches!( + writer.seal(&fixture.index), + Err(MetadataError::Integrity) + )); + assert_eq!(budget.used(), 0); + Ok(()) +} + +#[tokio::test] +async fn stored_child_kind_must_match_the_typed_dependency() -> Result { + let fixture = fixture(ObjectFormat::Sha1, 2).await?; + let budget = DiskBudget::new(128 << 20); + let mut writer = builder(&fixture, budget.clone(), fixture.identity)?; + let mut objects = fixture.objects.values().cloned().collect::>(); + let tree = objects + .iter_mut() + .find(|(object, _)| object.kind == ObjectKind::Tree) + .ok_or("tree")?; + tree.1[0].expected_kind = ObjectKind::Tree; + fill(&mut writer, &objects)?; + assert!(matches!( + writer.seal(&fixture.index), + Err(MetadataError::Integrity) + )); + assert_eq!(budget.used(), 0); + Ok(()) +} + +#[tokio::test] +async fn artifact_tampering_and_wrong_repository_are_rejected_before_reading_rows() -> Result { + let fixture = fixture(ObjectFormat::Sha1, 2).await?; + let budget = DiskBudget::new(128 << 20); + let mut writer = builder(&fixture, budget.clone(), fixture.identity)?; + fill( + &mut writer, + &fixture.objects.values().cloned().collect::>(), + )?; + let segment = writer.seal(&fixture.index)?; + for mode in ["bytes", "identity", "size"] { + let file = tempfile::NamedTempFile::new_in(fixture.root.path())?; + std::fs::copy(segment.path(), file.path())?; + let mut descriptor = segment.descriptor(); + match mode { + "bytes" => { + use std::io::{Seek, SeekFrom, Write}; + let mut handle = file.reopen()?; + handle.seek(SeekFrom::Start(100))?; + handle.write_all(&[255])?; + } + "identity" => descriptor.identity.repository = [39; 16], + "size" => descriptor.size += 1, + _ => unreachable!(), + } + let reservation = budget.try_reserve(descriptor.size)?; + assert!(matches!( + MetadataSegment::open(file, reservation, descriptor, 64), + Err(MetadataError::Integrity) + )); + } + drop(segment); + assert_eq!(budget.used(), 0); + Ok(()) +} + +#[tokio::test] +async fn large_inventory_and_wide_tree_use_bounded_pages_and_disk_admission() -> Result { + let fixture = fixture(ObjectFormat::Sha1, 1600).await?; + let budget = DiskBudget::new(128 << 20); + let mut writer = builder(&fixture, budget.clone(), fixture.identity)?; + fill( + &mut writer, + &fixture.objects.values().cloned().collect::>(), + )?; + let segment = writer.seal(&fixture.index)?; + assert_eq!(budget.used(), segment.descriptor().size); + let mut after = None; + let mut count = 0; + loop { + let page = segment.headers_after(after)?; + if page.is_empty() { + break; + } + assert!(page.len() <= PAGE_OBJECTS); + assert!( + page.windows(2) + .all(|pair| pair[0].object.oid < pair[1].object.oid) + ); + count += page.len(); + after = page.last().map(|header| header.object.oid); + } + assert_eq!(count, fixture.objects.len()); + let tree = fixture + .objects + .values() + .find(|(object, _)| object.kind == ObjectKind::Tree) + .ok_or("tree")? + .0; + let mut after = None; + let mut edges = 0; + loop { + let page = segment.edges_after(tree.oid, after)?; + if page.is_empty() { + break; + } + assert!(page.len() <= PAGE_OBJECTS); + edges += page.len(); + after = page.last().map(|edge| edge.child); + } + assert_eq!(edges, 1600); + drop(segment); + assert_eq!(budget.used(), 0); + assert!(matches!( + MetadataBuilder::new( + fixture.root.path(), + DiskBudget::new(growth::INITIAL_BYTES * 3 - 1), + fixture.identity, + limits() + ), + Err(MetadataError::Budget(_)) + )); + Ok(()) +} diff --git a/crates/canopy-server/src/packs/metadata/transport.rs b/crates/canopy-server/src/packs/metadata/transport.rs new file mode 100644 index 0000000..763dac0 --- /dev/null +++ b/crates/canopy-server/src/packs/metadata/transport.rs @@ -0,0 +1,323 @@ +use super::*; +use bytes::Bytes; +use canopy_object_storage::artifact::{ + ArtifactDescriptor, ArtifactKey, ArtifactKind, ArtifactStore, +}; +use futures_core::Stream; +use std::{ + future::Future, + io::{Seek, SeekFrom, Write}, + pin::Pin, + sync::Arc, + task::{Context, Poll}, +}; +use tokio_util::io::StreamReader; + +/// A certified catalog must bind both the canonical inventory and exact stored +/// bytes. The manifest digest authenticates every part before it reaches disk. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct StoredSegment { + pub segment: SegmentDescriptor, + pub artifact: ArtifactDescriptor, +} +impl StoredSegment { + fn validate(self, store: &ArtifactStore) -> Result<(), MetadataError> { + validate_identity(self.segment.identity)?; + if self.segment.identity.repository != store.repository() + || self.segment.size != self.artifact.size + || self.segment.digest != self.artifact.digest + { + return Err(MetadataError::Integrity); + } + Ok(()) + } + fn key(self) -> ArtifactKey { + key(self.segment.identity) + } +} +fn key(identity: SegmentIdentity) -> ArtifactKey { + ArtifactKey { + operation: identity.operation, + binding_digest: identity.pack_digest, + kind: ArtifactKind::Metadata, + } +} + +impl MetadataSegment { + /// Upload owns a segment pin. Each background file read also owns the pin, + /// so cancellation cannot release disk admission while a read is queued. + pub async fn upload( + self: Arc, + store: &ArtifactStore, + ) -> Result { + let segment = self.descriptor(); + if segment.identity.repository != store.repository() { + return Err(MetadataError::Integrity); + } + let artifact = upload_file( + self, + store, + key(segment.identity), + segment.size, + segment.digest, + ) + .await?; + Ok(StoredSegment { segment, artifact }) + } + + /// Admission precedes file creation and provider reads. Blocking writes, + /// sync, hashing and SQLite open retain the spool admission through cancel. + /// A canceled/failed download never returns a usable metadata connection. + pub async fn download( + root: &Path, + budget: DiskBudget, + store: &ArtifactStore, + stored: StoredSegment, + limits: MetadataLimits, + ) -> Result, MetadataError> { + Self::download_for_reader(root, budget, store, stored, limits, None).await + } + + pub(in crate::packs) async fn download_for_reader( + root: &Path, + budget: DiskBudget, + store: &ArtifactStore, + stored: StoredSegment, + limits: MetadataLimits, + reader: Option, + ) -> Result, MetadataError> { + stored.validate(store)?; + let admitted = download_file_for_reader( + root, + budget, + store, + stored.key(), + stored.artifact, + limits, + reader, + ) + .await?; + tokio::task::spawn_blocking(move || { + Ok(Arc::new(Self::open_admitted( + admitted, + stored.segment, + limits.cache_kib, + )?)) + }) + .await? + } +} + +pub(crate) trait PinnedFile: Send + Sync + 'static { + fn open(&self) -> io::Result; +} +impl PinnedFile for MetadataSegment { + fn open(&self) -> io::Result { + File::open(MetadataSegment::path(self)) + } +} + +pub(crate) async fn upload_file( + owner: Arc, + store: &ArtifactStore, + key: ArtifactKey, + size: u64, + digest: [u8; 32], +) -> Result { + let source = tokio::task::spawn_blocking(move || { + let file = owner.open()?; + if file.metadata()?.len() != size { + return Err(MetadataError::Integrity); + } + Ok::<_, MetadataError>(Arc::new(ReadPin { + file: Mutex::new(file), + _owner: owner, + size, + })) + }) + .await??; + let mut input = StreamReader::new(SegmentStream { + source, + offset: 0, + job: None, + failed: false, + }); + Ok(store.put(key, size, digest, &mut input).await?) +} + +pub(in crate::packs) async fn download_file_for_reader( + root: &Path, + budget: DiskBudget, + store: &ArtifactStore, + key: ArtifactKey, + artifact: ArtifactDescriptor, + limits: MetadataLimits, + reader: Option, +) -> Result { + if limits.cache_kib == 0 + || limits.cache_kib > i32::MAX as u32 + || artifact.size > limits.max_file_bytes + || limits.max_file_bytes > canopy_object_storage::external::MAX_ARTIFACT_BYTES + { + return Err(MetadataError::Limit); + } + let reservation = budget.try_reserve(artifact.size)?; + let root = root.to_owned(); + let spool = tokio::task::spawn_blocking(move || { + Ok::<_, MetadataError>(Arc::new(DownloadSpool { + admitted: Mutex::new( + AdmittedFile::new( + tempfile::Builder::new() + .prefix("canopy-metadata-download-") + .tempfile_in(root)?, + reservation, + ) + .with_reader(reader), + ), + })) + }) + .await??; + let mut reader = store.read(key, artifact).await?; + while let Some(bytes) = reader.next().await? { + let spool = Arc::clone(&spool); + tokio::task::spawn_blocking(move || { + let mut admitted = spool + .admitted + .lock() + .map_err(|_| MetadataError::Integrity)?; + admitted.file_mut().write_all(&bytes)?; + Ok::<_, MetadataError>(()) + }) + .await??; + } + tokio::task::spawn_blocking(move || { + let spool = Arc::try_unwrap(spool).map_err(|_| MetadataError::Integrity)?; + let admitted = spool + .admitted + .into_inner() + .map_err(|_| MetadataError::Integrity)?; + admitted.file().as_file().sync_all()?; + Ok(admitted) + }) + .await? +} + +// Field order closes/unlinks the private file before releasing admission. +struct DownloadSpool { + admitted: Mutex, +} + +// Unlike a bare tokio::fs::File, each pending blocking task owns its admission +// pin. Each stream serializes its reads; its owner defines how handles open. +struct ReadPin { + file: Mutex, + _owner: Arc, + size: u64, +} +struct SegmentStream { + source: Arc>, + offset: u64, + job: Option>>, + failed: bool, +} +impl Stream for SegmentStream { + type Item = io::Result; + fn poll_next(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { + if self.failed || self.offset == self.source.size { + return Poll::Ready(None); + } + if self.job.is_none() { + let source = Arc::clone(&self.source); + let offset = self.offset; + let length = (source.size - offset).min(64 << 10) as usize; + self.job = Some(tokio::task::spawn_blocking(move || { + let mut file = source + .file + .lock() + .map_err(|_| io::Error::other("metadata read lock poisoned"))?; + file.seek(SeekFrom::Start(offset))?; + let mut buffer = vec![0; length]; + file.read_exact(&mut buffer)?; + Ok(Bytes::from(buffer)) + })); + } + let result = match Pin::new(self.job.as_mut().expect("scheduled read")).poll(cx) { + Poll::Pending => return Poll::Pending, + Poll::Ready(result) => result, + }; + self.job = None; + match result.unwrap_or_else(|error| Err(io::Error::other(error))) { + Ok(bytes) => { + self.offset += bytes.len() as u64; + Poll::Ready(Some(Ok(bytes))) + } + Err(error) => { + self.failed = true; + Poll::Ready(Some(Err(error))) + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn canceled_queued_read_keeps_file_and_admission_until_the_worker_finishes() + -> Result<(), Box> { + let runtime = tokio::runtime::Builder::new_multi_thread() + .worker_threads(1) + .max_blocking_threads(1) + .enable_all() + .build()?; + let (_root, segment, budget) = runtime.block_on(super::super::tests::prepared_segment())?; + let path = segment.path().to_owned(); + let weak = Arc::downgrade(&segment); + let charged = budget.used(); + assert!(charged > 0); + let (ready_tx, ready_rx) = std::sync::mpsc::channel(); + let (release_tx, release_rx) = std::sync::mpsc::channel(); + let blocker = runtime.spawn_blocking(move || { + ready_tx.send(()).unwrap(); + release_rx.recv().unwrap(); + }); + ready_rx.recv_timeout(std::time::Duration::from_secs(5))?; + let mut stream = SegmentStream { + source: Arc::new(ReadPin { + file: Mutex::new(File::open(&path)?), + size: segment.descriptor().size, + _owner: segment, + }), + offset: 0, + job: None, + failed: false, + }; + { + let _entered = runtime.enter(); + let mut context = Context::from_waker(std::task::Waker::noop()); + assert!(Pin::new(&mut stream).poll_next(&mut context).is_pending()); + } + assert!(stream.job.is_some()); + drop(stream); + assert_eq!(budget.used(), charged); + assert!(weak.upgrade().is_some()); + assert!(path.exists()); + // Always release before assertions that could unwind: a stopped + // blocking worker must not hang runtime shutdown if this test fails. + release_tx.send(())?; + runtime.block_on(async { + blocker.await?; + tokio::time::timeout(std::time::Duration::from_secs(5), async { + while budget.used() != 0 { + tokio::time::sleep(std::time::Duration::from_millis(5)).await; + } + }) + .await?; + Ok::<_, Box>(()) + })?; + assert!(weak.upgrade().is_none()); + assert!(!path.exists()); + Ok(()) + } +} diff --git a/crates/canopy-server/src/packs/metadata/writer.rs b/crates/canopy-server/src/packs/metadata/writer.rs new file mode 100644 index 0000000..561b11c --- /dev/null +++ b/crates/canopy-server/src/packs/metadata/writer.rs @@ -0,0 +1,441 @@ +use super::*; + +/// Disk-backed verification spool. Header/edge batches are atomic and bounded. +/// Supplied canonical headers must come from the trusted streaming verifier. +pub struct MetadataBuilder { + // Close SQLite (and its journal) before deleting the spool or releasing + // admission on a failed or canceled builder. + connection: Connection, + admitted: AdmittedFile, + identity: SegmentIdentity, + limits: MetadataLimits, + failed: bool, +} +impl MetadataBuilder { + #[cfg(test)] + pub(in crate::packs) fn admitted_bytes(&mut self) -> u64 { + self.admitted.reservation().bytes() + } + pub fn new( + root: &Path, + budget: DiskBudget, + identity: SegmentIdentity, + limits: MetadataLimits, + ) -> Result { + validate_identity(identity)?; + if limits.cache_kib == 0 + || limits.cache_kib > i32::MAX as u32 + || limits.max_file_bytes < 16 << 10 + || limits.max_file_bytes > canopy_object_storage::external::MAX_ARTIFACT_BYTES + || !limits.max_file_bytes.is_multiple_of(4096) + { + return Err(MetadataError::Limit); + } + let reservation = growth::reserve(&budget, limits.max_file_bytes)?; + let file = tempfile::Builder::new() + .prefix("canopy-metadata-") + .tempfile_in(root)?; + let mut admitted = AdmittedFile::new(file, reservation); + let mut connection = Connection::open(admitted.file().path())?; + connection.execute_batch("PRAGMA page_size=4096; PRAGMA journal_mode=DELETE; PRAGMA synchronous=FULL; PRAGMA foreign_keys=ON; PRAGMA trusted_schema=OFF; PRAGMA mmap_size=0;")?; + connection.pragma_update(None, "cache_size", -(limits.cache_kib as i64))?; + growth::configure(&connection, &mut admitted)?; + growth::transaction( + &mut connection, + &mut admitted, + limits.max_file_bytes, + |tx| { + tx.execute_batch(SCHEMA)?; + Ok::<_, MetadataError>(()) + }, + )?; + Ok(Self { + admitted, + connection, + identity, + limits, + failed: false, + }) + } + + pub fn put_objects(&mut self, objects: &[CanonicalObject]) -> Result<(), MetadataError> { + self.check_healthy()?; + if objects.is_empty() || objects.len() > PAGE_OBJECTS { + return Err(MetadataError::Limit); + } + if objects.iter().any(|object| { + object.oid.format() != self.identity.format + || object.oid.is_zero() + || object.size > i64::MAX as u64 + }) { + return Err(MetadataError::Integrity); + } + growth::transaction( + &mut self.connection, + &mut self.admitted, + self.limits.max_file_bytes, + |transaction| { + { + let mut insert = transaction.prepare_cached("INSERT INTO objects(oid, kind, size, digest) VALUES (?1,?2,?3,?4) ON CONFLICT DO NOTHING")?; + let mut existing = transaction.prepare_cached( + "SELECT oid, kind, size, digest FROM objects WHERE oid = ?1", + )?; + for object in objects { + insert.execute(params![ + object.oid.as_ref(), + object.kind.git_name(), + object.size as i64, + object.digest.as_slice() + ])?; + if existing.query_row([object.oid.as_ref()], canonical)? != *object { + return Err(MetadataError::IdentityConflict); + } + } + } + Ok(()) + }, + ) + } + + pub fn put_edges( + &mut self, + parent: ObjectId, + edges: &[TypedEdge], + ) -> Result<(), MetadataError> { + self.check_healthy()?; + if edges.is_empty() || edges.len() > PAGE_OBJECTS { + return Err(MetadataError::Limit); + } + if parent.format() != self.identity.format + || parent.is_zero() + || edges + .iter() + .any(|edge| edge.child.format() != self.identity.format || edge.child.is_zero()) + { + return Err(MetadataError::Integrity); + } + growth::transaction( + &mut self.connection, + &mut self.admitted, + self.limits.max_file_bytes, + |transaction| { + let parent_kind = transaction + .query_row( + "SELECT kind FROM objects WHERE oid = ?1", + [parent.as_ref()], + |row| kind(&row.get::<_, String>(0)?), + ) + .optional()? + .ok_or(MetadataError::Integrity)?; + if parent_kind == ObjectKind::Blob + || edges.iter().any(|edge| match parent_kind { + ObjectKind::Tree => { + !matches!(edge.expected_kind, ObjectKind::Blob | ObjectKind::Tree) + } + ObjectKind::Commit => { + !matches!(edge.expected_kind, ObjectKind::Commit | ObjectKind::Tree) + } + _ => false, + }) + { + return Err(MetadataError::Integrity); + } + { + let mut insert = transaction.prepare_cached("INSERT INTO object_edges(parent,child,expected_kind) VALUES (?1,?2,?3) ON CONFLICT DO NOTHING")?; + let mut existing = transaction.prepare_cached( + "SELECT expected_kind FROM object_edges WHERE parent = ?1 AND child = ?2", + )?; + for edge in edges { + insert.execute(params![ + parent.as_ref(), + edge.child.as_ref(), + edge.expected_kind.git_name() + ])?; + if existing + .query_row(params![parent.as_ref(), edge.child.as_ref()], |row| { + kind(&row.get::<_, String>(0)?) + })? + != edge.expected_kind + { + return Err(MetadataError::IdentityConflict); + } + } + } + Ok(()) + }, + ) + } + + fn check_healthy(&self) -> Result<(), MetadataError> { + if self.failed { + Err(MetadataError::Integrity) + } else { + Ok(()) + } + } + + /// Atomically import one complete decoded witness. No header or dependency + /// survives a failed replay, and failure permanently prevents sealing. + /// Physical pack binding and global typed closure are separate requirements. + pub fn put_verified( + &mut self, + witness: super::super::verification::VerifiedObject, + ) -> Result<(), MetadataError> { + self.put_verified_batch(vec![witness]) + } + + /// Bound retained witnesses and amortize SQLite durability over up to one + /// page. Any failure rolls back the whole batch and poisons this builder. + /// Run on a blocking worker that owns the admitted builder and witnesses + /// until it finishes, even when the requesting async future is canceled. + pub fn put_verified_batch( + &mut self, + mut witnesses: Vec, + ) -> Result<(), MetadataError> { + self.check_healthy()?; + self.failed = true; + if witnesses.is_empty() || witnesses.len() > PAGE_OBJECTS { + return Err(MetadataError::Limit); + } + growth::transaction( + &mut self.connection, + &mut self.admitted, + self.limits.max_file_bytes, + |transaction| { + for witness in &mut witnesses { + let object = witness.object(); + if object.oid.format() != self.identity.format + || object.oid.is_zero() + || object.size > i64::MAX as u64 + { + return Err(MetadataError::Integrity); + } + transaction.execute("INSERT INTO objects(oid, kind, size, digest) VALUES (?1,?2,?3,?4) ON CONFLICT DO NOTHING", params![object.oid.as_ref(),object.kind.git_name(),object.size as i64,object.digest.as_slice()])?; + let existing = transaction.query_row( + "SELECT oid, kind, size, digest FROM objects WHERE oid = ?1", + [object.oid.as_ref()], + canonical, + )?; + if existing != object { + return Err(MetadataError::IdentityConflict); + } + { + let mut insert = transaction.prepare_cached("INSERT INTO object_edges(parent,child,expected_kind) VALUES (?1,?2,?3) ON CONFLICT DO NOTHING")?; + let mut existing = transaction.prepare_cached( + "SELECT expected_kind FROM object_edges WHERE parent = ?1 AND child = ?2", + )?; + witness.replay(|edges| { + for edge in edges { + if edge.child.format() != self.identity.format + || edge.child.is_zero() + || match object.kind { + ObjectKind::Blob => true, + ObjectKind::Tree => !matches!( + edge.expected_kind, + ObjectKind::Tree | ObjectKind::Blob + ), + ObjectKind::Commit => !matches!( + edge.expected_kind, + ObjectKind::Tree | ObjectKind::Commit + ), + ObjectKind::Tag => false, + } + { + return Err(MetadataError::Integrity); + } + insert.execute(params![ + object.oid.as_ref(), + edge.child.as_ref(), + edge.expected_kind.git_name() + ])?; + if existing.query_row( + params![object.oid.as_ref(), edge.child.as_ref()], + |row| kind(&row.get::<_, String>(0)?), + )? != edge.expected_kind + { + return Err(MetadataError::IdentityConflict); + } + } + Ok(()) + })?; + } + } + Ok(()) + }, + )?; + self.failed = false; + Ok(()) + } + + pub fn seal(mut self, index: &PackIndex) -> Result { + self.check_healthy()?; + let end = self + .identity + .first_ordinal + .checked_add(self.identity.object_count) + .ok_or(MetadataError::Integrity)?; + if index.format() != self.identity.format + || index.pack_checksum() != self.identity.git_checksum + || end > index.len() + { + return Err(MetadataError::Integrity); + } + let count: u64 = self + .connection + .query_row("SELECT count(*) FROM objects", [], |row| row.get(0))?; + if count != u64::from(self.identity.object_count) { + return Err(MetadataError::Integrity); + } + let wrong_child: bool = self.connection.query_row("SELECT EXISTS(SELECT 1 FROM object_edges e JOIN objects o ON o.oid = e.child WHERE o.kind != e.expected_kind)", [], |row| row.get(0))?; + if wrong_child { + return Err(MetadataError::Integrity); + } + let mut inventory = inventory_seed(self.identity); + let mut native = index.ids_from(self.identity.first_ordinal)?; + let mut after = Vec::new(); + let mut ordinal = 0; + let mut total_edges = 0_u64; + let mut first_oid = None; + let mut last_oid = None; + loop { + let objects = { + let mut statement = self.connection.prepare_cached("SELECT oid, kind, size, digest FROM objects WHERE oid > ?1 ORDER BY oid LIMIT ?2")?; + statement + .query_map(params![after, PAGE_OBJECTS as i64], canonical)? + .collect::>>()? + }; + if objects.is_empty() { + break; + } + // Consume native ordinals once, outside the replayable SQL body. + for object in &objects { + if native.next().transpose()? != Some(object.oid) { + return Err(MetadataError::Integrity); + } + } + let (next_inventory, next_ordinal, next_total, next_first, next_last, next_after) = + growth::transaction( + &mut self.connection, + &mut self.admitted, + self.limits.max_file_bytes, + |transaction| { + let mut inventory = inventory; + let mut ordinal = ordinal; + let mut total_edges = total_edges; + let mut first_oid = first_oid; + let mut last_oid = last_oid; + let mut after = after.clone(); + for &object in &objects { + let mut chain = edge_seed(object.oid); + let mut edge_count = 0_u64; + let mut tree_count = 0_u64; + { + let mut statement = transaction.prepare_cached("SELECT child, expected_kind FROM object_edges WHERE parent = ?1 ORDER BY child")?; + let mut rows = statement.query([object.oid.as_ref()])?; + while let Some(row) = rows.next()? { + let child = oid(row.get(0)?)?; + let expected_kind = kind(&row.get::<_, String>(1)?)?; + let mut record = [0; 33]; + let width = child.len(); + record[..width].copy_from_slice(&child); + record[width] = kind_code(expected_kind); + chain = fold(chain, edge_count, &record[..width + 1]); + edge_count = + edge_count.checked_add(1).ok_or(MetadataError::Limit)?; + tree_count += u64::from(expected_kind == ObjectKind::Tree); + } + } + if (object.kind == ObjectKind::Blob && edge_count != 0) + || (object.kind == ObjectKind::Tag && edge_count != 1) + || (object.kind == ObjectKind::Commit && tree_count != 1) + { + return Err(MetadataError::Integrity); + } + let edge_count_sql = + i64::try_from(edge_count).map_err(|_| MetadataError::Limit)?; + transaction.execute( + "UPDATE objects SET edge_count = ?2, edge_digest = ?3 WHERE oid = ?1", + params![object.oid.as_ref(), edge_count_sql, chain.as_slice()], + )?; + inventory = fold_header( + inventory, + ordinal, + ObjectHeader { + object, + edge_count, + edge_digest: chain, + }, + ); + ordinal += 1; + total_edges = total_edges + .checked_add(edge_count) + .ok_or(MetadataError::Limit)?; + first_oid.get_or_insert(object.oid); + last_oid = Some(object.oid); + after = object.oid.to_vec(); + } + Ok::<_, MetadataError>(( + inventory, + ordinal, + total_edges, + first_oid, + last_oid, + after, + )) + }, + )?; + inventory = next_inventory; + ordinal = next_ordinal; + total_edges = next_total; + first_oid = next_first; + last_oid = next_last; + after = next_after; + } + if ordinal != count { + return Err(MetadataError::Integrity); + } + let identity = self.identity; + growth::transaction( + &mut self.connection, + &mut self.admitted, + self.limits.max_file_bytes, + |tx| { + tx.execute( + "INSERT INTO segment_identity VALUES(1,?1,?2,?3,?4,?5,?6,?7,?8,?9)", + params![ + identity.repository.as_slice(), + identity.operation.as_slice(), + identity.format.as_str(), + identity.pack_digest.as_slice(), + identity.git_checksum.as_ref(), + identity.first_ordinal, + identity.object_count, + i64::try_from(total_edges).map_err(|_| MetadataError::Limit)?, + inventory.as_slice() + ], + )?; + Ok::<_, MetadataError>(()) + }, + )?; + self.connection + .close() + .map_err(|(_, error)| MetadataError::Sql(error))?; + self.admitted.clean_journal()?; + self.admitted.file().as_file().sync_all()?; + let size = self.admitted.file().as_file().metadata()?.len(); + if size > self.limits.max_file_bytes { + return Err(MetadataError::Limit); + } + let descriptor = SegmentDescriptor { + identity, + edge_count: total_edges, + inventory_digest: inventory, + first_oid: first_oid.ok_or(MetadataError::Integrity)?, + last_oid: last_oid.ok_or(MetadataError::Integrity)?, + size, + digest: file_digest(self.admitted.file().path(), size)?, + }; + self.admitted.reservation().resize(size)?; + MetadataSegment::open_admitted(self.admitted, descriptor, self.limits.cache_kib) + } +} diff --git a/crates/canopy-server/src/packs/mod.rs b/crates/canopy-server/src/packs/mod.rs new file mode 100644 index 0000000..41463d2 --- /dev/null +++ b/crates/canopy-server/src/packs/mod.rs @@ -0,0 +1,17 @@ +//! Immutable native-pack metadata, outside the Repository Cell's write path. +//! +//! These structures are inputs to trusted catalog verification. Native pack +//! membership alone does not certify graph closure or authorize object reads. + +pub mod catalog; +pub mod closure; +pub mod directory; +pub mod metadata; +pub mod sources; +pub mod verification; + +pub(crate) mod input_artifact; +pub use input_artifact::InputRootError; +pub mod publication; +pub mod ref_state; +pub mod wire_request; diff --git a/crates/canopy-server/src/packs/publication/attestation.rs b/crates/canopy-server/src/packs/publication/attestation.rs new file mode 100644 index 0000000..00e5a71 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/attestation.rs @@ -0,0 +1,255 @@ +//! Trusted service issuance and bounded durable registration. No ref or current +//! catalog is changed here; the later final publisher must authenticate this +//! certificate, whether carried inline or recovered from a checkpoint. +use super::*; +use super::{ + certificate::CertificateData, + commands::{authorized, check_pin, fact, load, matched}, + sql::*, +}; +use cellule_runtime::{Committed, InvocationError, MutationIdentity, primitives::sql::SqlCell}; +use tokio::time::timeout_at; + +#[derive(Debug, thiserror::Error)] +pub enum CatalogAttestationError { + #[error("catalog attestation base is inactive")] + Base(#[from] PreparationBaseError), + #[error("catalog input custody failed")] + Inputs(#[from] InputCheckpointError), + #[error("catalog attestation SQL capability failed")] + Capability(#[from] Error), + #[error("catalog attestation issuer query failed")] + Query(#[source] Box>>), + #[error("catalog attestation registration failed")] + Command(#[source] Box>), + #[error("catalog attestation encoding failed")] + Codec(#[from] CodecError), +} +impl PreparedCatalog { + /// Uses the already trusted application SQL capability, as signed-push + /// nonce issuance does. Never expose the repository seed or raw SQL surface + /// to product clients. Raw descriptors cannot call this factory. + pub async fn certificate(&self) -> Result { + let (_, deadline) = self.base.live_lease()?; + timeout_at(deadline, self.issue_certificate(None, None)) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } + /// Optional durable checkpoint. The final publisher may instead carry the + /// certificate inline and persist its facts with refs in the same command. + pub async fn attest( + &self, + identity: MutationIdentity, + ) -> Result, CatalogAttestationError> { + let (_, deadline) = self.base.live_lease()?; + timeout_at(deadline, self.attest_inner(identity)) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } + async fn attest_inner( + &self, + identity: MutationIdentity, + ) -> Result, CatalogAttestationError> { + let certificate = self.issue_certificate(None, None).await?; + self.ensure_live()?; + let (client, target, _) = self.base.capability(); + client + .command::(target, identity, certificate) + .await + .map_err(|error| CatalogAttestationError::Command(Box::new(error))) + } + pub(super) async fn issue_certificate( + &self, + refs_digest: Option<[u8; 32]>, + completion_digest: Option<[u8; 32]>, + ) -> Result { + let mut data = CertificateData::from_prepared(self); + data.refs_digest = refs_digest; + data.completion_digest = completion_digest; + issue_data_certificate(&self.base, data).await + } +} +/// Shared trusted issuer. Only private verified preparation factories construct +/// these facts; decoded descriptors cannot invoke it from a product surface. +pub(super) async fn issue_data_certificate( + base: &PreparationBaseResolver, + data: CertificateData, +) -> Result { + let (client, target, check) = base.capability(); + let live = client + .query::(target, None, check.clone()) + .await + .map_err(|error| PreparationBaseError::Query(Box::new(error)))? + .output + .ok_or(PreparationBaseError::Inactive)?; + if live.token != base.context_token() + || live.base != base.retention_floor() + || live.format != data.catalog.format + { + return Err(PreparationBaseError::Context.into()); + } + if let Some(digest) = data.input_checkpoint_digest { + inputs::verify_digest(base, digest).await?; + } + let sql = SqlCell::::new(client.clone(), target.clone())?; + let mut statements = vec![ + SqlStatement { sql: "SELECT push_cert_seed FROM repository_identity WHERE singleton=1 AND repository_id=?1 AND object_format=?2".into(), parameters:vec![blob(live.token.repository),SqlValue::Text(live.format.as_str().into())] }, + SqlStatement { sql: GENERATION.into(), parameters:vec![number(base.generation_fact().generation)?] }, + ]; + if data.compaction { + statements.push(access_statement(&check.actor)); + } + let result = sql + .query(None, SqlBatch { statements }) + .await + .map_err(|error| CatalogAttestationError::Query(Box::new(error)))?; + if data.compaction + && !decode_access( + result + .output + .get(2..) + .ok_or(Error::Command("missing compaction issuer authority"))?, + )? + .is_some_and(|role| role >= TokenScope::Admin) + { + return Err(PreparationBaseError::Inactive.into()); + } + let seed = seed(&result.output)?; + if generation( + result + .output + .get(1..) + .ok_or(Error::Command("missing selected generation"))?, + live.token.repository, + live.format, + )? != base.generation_fact() + { + return Err(PreparationBaseError::Context.into()); + } + let certificate = CatalogCertificate::seal(&data, &seed)?; + base.live_lease()?; + Ok(certificate) +} + +pub(super) fn seed(sets: &[SqlResultSet]) -> cellule_runtime::Result<[u8; 32]> { + let Some([value]) = rows(sets)?.first().map(Vec::as_slice) else { + return Err(Error::Command("catalog issuer secret is absent")); + }; + fixed(value) +} +fn deny(reason: PreparationDenial) -> CommandResult { + CommandResult::Rejected(AttestationOutcome::Denied(reason)) +} +pub struct RegisterCatalogAttestation; +impl Command for RegisterCatalogAttestation { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 17; + const CODEC_VERSION: u32 = 1; + type Input = CatalogCertificate; + type Output = AttestationOutcome; + fn execute( + context: &mut CommandContext<'_, '_>, + certificate: CatalogCertificate, + ) -> cellule_runtime::Result> { + let data = certificate.data()?; + if data.tenant != *context.target().tenant().as_bytes() + || data.application != *context.target().application().as_bytes() + || data.token.owner != context.owner_fence() + { + return Ok(deny(PreparationDenial::Stale)); + } + let Some(format) = authorized( + context, + data.token.repository, + &data.actor, + if data.compaction { + TokenScope::Admin + } else { + TokenScope::Write + }, + )? + else { + return Ok(deny(PreparationDenial::Unauthorized)); + }; + let Some(row) = load(context, data.token)? else { + return Ok(deny(PreparationDenial::Missing)); + }; + if !matched( + &row, + &LeaseCheck { + token: data.token, + actor: data.actor.clone(), + }, + ) { + return Ok(deny(PreparationDenial::Stale)); + } + if row.expires <= now(context.now_ms())? { + return Ok(deny(PreparationDenial::Expired)); + } + check_pin(context, &row)?; + if format != data.catalog.format + || !super::publish::retention_matches(context, &data, row.generation, format)? + || fact( + context, + data.token.repository, + format, + Some(data.base.generation), + )? != data.base + { + return Ok(deny(PreparationDenial::Conflict)); + } + let key = seed(&context.sql(&statement( + "SELECT push_cert_seed FROM repository_identity WHERE singleton=1", + vec![], + ))?)?; + if !certificate.authenticated(&key) { + return Ok(deny(PreparationDenial::Unauthorized)); + } + let bytes = certificate.bytes()?; + let digest = *blake3::hash(&bytes).as_bytes(); + let previous = context.sql(&statement( + "SELECT attestation,attestation_digest FROM catalog_operations WHERE id=?1", + vec![blob(data.token.operation)], + ))?; + let pin_previous = context.sql(&statement( + "SELECT attestation,attestation_digest FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2", + vec![blob(data.token.owner.incarnation.as_bytes()),number(data.token.attempt)?], + ))?; + if rows(&previous)? != rows(&pin_previous)? { + return Err(Error::Command( + "catalog attestation differs from its retention pin", + )); + } + match rows(&previous)?.first().map(Vec::as_slice) { + Some([SqlValue::Null, SqlValue::Null]) => { + let result = context.sql(&statement("UPDATE catalog_operations SET attestation=?1,attestation_digest=?2 WHERE id=?3 AND attestation IS NULL",vec![blob(&bytes),blob(digest),blob(data.token.operation)]))?; + if result.first().is_none_or(|set| set.rows_affected != 1) { + return Err(Error::Command( + "catalog attestation registration changed no rows", + )); + } + let pinned = context.sql(&statement( + "UPDATE catalog_leases SET attestation=?1,attestation_digest=?2 WHERE incarnation=?3 AND admission_sequence=?4 AND attestation IS NULL", + vec![blob(&bytes),blob(digest),blob(data.token.owner.incarnation.as_bytes()),number(data.token.attempt)?], + ))?; + if pinned.first().is_none_or(|set| set.rows_affected != 1) { + return Err(Error::Command( + "catalog attestation retention changed no rows", + )); + } + } + Some([SqlValue::Blob(stored), stored_digest]) => { + if *stored != bytes || fixed::<32>(stored_digest)? != digest { + return Ok(deny(PreparationDenial::Conflict)); + } + } + _ => return Err(Error::Command("invalid stored catalog attestation")), + } + Ok(CommandResult::Success(AttestationOutcome::Registered( + RegisteredCatalog { + token: data.token, + certificate_digest: digest, + }, + ))) + } +} diff --git a/crates/canopy-server/src/packs/publication/base.rs b/crates/canopy-server/src/packs/publication/base.rs new file mode 100644 index 0000000..df527c3 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/base.rs @@ -0,0 +1,244 @@ +//! Authoritative preparation pin -> conditional certified base resolution. +//! The caller supplies the trusted application CellClient capability. Raw root +//! descriptors or decoded lease DTOs cannot construct this resolver. +use super::*; +use crate::packs::{ + catalog::{CatalogFiles, CatalogIndexes, CatalogReader}, + closure::{BaseBatch, BaseObject, BaseResolver, ClosureBase, ClosureContext, ClosureError}, + directory::index::IndexError, + metadata::PAGE_OBJECTS, +}; +use cellule_runtime::{CellClient, CellTarget, InvocationError, MutationIdentity, Receipt}; +use std::{sync::Arc, time::Duration}; +use tokio::time::{Instant, timeout_at}; + +#[derive(Debug, thiserror::Error)] +pub enum PreparationBaseError { + #[error("authoritative preparation query failed")] + Query(#[source] Box>>), + #[error("authoritative preparation frontier query failed")] + Frontier(#[source] Box>>), + #[error("preparation renewal failed")] + Command(#[source] Box>), + #[error("preparation catalog loading failed")] + Catalog(#[from] IndexError), + #[error("preparation has no active matching lease")] + Inactive, + #[error("preparation lease context is inconsistent")] + Context, +} +pub struct PreparationBaseResolver { + pub(super) session: PreparationSession, + selected: GenerationFact, + reader: Option>, + files: Arc, + indexes: Arc, +} +impl PreparationBaseResolver { + pub(super) fn capability(&self) -> (&CellClient, &CellTarget, &LeaseCheck) { + self.session.capability() + } + pub async fn open( + client: CellClient, + target: CellTarget, + check: LeaseCheck, + indexes: Arc, + files: Arc, + minimum: Option, + ) -> Result { + let session = PreparationSession::open(client, target, check, minimum).await?; + Self::from_session(session, indexes, files).await + } + pub(super) async fn from_session( + session: PreparationSession, + indexes: Arc, + files: Arc, + ) -> Result { + let (lease, deadline) = session.live_lease()?; + if indexes.store().repository() != lease.token.repository + || indexes.sources().format() != lease.format + { + return Err(PreparationBaseError::Context); + } + let reader = if let Some(catalog) = lease.base.catalog { + Some(Arc::new( + timeout_at(deadline, CatalogReader::open(Arc::clone(&indexes), catalog)) + .await + .map_err(|_| PreparationBaseError::Inactive)??, + )) + } else { + None + }; + session.live_lease()?; + Ok(Self { + session, + selected: lease.base, + reader, + files, + indexes, + }) + } + pub(super) fn live_lease(&self) -> Result<(PreparationLease, Instant), PreparationBaseError> { + self.session.live_lease() + } + pub(super) fn indexes(&self) -> Arc { + Arc::clone(&self.indexes) + } + pub(super) fn files(&self) -> Arc { + Arc::clone(&self.files) + } + pub(super) fn catalog_parts( + &self, + ) -> ( + super::super::directory::snapshot::DirectorySnapshot, + Option, + ) { + match &self.reader { + Some(reader) => (reader.directory(), reader.source_root()), + None => ( + super::super::directory::snapshot::DirectorySnapshot::empty( + self.session.lease.token.repository, + self.session.lease.format, + ), + None, + ), + } + } + pub fn context(&self) -> ClosureContext { + ClosureContext { + repository: self.session.lease.token.repository, + operation: self.session.lease.token.artifact_operation, + format: self.session.lease.format, + base: self.selected.catalog.map(|catalog| ClosureBase { + catalog, + generation: self.selected.generation, + }), + } + } + pub(super) fn context_token(&self) -> PreparationToken { + self.session.lease.token + } + pub(super) fn generation_fact(&self) -> GenerationFact { + self.selected + } + pub(super) fn retention_floor(&self) -> GenerationFact { + self.session.lease.base + } + /// Select only facts read through the exact active attempt. The original + /// floor, namespace, deadline and renewal fence are shared by all selections. + pub(super) async fn select_current(&self) -> Result { + let (_, deadline) = self.live_lease()?; + timeout_at(deadline, async { + let started = Instant::now(); + let frontier = self + .session + .client + .query::( + &self.session.target, + None, + self.session.check.clone(), + ) + .await + .map_err(|error| PreparationBaseError::Frontier(Box::new(error)))? + .output + .ok_or(PreparationBaseError::Inactive)?; + if frontier.lease.token != self.session.lease.token + || frontier.lease.base != self.session.lease.base + || frontier.lease.format != self.session.lease.format + || frontier.current.generation < self.selected.generation + { + return Err(PreparationBaseError::Context); + } + let remaining = (frontier.lease.expires_at_ms - frontier.lease.observed_at_ms) as u64; + let observed_deadline = started + .checked_add(Duration::from_millis(remaining.min(MAX_LEASE_MS))) + .ok_or(PreparationBaseError::Context)?; + let deadline = { + let mut shared = self + .session + .deadline + .lock() + .map_err(|_| PreparationBaseError::Context)?; + *shared = (*shared).min(observed_deadline); + deadline.min(*shared) + }; + let reader = if frontier.current == self.selected { + self.reader.clone() + } else { + match frontier.current.catalog { + Some(catalog) => Some(Arc::new( + timeout_at( + deadline, + CatalogReader::open(Arc::clone(&self.indexes), catalog), + ) + .await + .map_err(|_| PreparationBaseError::Inactive)??, + )), + None => None, + } + }; + self.live_lease()?; + if Instant::now() >= deadline { + return Err(PreparationBaseError::Inactive); + } + Ok(Self { + session: self.session.clone(), + selected: frontier.current, + reader, + files: Arc::clone(&self.files), + indexes: Arc::clone(&self.indexes), + }) + }) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } + pub async fn renew( + &self, + identity: MutationIdentity, + lease_ms: u64, + ) -> Result<(), PreparationBaseError> { + self.session.renew(identity, lease_ms).await + } +} +impl BaseResolver for PreparationBaseResolver { + async fn resolve( + &self, + base: ClosureBase, + ids: &[crate::ObjectId], + ) -> Result { + if ids.len() > PAGE_OBJECTS + || self.context().base != Some(base) + || ids + .iter() + .any(|oid| oid.format() != self.session.lease.format || oid.is_zero()) + { + return Err(ClosureError::Integrity); + } + let (_, deadline) = self + .session + .live_lease() + .map_err(|_| ClosureError::LeaseExpired)?; + let reader = self.reader.as_ref().ok_or(ClosureError::Integrity)?; + let headers = timeout_at(deadline, reader.headers(ids, &*self.files, &*self.files)) + .await + .map_err(|_| ClosureError::LeaseExpired)??; + self.session + .live_lease() + .map_err(|_| ClosureError::LeaseExpired)?; + if Instant::now() >= deadline { + return Err(ClosureError::LeaseExpired); + } + Ok(BaseBatch { + base, + objects: headers + .into_iter() + .map(|header| { + header.map(|header| BaseObject { + header, + certified: true, + }) + }) + .collect(), + }) + } +} diff --git a/crates/canopy-server/src/packs/publication/certificate.rs b/crates/canopy-server/src/packs/publication/certificate.rs new file mode 100644 index 0000000..9d4648d --- /dev/null +++ b/crates/canopy-server/src/packs/publication/certificate.rs @@ -0,0 +1,333 @@ +//! Bounded conditional catalog certificate. Only the prepared catalog factory +//! supplies signing facts; decoded bytes remain untrusted until MAC verification. +use super::*; +use crate::packs::directory::index::codec::{artifact, fixed, read_artifact}; + +pub const CERTIFICATE_BYTES: u32 = 1024; +const PAYLOAD_BYTES: u32 = 960; +const DOMAIN: &[u8] = b"canopy.catalog-attestation.v4\0"; + +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct CatalogCertificate(pub(super) CertificateEnvelope); +/// Shared bounded MAC carrier. Each typed proof validates its own domain/data. +#[derive(Clone, Debug, PartialEq, Eq)] +pub(super) struct CertificateEnvelope { + pub(super) body: Vec, + tag: [u8; 32], +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub(super) struct CertificateData { + pub(super) compaction: bool, + pub(super) tenant: [u8; 16], + pub(super) application: [u8; 16], + pub(super) token: PreparationToken, + pub(super) actor: String, + pub(super) retention_floor: u64, + pub(super) retention_certificate: Option<[u8; 32]>, + pub(super) base: GenerationFact, + pub(super) catalog: StoredCatalog, + pub(super) object_count: u64, + pub(super) edge_count: u64, + pub(super) input_count: u64, + pub(super) input_checkpoint_digest: Option<[u8; 32]>, + pub(super) inputs_digest: [u8; 32], + pub(super) inventory_digest: [u8; 32], + /// Exact ref-plan and ancestry evidence binding, minted only after checks + /// against this privately constructed prepared catalog. None is catalog-only. + pub(super) refs_digest: Option<[u8; 32]>, + /// Exact network completion payload, distinct from a refs-only publication. + pub(super) completion_digest: Option<[u8; 32]>, +} +impl CertificateData { + pub(super) fn from_prepared(prepared: &PreparedCatalog) -> Self { + let (_, target, check) = prepared.base.capability(); + Self { + compaction: false, + tenant: *target.tenant().as_bytes(), + application: *target.application().as_bytes(), + token: prepared.token(), + actor: check.actor.clone(), + retention_floor: prepared.base.retention_floor().generation, + retention_certificate: prepared.base.retention_floor().certificate, + base: prepared.base(), + catalog: prepared.catalog(), + object_count: prepared.object_count(), + edge_count: prepared.edge_count(), + input_count: prepared.input_count(), + input_checkpoint_digest: prepared.input_checkpoint_digest, + inputs_digest: prepared.inputs_digest(), + inventory_digest: prepared.inventory_digest(), + refs_digest: None, + completion_digest: None, + } + } + fn validate(&self) -> Result<(), CodecError> { + self.base.validate()?; + self.catalog + .validate() + .map_err(|_| CodecError::Invalid("invalid attested catalog"))?; + if validate_component(&self.actor).is_err() + || self.retention_floor > self.base.generation + || (self.retention_floor == 0) != self.retention_certificate.is_none() + || self.catalog.repository != self.token.repository + || self.catalog.operation != self.token.artifact_operation + || self.base.catalog.is_some_and(|base| { + base.repository != self.token.repository || base.format != self.catalog.format + }) + || [self.object_count, self.edge_count, self.input_count] + .iter() + .any(|value| *value > i64::MAX as u64) + || (self.input_count == 0) != (self.object_count == 0) + || (self.object_count == 0 && self.edge_count != 0) + || (self.input_checkpoint_digest.is_some() + && (self.input_count == 0 || self.compaction)) + || (self.compaction && (self.refs_digest.is_some() || self.completion_digest.is_some())) + { + return Err(CodecError::Invalid("invalid catalog attestation facts")); + } + Ok(()) + } +} +impl WireValue for CertificateData { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.validate()?; + e.write_bytes(DOMAIN)?; + e.write_bool(self.compaction)?; + e.write_bytes(&self.tenant)?; + e.write_bytes(&self.application)?; + self.token.encode(e)?; + e.write_text(&self.actor)?; + e.write_u64(self.retention_floor)?; + e.write_bool(self.retention_certificate.is_some())?; + if let Some(certificate) = self.retention_certificate { + e.write_bytes(&certificate)?; + } + // The token already binds repository and proposed creating namespace. + // Encode that context once; reconstruct the existing typed structures + // on decode instead of repeating full catalog descriptor domains. + e.write_u8(self.catalog.format.bytes() as u8)?; + e.write_u64(self.base.generation)?; + e.write_bool(self.base.catalog.is_some())?; + if let Some(catalog) = self.base.catalog { + e.write_bytes(&catalog.operation)?; + artifact(e, catalog.artifact)?; + } + self.base.refs.encode(e)?; + self.base + .certificate + .as_ref() + .map(|v| v.to_vec()) + .encode(e)?; + artifact(e, self.catalog.artifact)?; + e.write_u64(self.object_count)?; + e.write_u64(self.edge_count)?; + e.write_u64(self.input_count)?; + e.write_bool(self.input_checkpoint_digest.is_some())?; + if let Some(digest) = self.input_checkpoint_digest { + e.write_bytes(&digest)?; + } + e.write_bytes(&self.inputs_digest)?; + e.write_bytes(&self.inventory_digest)?; + e.write_bool(self.refs_digest.is_some())?; + if let Some(digest) = self.refs_digest { + e.write_bytes(&digest)?; + } + e.write_bool(self.completion_digest.is_some())?; + if let Some(digest) = self.completion_digest { + e.write_bytes(&digest)?; + } + Ok(()) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + if d.read_bytes()? != DOMAIN { + return Err(CodecError::Invalid("invalid catalog attestation domain")); + } + let compaction = d.read_bool()?; + let tenant = fixed(d)?; + let application = fixed(d)?; + let token = PreparationToken::decode(d)?; + let actor = d.read_text()?.into(); + let retention_floor = d.read_u64()?; + let retention_certificate = if d.read_bool()? { + Some(fixed(d)?) + } else { + None + }; + let format = match d.read_u8()? { + 20 => ObjectFormat::Sha1, + 32 => ObjectFormat::Sha256, + _ => return Err(CodecError::Invalid("invalid attested catalog format")), + }; + let generation = d.read_u64()?; + let catalog = if d.read_bool()? { + Some(StoredCatalog { + repository: token.repository, + operation: fixed(d)?, + format, + artifact: read_artifact(d)?, + }) + } else { + None + }; + let refs = Option::::decode(d)?; + let certificate = Option::>::decode(d)? + .map(|v| { + v.try_into() + .map_err(|_| CodecError::Invalid("invalid generation certificate")) + }) + .transpose()?; + let base = GenerationFact { + generation, + catalog, + refs, + certificate, + }; + let catalog = StoredCatalog { + repository: token.repository, + operation: token.artifact_operation, + format, + artifact: read_artifact(d)?, + }; + let value = Self { + compaction, + tenant, + application, + token, + actor, + retention_floor, + retention_certificate, + base, + catalog, + object_count: d.read_u64()?, + edge_count: d.read_u64()?, + input_count: d.read_u64()?, + input_checkpoint_digest: if d.read_bool()? { + Some(fixed(d)?) + } else { + None + }, + inputs_digest: fixed(d)?, + inventory_digest: fixed(d)?, + refs_digest: if d.read_bool()? { + Some(fixed(d)?) + } else { + None + }, + completion_digest: if d.read_bool()? { + Some(fixed(d)?) + } else { + None + }, + }; + value.validate()?; + Ok(value) + } +} +impl CertificateEnvelope { + pub(super) fn seal(data: &impl WireValue, seed: &[u8; 32]) -> Result { + let mut encoder = BoundedEncoder::new(PAYLOAD_BYTES)?; + data.encode(&mut encoder)?; + let body = encoder.finish(); + let tag = *mac(seed, &body).as_bytes(); + Ok(Self { body, tag }) + } + pub(super) fn data(&self) -> Result { + let mut decoder = BoundedDecoder::new(&self.body, PAYLOAD_BYTES)?; + let value = T::decode(&mut decoder)?; + decoder.finish()?; + Ok(value) + } + pub(super) fn authenticated(&self, seed: &[u8; 32]) -> bool { + mac(seed, &self.body) == blake3::Hash::from_bytes(self.tag) + } +} +impl CatalogCertificate { + pub(super) fn seal(data: &CertificateData, seed: &[u8; 32]) -> Result { + Ok(Self(CertificateEnvelope::seal(data, seed)?)) + } + pub(super) fn data(&self) -> Result { + self.0.data() + } + pub(super) fn authenticated(&self, seed: &[u8; 32]) -> bool { + self.0.authenticated(seed) + } + pub(super) fn bytes(&self) -> Result, CodecError> { + let mut encoder = BoundedEncoder::new(CERTIFICATE_BYTES)?; + self.encode(&mut encoder)?; + Ok(encoder.finish()) + } +} +fn mac(seed: &[u8; 32], bytes: &[u8]) -> blake3::Hash { + // Reuse the repository secret with a separate key derivation domain; + // public signed-push nonces and catalog certificates share no MAC key. + let key = blake3::derive_key("canopy.catalog-attestation-key.v1", seed); + blake3::keyed_hash(&key, bytes) +} +impl WireValue for CertificateEnvelope { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + if self.body.is_empty() || self.body.len() > PAYLOAD_BYTES as usize { + return Err(CodecError::Invalid("certificate size")); + } + e.write_bytes(&self.body)?; + e.write_bytes(&self.tag) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let body = d.read_bytes()?; + if body.is_empty() || body.len() > PAYLOAD_BYTES as usize { + return Err(CodecError::Invalid("certificate size")); + } + Ok(Self { + body: body.into(), + tag: fixed(d)?, + }) + } +} +impl WireValue for CatalogCertificate { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.data()?; + self.0.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self(CertificateEnvelope::decode(d)?); + value.data()?; + Ok(value) + } +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct RegisteredCatalog { + pub token: PreparationToken, + pub certificate_digest: [u8; 32], +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum AttestationOutcome { + Registered(RegisteredCatalog), + Denied(PreparationDenial), +} +impl WireValue for AttestationOutcome { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + match self { + Self::Registered(value) => { + e.write_u8(0)?; + value.token.encode(e)?; + e.write_bytes(&value.certificate_digest) + } + Self::Denied(reason) => PreparationReply::Denied(*reason).encode(e), + } + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + match d.read_u8()? { + 0 => Ok(Self::Registered(RegisteredCatalog { + token: PreparationToken::decode(d)?, + certificate_digest: fixed(d)?, + })), + 1 => Ok(Self::Denied(PreparationDenial::Unauthorized)), + 2 => Ok(Self::Denied(PreparationDenial::Conflict)), + 3 => Ok(Self::Denied(PreparationDenial::Stale)), + 4 => Ok(Self::Denied(PreparationDenial::Expired)), + 5 => Ok(Self::Denied(PreparationDenial::Capacity)), + 6 => Ok(Self::Denied(PreparationDenial::Missing)), + _ => Err(CodecError::Invalid("invalid attestation outcome")), + } + } +} diff --git a/crates/canopy-server/src/packs/publication/codec.rs b/crates/canopy-server/src/packs/publication/codec.rs new file mode 100644 index 0000000..d60ad89 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/codec.rs @@ -0,0 +1,338 @@ +use super::*; +use crate::packs::directory::index::codec::fixed; + +fn invalid() -> CodecError { + CodecError::Invalid("invalid preparation context") +} +pub(in crate::packs) fn artifact_valid(operation: [u8; 16]) -> Result<(), CodecError> { + let sequence = u64::from_be_bytes(operation[8..].try_into().map_err(|_| invalid())?); + if &operation[..8] != b"CANOPY01" || sequence == 0 || sequence > i64::MAX as u64 { + return Err(invalid()); + } + Ok(()) +} +fn owner_valid(owner: OwnerFence) -> Result<(), CodecError> { + if owner.epoch == 0 { + Err(invalid()) + } else { + Ok(()) + } +} +fn owner_encode(owner: OwnerFence, e: &mut BoundedEncoder) -> Result<(), CodecError> { + owner_valid(owner)?; + e.write_bytes(owner.incarnation.as_bytes())?; + e.write_u64(owner.epoch) +} +fn owner_decode(d: &mut BoundedDecoder<'_>) -> Result { + let owner = OwnerFence { + incarnation: IncarnationId::from_bytes(fixed(d)?), + epoch: d.read_u64()?, + }; + owner_valid(owner)?; + Ok(owner) +} +fn repo_valid(repo: [u8; 16]) -> Result<(), CodecError> { + crate::validate_repository_id(repo).map_err(|_| invalid()) +} +fn actor_valid(actor: &str) -> Result<(), CodecError> { + validate_component(actor).map_err(|_| invalid()) +} +fn lease_valid(ms: u64) -> Result<(), CodecError> { + if ms == 0 || ms > MAX_LEASE_MS { + Err(invalid()) + } else { + Ok(()) + } +} +impl WireValue for PreparationToken { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + repo_valid(self.repository)?; + artifact_valid(self.artifact_operation)?; + if self.operation == [0; 16] || self.attempt == 0 || self.attempt > i64::MAX as u64 { + return Err(invalid()); + } + e.write_bytes(&self.repository)?; + e.write_bytes(&self.operation)?; + e.write_bytes(&self.artifact_operation)?; + e.write_bytes(&self.request_digest)?; + owner_encode(self.owner, e)?; + e.write_u64(self.attempt) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + repository: fixed(d)?, + operation: fixed(d)?, + artifact_operation: fixed(d)?, + request_digest: fixed(d)?, + owner: owner_decode(d)?, + attempt: d.read_u64()?, + }; + repo_valid(value.repository)?; + artifact_valid(value.artifact_operation)?; + if value.operation == [0; 16] || value.attempt == 0 || value.attempt > i64::MAX as u64 { + return Err(invalid()); + } + Ok(value) + } +} +impl WireValue for GenerationFact { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.validate()?; + e.write_u64(self.generation)?; + self.catalog.encode(e)?; + self.refs.encode(e)?; + self.certificate.as_ref().map(|v| v.to_vec()).encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let generation = d.read_u64()?; + let catalog = Option::::decode(d)?; + let refs = Option::::decode(d)?; + let certificate = Option::>::decode(d)? + .map(|v| v.try_into().map_err(|_| invalid())) + .transpose()?; + let value = Self { + generation, + catalog, + refs, + certificate, + }; + value.validate()?; + Ok(value) + } +} +impl WireValue for PreparationLease { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.validate()?; + self.token.encode(e)?; + self.base.encode(e)?; + e.write_u8(self.format.bytes() as u8)?; + e.write_i64(self.observed_at_ms)?; + e.write_i64(self.expires_at_ms) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let token = PreparationToken::decode(d)?; + let base = GenerationFact::decode(d)?; + let format = match d.read_u8()? { + 20 => ObjectFormat::Sha1, + 32 => ObjectFormat::Sha256, + _ => return Err(invalid()), + }; + let value = Self { + token, + base, + format, + observed_at_ms: d.read_i64()?, + expires_at_ms: d.read_i64()?, + }; + value.validate()?; + Ok(value) + } +} +impl PreparationLease { + fn validate(self) -> Result<(), CodecError> { + self.base.validate()?; + if self.observed_at_ms < 0 + || self.expires_at_ms <= self.observed_at_ms + || self.base.catalog.is_some_and(|catalog| { + catalog.repository != self.token.repository || catalog.format != self.format + }) + { + return Err(invalid()); + } + Ok(()) + } +} +impl PreparationFrontier { + pub(super) fn validate(self) -> Result<(), CodecError> { + self.lease.validate()?; + self.current.validate()?; + if self.current.generation < self.lease.base.generation + || (self.current.generation == self.lease.base.generation + && self.current != self.lease.base) + || self.current.catalog.is_some_and(|catalog| { + catalog.repository != self.lease.token.repository + || catalog.format != self.lease.format + }) + { + return Err(invalid()); + } + Ok(()) + } +} +impl WireValue for PreparationFrontier { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.validate()?; + self.lease.encode(e)?; + self.current.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + lease: PreparationLease::decode(d)?, + current: GenerationFact::decode(d)?, + }; + value.validate()?; + Ok(value) + } +} +impl WireValue for PreparationReply { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + match self { + Self::Granted(lease) => { + e.write_u8(0)?; + lease.encode(e) + } + Self::Denied(reason) => e.write_u8(match reason { + PreparationDenial::Unauthorized => 1, + PreparationDenial::Conflict => 2, + PreparationDenial::Stale => 3, + PreparationDenial::Expired => 4, + PreparationDenial::Capacity => 5, + PreparationDenial::Missing => 6, + }), + } + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + Ok(match d.read_u8()? { + 0 => Self::Granted(Box::new(PreparationLease::decode(d)?)), + 1 => Self::Denied(PreparationDenial::Unauthorized), + 2 => Self::Denied(PreparationDenial::Conflict), + 3 => Self::Denied(PreparationDenial::Stale), + 4 => Self::Denied(PreparationDenial::Expired), + 5 => Self::Denied(PreparationDenial::Capacity), + 6 => Self::Denied(PreparationDenial::Missing), + _ => return Err(invalid()), + }) + } +} +impl WireValue for StagingLease { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + if self.observed_at_ms < 0 || self.expires_at_ms <= self.observed_at_ms { + return Err(invalid()); + } + self.token.encode(e)?; + e.write_u8(self.format.bytes() as u8)?; + e.write_i64(self.observed_at_ms)?; + e.write_i64(self.expires_at_ms) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let token = PreparationToken::decode(d)?; + let format = match d.read_u8()? { + 20 => ObjectFormat::Sha1, + 32 => ObjectFormat::Sha256, + _ => return Err(invalid()), + }; + let value = Self { + token, + format, + observed_at_ms: d.read_i64()?, + expires_at_ms: d.read_i64()?, + }; + if value.observed_at_ms < 0 || value.expires_at_ms <= value.observed_at_ms { + return Err(invalid()); + } + Ok(value) + } +} +impl WireValue for StagingReply { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + match self { + Self::Granted(lease) => { + e.write_u8(0)?; + lease.encode(e) + } + Self::Denied(reason) => PreparationReply::Denied(*reason).encode(e), + } + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + Ok(match d.read_u8()? { + 0 => Self::Granted(Box::new(StagingLease::decode(d)?)), + 1 => Self::Denied(PreparationDenial::Unauthorized), + 2 => Self::Denied(PreparationDenial::Conflict), + 3 => Self::Denied(PreparationDenial::Stale), + 4 => Self::Denied(PreparationDenial::Expired), + 5 => Self::Denied(PreparationDenial::Capacity), + 6 => Self::Denied(PreparationDenial::Missing), + _ => return Err(invalid()), + }) + } +} +impl WireValue for BeginRequest { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + repo_valid(self.repository)?; + actor_valid(&self.actor)?; + lease_valid(self.lease_ms)?; + if self.operation == [0; 16] { + return Err(invalid()); + } + e.write_bytes(&self.repository)?; + e.write_bytes(&self.operation)?; + e.write_bytes(&self.request_digest)?; + e.write_text(&self.actor)?; + e.write_u64(self.lease_ms) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + repository: fixed(d)?, + operation: fixed(d)?, + request_digest: fixed(d)?, + actor: d.read_text()?.into(), + lease_ms: d.read_u64()?, + }; + repo_valid(value.repository)?; + actor_valid(&value.actor)?; + lease_valid(value.lease_ms)?; + if value.operation == [0; 16] { + return Err(invalid()); + } + Ok(value) + } +} +impl WireValue for LeaseCheck { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + actor_valid(&self.actor)?; + self.token.encode(e)?; + e.write_text(&self.actor) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + token: PreparationToken::decode(d)?, + actor: d.read_text()?.into(), + }; + actor_valid(&value.actor)?; + Ok(value) + } +} +impl WireValue for LeaseRequest { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + lease_valid(self.lease_ms)?; + self.check.encode(e)?; + e.write_u64(self.lease_ms) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + check: LeaseCheck::decode(d)?, + lease_ms: d.read_u64()?, + }; + lease_valid(value.lease_ms)?; + Ok(value) + } +} +impl WireValue for MaintenanceRequest { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + repo_valid(self.repository)?; + actor_valid(&self.actor)?; + e.write_bytes(&self.repository)?; + e.write_text(&self.actor)?; + owner_encode(self.owner, e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + repository: fixed(d)?, + actor: d.read_text()?.into(), + owner: owner_decode(d)?, + }; + repo_valid(value.repository)?; + actor_valid(&value.actor)?; + Ok(value) + } +} diff --git a/crates/canopy-server/src/packs/publication/commands.rs b/crates/canopy-server/src/packs/publication/commands.rs new file mode 100644 index 0000000..ed03456 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/commands.rs @@ -0,0 +1,495 @@ +use super::sql::*; +use super::*; +fn denied(reason: PreparationDenial) -> CommandResult { + CommandResult::Rejected(PreparationReply::Denied(reason)) +} +pub(super) fn authorized( + context: &CommandContext<'_, '_>, + repository: [u8; 16], + actor: &str, + role: TokenScope, +) -> cellule_runtime::Result> { + validate_component(actor)?; + if context.target() + != &crate::repository_target( + context.target().tenant(), + context.target().application(), + repository, + )? + { + return Ok(None); + } + if !decode_access(&context.sql(&SqlBatch { + statements: vec![access_statement(actor)], + })?)? + .is_some_and(|value| value >= role) + { + return Ok(None); + } + identity(&context.sql(&statement(IDENTITY, vec![]))?, repository) +} +pub(super) fn load( + context: &CommandContext<'_, '_>, + token: PreparationToken, +) -> cellule_runtime::Result> { + operation( + &context.sql(&statement(OPERATION, vec![blob(token.operation)]))?, + token.repository, + token.operation, + ) +} +pub(super) fn fact( + context: &CommandContext<'_, '_>, + repository: [u8; 16], + format: ObjectFormat, + base: Option, +) -> cellule_runtime::Result { + generation( + &context.sql(&match base { + None => statement(CURRENT, vec![]), + Some(value) => statement(GENERATION, vec![number(value)?]), + })?, + repository, + format, + ) +} +pub(super) fn matched(row: &Operation, check: &LeaseCheck) -> bool { + row.actor == check.actor && row.token == check.token +} +pub(super) fn pin(sets: &[SqlResultSet], row: &Operation) -> cellule_runtime::Result<()> { + let Some( + [ + operation, + epoch, + artifact_operation, + generation, + SqlValue::Integer(expires), + ], + ) = rows(sets)?.first().map(Vec::as_slice) + else { + return Err(Error::Command("catalog attempt pin is absent")); + }; + if fixed::<16>(operation)? != row.token.operation + || u64::from_be_bytes(fixed(epoch)?) != row.token.owner.epoch + || fixed::<16>(artifact_operation)? != row.token.artifact_operation + || optional_generation(generation)? != row.generation + || *expires != row.expires + { + return Err(Error::Command("catalog attempt pin differs")); + } + Ok(()) +} +pub(super) fn pin_query(token: PreparationToken) -> cellule_runtime::Result { + Ok(statement( + "SELECT operation,owner_epoch,artifact_operation,generation,expires_at_ms FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2", + vec![ + blob(token.owner.incarnation.as_bytes()), + number(token.attempt)?, + ], + )) +} +pub(super) fn check_pin( + context: &CommandContext<'_, '_>, + row: &Operation, +) -> cellule_runtime::Result<()> { + pin(&context.sql(&pin_query(row.token)?)?, row) +} +pub(super) fn token( + context: &CommandContext<'_, '_>, + repository: [u8; 16], + operation: [u8; 16], + request_digest: [u8; 32], +) -> cellule_runtime::Result { + if operation == [0; 16] || context.sequence() == 0 || context.sequence() > i64::MAX as u64 { + return Err(Error::Command("invalid catalog attempt identity")); + } + Ok(PreparationToken { + repository, + operation, + artifact_operation: allocate_artifacts(context)?, + request_digest, + owner: context.owner_fence(), + attempt: context.sequence(), + }) +} + +pub(super) fn logical_available( + context: &CommandContext<'_, '_>, + input: &BeginRequest, +) -> cellule_runtime::Result { + // A logical outcome already exists: callers must look it up before + // native preparation. Never allocate another namespace for a completed + // push, or admit an identity conflicting with a pending network push. + if !rows(&context.sql(&statement( + "SELECT id FROM catalog_compactions WHERE id=?1 UNION ALL SELECT id FROM catalog_initialization WHERE id=?1", + vec![blob(input.operation)], + ))?)? + .is_empty() + { + return Ok(false); + } + let saved = context.sql(&statement( + "SELECT actor,request_digest,response_id,publication FROM pushes WHERE id=?1", + vec![blob(input.operation)], + ))?; + if let Some(row) = rows(&saved)?.first() { + let [SqlValue::Text(actor), digest, response, publication] = row.as_slice() else { + return Err(Error::Command("invalid preparation push identity")); + }; + if *actor != input.actor + || fixed::<32>(digest)? != input.request_digest + || *response != SqlValue::Null + || *publication != SqlValue::Null + { + return Ok(false); + } + } + Ok(true) +} + +pub struct BeginPreparation; +impl Command for BeginPreparation { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 11; + const CODEC_VERSION: u32 = 1; + type Input = BeginRequest; + type Output = PreparationReply; + fn execute( + context: &mut CommandContext<'_, '_>, + input: BeginRequest, + ) -> cellule_runtime::Result> { + let Some(format) = authorized(context, input.repository, &input.actor, TokenScope::Write)? + else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + if !logical_available(context, &input)? { + return Ok(denied(PreparationDenial::Conflict)); + } + let now = now(context.now_ms())?; + let expires = expiry(now, input.lease_ms)?; + if let Some(existing) = operation( + &context.sql(&statement(OPERATION, vec![blob(input.operation)]))?, + input.repository, + input.operation, + )? { + if existing.actor != input.actor + || existing.token.request_digest != input.request_digest + { + return Ok(denied(PreparationDenial::Conflict)); + } + if existing.token.owner != context.owner_fence() { + return Ok(denied(PreparationDenial::Stale)); + } + if existing.expires <= now { + return Ok(denied(PreparationDenial::Expired)); + } + if existing.generation.is_none() { + return Ok(denied(PreparationDenial::Conflict)); + } + check_pin(context, &existing)?; + let base = fact(context, input.repository, format, existing.generation)?; + return Ok(CommandResult::Success(PreparationReply::Granted(Box::new( + grant(&existing, format, base, now)?, + )))); + } + if !quota(context, true)? { + return Ok(denied(PreparationDenial::Capacity)); + } + let base = fact(context, input.repository, format, None)?; + let new_token = token( + context, + input.repository, + input.operation, + input.request_digest, + )?; + insert_lease(context, new_token, Some(base.generation), expires)?; + context.sql(&statement("INSERT INTO catalog_operations(id,actor,request_digest,incarnation,owner_epoch,admission_sequence,artifact_operation,generation,expires_at_ms) VALUES(?1,?2,?3,?4,?5,?6,?7,?8,?9)",vec![blob(input.operation),SqlValue::Text(input.actor.clone()),blob(input.request_digest),blob(new_token.owner.incarnation.as_bytes()),blob(new_token.owner.epoch.to_be_bytes()),number(new_token.attempt)?,blob(new_token.artifact_operation),number(base.generation)?,SqlValue::Integer(expires)]))?; + let row = Operation { + actor: input.actor, + token: new_token, + generation: Some(base.generation), + expires, + }; + Ok(CommandResult::Success(PreparationReply::Granted(Box::new( + grant(&row, format, base, now)?, + )))) + } +} + +pub struct ClaimPreparation; +impl Command for ClaimPreparation { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 12; + const CODEC_VERSION: u32 = 1; + type Input = LeaseRequest; + type Output = PreparationReply; + fn execute( + context: &mut CommandContext<'_, '_>, + input: LeaseRequest, + ) -> cellule_runtime::Result> { + let check = input.check; + let Some(format) = authorized( + context, + check.token.repository, + &check.actor, + TokenScope::Write, + )? + else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + let Some(existing) = load(context, check.token)? else { + return Ok(denied(PreparationDenial::Missing)); + }; + if !matched(&existing, &check) { + return Ok(denied(PreparationDenial::Stale)); + } + if existing.generation.is_none() { + return Ok(denied(PreparationDenial::Conflict)); + } + check_pin(context, &existing)?; + if !quota(context, false)? { + return Ok(denied(PreparationDenial::Capacity)); + } + let now = now(context.now_ms())?; + let expires = expiry(now, input.lease_ms)?; + let base = fact(context, check.token.repository, format, None)?; + let next = token( + context, + check.token.repository, + check.token.operation, + check.token.request_digest, + )?; + insert_lease(context, next, Some(base.generation), expires)?; + // Keep the previous pin unchanged, even when rebasing to a new root. + context.sql(&statement("UPDATE catalog_operations SET incarnation=?1,owner_epoch=?2,admission_sequence=?3,generation=?4,expires_at_ms=?5,attestation=NULL,attestation_digest=NULL,artifact_operation=?7 WHERE id=?6",vec![blob(next.owner.incarnation.as_bytes()),blob(next.owner.epoch.to_be_bytes()),number(next.attempt)?,number(base.generation)?,SqlValue::Integer(expires),blob(next.operation),blob(next.artifact_operation)]))?; + let row = Operation { + actor: check.actor, + token: next, + generation: Some(base.generation), + expires, + }; + Ok(CommandResult::Success(PreparationReply::Granted(Box::new( + grant(&row, format, base, now)?, + )))) + } +} + +pub struct RenewPreparation; +impl Command for RenewPreparation { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 13; + const CODEC_VERSION: u32 = 1; + type Input = LeaseRequest; + type Output = PreparationReply; + fn execute( + context: &mut CommandContext<'_, '_>, + input: LeaseRequest, + ) -> cellule_runtime::Result> { + let check = input.check; + let Some(format) = authorized( + context, + check.token.repository, + &check.actor, + TokenScope::Write, + )? + else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + if check.token.owner != context.owner_fence() { + return Ok(denied(PreparationDenial::Stale)); + } + let Some(mut existing) = load(context, check.token)? else { + return Ok(denied(PreparationDenial::Missing)); + }; + if !matched(&existing, &check) { + return Ok(denied(PreparationDenial::Stale)); + } + if existing.generation.is_none() { + return Ok(denied(PreparationDenial::Conflict)); + } + check_pin(context, &existing)?; + let now = now(context.now_ms())?; + if existing.expires <= now { + return Ok(denied(PreparationDenial::Expired)); + } + let base = fact(context, check.token.repository, format, existing.generation)?; + existing.expires = existing.expires.max(expiry(now, input.lease_ms)?); + context.sql(&statement("UPDATE catalog_leases SET expires_at_ms=?1 WHERE incarnation=?2 AND admission_sequence=?3",vec![SqlValue::Integer(existing.expires),blob(existing.token.owner.incarnation.as_bytes()),number(existing.token.attempt)?]))?; + context.sql(&statement( + "UPDATE catalog_operations SET expires_at_ms=?1 WHERE id=?2", + vec![ + SqlValue::Integer(existing.expires), + blob(existing.token.operation), + ], + ))?; + Ok(CommandResult::Success(PreparationReply::Granted(Box::new( + grant(&existing, format, base, now)?, + )))) + } +} + +pub struct AbortPreparation; +impl Command for AbortPreparation { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 14; + const CODEC_VERSION: u32 = 1; + type Input = LeaseCheck; + type Output = bool; + fn execute( + context: &mut CommandContext<'_, '_>, + check: LeaseCheck, + ) -> cellule_runtime::Result> { + if authorized( + context, + check.token.repository, + &check.actor, + TokenScope::Write, + )? + .is_none() + || check.token.owner != context.owner_fence() + { + return Ok(CommandResult::Rejected(false)); + } + let Some(existing) = load(context, check.token)? else { + return Ok(CommandResult::Rejected(false)); + }; + if !matched(&existing, &check) { + return Ok(CommandResult::Rejected(false)); + } + // Aborting stops admission but cannot retract an already borrowed pin. + context.sql(&statement( + "DELETE FROM catalog_operations WHERE id=?1", + vec![blob(check.token.operation)], + ))?; + Ok(CommandResult::Success(true)) + } +} + +pub struct CheckPreparation; +impl Query for CheckPreparation { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 15; + const CODEC_VERSION: u32 = 1; + type Input = LeaseCheck; + type Output = Option; + fn execute( + context: &mut QueryContext<'_>, + check: LeaseCheck, + ) -> cellule_runtime::Result> { + validate_component(&check.actor)?; + if !decode_access(&context.sql(&SqlBatch { + statements: vec![access_statement(&check.actor)], + })?)? + .is_some_and(|role| role >= TokenScope::Write) + { + return Ok(None); + } + let Some(format) = identity( + &context.sql(&statement(IDENTITY, vec![]))?, + check.token.repository, + )? + else { + return Ok(None); + }; + let Some(row) = operation( + &context.sql(&statement(OPERATION, vec![blob(check.token.operation)]))?, + check.token.repository, + check.token.operation, + )? + else { + return Ok(None); + }; + if !matched(&row, &check) { + return Ok(None); + } + let now = now(context.now_ms())?; + if row.expires <= now { + return Ok(None); + } + let Some(floor) = row.generation else { + return Ok(None); + }; + pin(&context.sql(&pin_query(row.token)?)?, &row)?; + let base = generation( + &context.sql(&statement(GENERATION, vec![number(floor)?]))?, + check.token.repository, + format, + )?; + Ok(Some(grant(&row, format, base, now)?)) + } +} + +/// Refresh catalog facts under the original attempt without allocating a new +/// namespace, changing its retention floor or submitting a durable Claim. +pub struct CheckPreparationFrontier; +impl Query for CheckPreparationFrontier { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 21; + const CODEC_VERSION: u32 = 1; + type Input = LeaseCheck; + type Output = Option; + fn execute( + context: &mut QueryContext<'_>, + check: LeaseCheck, + ) -> cellule_runtime::Result { + let Some(mut lease) = CheckPreparation::execute(context, check)? else { + return Ok(None); + }; + // Both facts come from this query's one committed SQLite snapshot. + let current = generation( + &context.sql(&statement(CURRENT, vec![]))?, + lease.token.repository, + lease.format, + )?; + // Sampling before the worker queue must not extend a usable lease. + lease.observed_at_ms = now(context.now_ms())?; + if lease.observed_at_ms >= lease.expires_at_ms { + return Ok(None); + } + let frontier = PreparationFrontier { lease, current }; + frontier.validate()?; + Ok(Some(frontier)) + } +} + +pub struct ReapPreparation; +pub(super) const REAP_GENERATIONS: &str = "DELETE FROM catalog_generations WHERE generation IN (SELECT g.generation FROM catalog_generations g WHERE g.generation>0 AND g.generation<(SELECT generation FROM catalog_state WHERE singleton=1) AND g.generation, + input: MaintenanceRequest, + ) -> cellule_runtime::Result> { + if input.owner != context.owner_fence() + || authorized(context, input.repository, &input.actor, TokenScope::Admin)?.is_none() + { + return Ok(CommandResult::Rejected(0)); + } + let now = now(context.now_ms())?; + let operations=context.sql(&statement("DELETE FROM catalog_operations WHERE id IN (SELECT id FROM catalog_operations WHERE expires_at_ms<=?1 ORDER BY expires_at_ms,id LIMIT ?2)",vec![SqlValue::Integer(now),number(REAP_ROWS)?]))?; + let leases=context.sql(&statement("DELETE FROM catalog_leases WHERE (incarnation,admission_sequence) IN (SELECT l.incarnation,l.admission_sequence FROM catalog_leases l WHERE l.expires_at_ms<=?1 AND NOT EXISTS(SELECT 1 FROM catalog_operations o WHERE o.incarnation=l.incarnation AND o.admission_sequence=l.admission_sequence) ORDER BY l.expires_at_ms,l.incarnation,l.admission_sequence LIMIT ?2)",vec![SqlValue::Integer(now),number(REAP_ROWS)?]))?; + // This only removes obsolete facts from the current SQL state. Old + // recovery snapshots retain their own facts. An independent attempt's + // floor protects all later facts, including generations selected by a + // read-only frontier refresh. Expired pins protect until actually reaped. + // Remote deletion still needs all recovery/backup/reader retention. + let generations = context.sql(&statement(REAP_GENERATIONS, vec![number(REAP_ROWS)?]))?; + let removed = operations + .first() + .ok_or(Error::Command("missing catalog operation reap"))? + .rows_affected + + leases + .first() + .ok_or(Error::Command("missing catalog lease reap"))? + .rows_affected + + generations + .first() + .ok_or(Error::Command("missing catalog generation reap"))? + .rows_affected; + Ok(CommandResult::Success(removed)) + } +} diff --git a/crates/canopy-server/src/packs/publication/compaction.rs b/crates/canopy-server/src/packs/publication/compaction.rs new file mode 100644 index 0000000..cf92ede --- /dev/null +++ b/crates/canopy-server/src/packs/publication/compaction.rs @@ -0,0 +1,197 @@ +//! Certified directory compaction. This changes neither refs nor native packs +//! and grants no authority to delete old artifacts or shorten retained floors. +use super::*; +use crate::packs::{ + catalog::CatalogSnapshot, + directory::{ + DirectoryBuilder, DirectoryPartitioner, RUN_TARGET_BYTES, + index::{NodeRef, codec::reference}, + snapshot::LEVEL_ZERO_ROOTS, + }, + metadata::{MetadataError, MetadataLimits}, +}; +use cellule_ltx::DiskBudget; +use std::{path::Path, sync::Arc}; +use tokio::time::timeout_at; +mod prepare; +mod publish; +mod range; +mod schedule; +pub use publish::{ + CheckCompletedCompaction, CompactionReply, PublishCatalogCompaction, PublishedCompaction, +}; +pub use range::CompactionSource; +use range::RangeSelection; +pub use schedule::{CompactionPlanner, CompactionPolicy, CompactionPressure}; + +#[derive(Clone, Copy)] +pub struct CompactionLimits { + pub input_runs: u32, + pub input_bytes: u64, + pub spool: MetadataLimits, + pub output: MetadataLimits, +} +impl Default for CompactionLimits { + fn default() -> Self { + Self { + input_runs: 128, + input_bytes: 256 << 20, + spool: MetadataLimits::default(), + output: MetadataLimits { + max_file_bytes: RUN_TARGET_BYTES, + ..MetadataLimits::default() + }, + } + } +} +impl CompactionLimits { + fn validate(self) -> Result<(), CatalogPreparationError> { + DirectoryPartitioner::validate_limits(self.output)?; + if self.input_runs == 0 + || self.input_runs > 4096 + || self.input_bytes == 0 + || self.input_bytes > 16 << 30 + { + return Err(MetadataError::Limit.into()); + } + Ok(()) + } +} +#[derive(Clone)] +enum Selection { + Ingress(Vec), + Range(Box), +} +/// Only preparation from query-derived certified inputs constructs this object. +/// It retains exact selected inputs for rebinding against a moving frontier. +pub struct PreparedCompaction { + base: Arc, + selected: Selection, + output: NodeRef, + catalog: StoredCatalog, + object_count: u64, + edge_count: u64, + input_count: u64, + inputs_digest: [u8; 32], + inventory_digest: [u8; 32], +} +impl PreparedCompaction { + pub(super) fn preparation_base(&self) -> &PreparationBaseResolver { + &self.base + } + pub fn catalog(&self) -> StoredCatalog { + self.catalog + } + pub fn token(&self) -> PreparationToken { + self.base.context_token() + } + pub fn object_count(&self) -> u64 { + self.object_count + } + pub fn input_count(&self) -> u64 { + self.input_count + } + pub fn inventory_digest(&self) -> [u8; 32] { + self.inventory_digest + } + pub fn base(&self) -> GenerationFact { + self.base.generation_fact() + } + pub async fn certificate(&self) -> Result { + let (_, deadline) = self.base.live_lease()?; + let (_, target, check) = self.base.capability(); + let data = certificate::CertificateData { + compaction: true, + tenant: *target.tenant().as_bytes(), + application: *target.application().as_bytes(), + token: self.token(), + actor: check.actor.clone(), + retention_floor: self.base.retention_floor().generation, + retention_certificate: self.base.retention_floor().certificate, + base: self.base(), + catalog: self.catalog, + object_count: self.object_count, + edge_count: self.edge_count, + input_count: self.input_count, + input_checkpoint_digest: None, + inputs_digest: self.inputs_digest, + inventory_digest: self.inventory_digest, + refs_digest: None, + completion_digest: None, + }; + timeout_at( + deadline, + attestation::issue_data_certificate(&self.base, data), + ) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } + /// Selected inputs must still occupy their certified positions. A range + /// selection can rebind unrelated level updates, but replaced inputs or new + /// overlapping target runs reject rather than resurrect obsolete placement. + /// Concurrent ingress and the current source tree are preserved. + pub async fn reconcile(&self) -> Result { + let (_, deadline) = self.base.live_lease()?; + timeout_at(deadline, async { + let base = Arc::new(self.base.select_current().await?); + let catalog = if base.generation_fact() == self.base() { + self.catalog + } else { + replacement(&base, &self.selected, self.output).await? + }; + base.live_lease()?; + Ok(Self { + base, + catalog, + selected: self.selected.clone(), + output: self.output, + object_count: self.object_count, + edge_count: self.edge_count, + input_count: self.input_count, + inputs_digest: self.inputs_digest, + inventory_digest: self.inventory_digest, + }) + }) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } +} +async fn replacement( + base: &PreparationBaseResolver, + selected: &Selection, + output: NodeRef, +) -> Result { + let (mut directory, sources) = base.catalog_parts(); + let indexes = base.indexes(); + match selected { + Selection::Ingress(selected) => { + if selected.len() < 2 + || selected + .iter() + .any(|root| !directory.level_zero.contains(root)) + { + return Err(crate::packs::directory::index::IndexError::Stale.into()); + } + directory.level_zero.retain(|root| !selected.contains(root)); + directory.append(indexes.ranges(), output).await?; + } + Selection::Range(selected) => { + selected + .replace( + &mut directory, + indexes.ranges(), + base.context().operation, + output, + ) + .await?; + } + } + let store = indexes.store(); + let operation = base.context_token().artifact_operation; + Ok(CatalogSnapshot { + directory: directory.upload(&store, operation).await?, + sources, + } + .upload(&store, operation) + .await?) +} diff --git a/crates/canopy-server/src/packs/publication/compaction/prepare.rs b/crates/canopy-server/src/packs/publication/compaction/prepare.rs new file mode 100644 index 0000000..ef081f7 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/compaction/prepare.rs @@ -0,0 +1,230 @@ +use super::*; +use crate::packs::directory::snapshot::RunLoader; + +impl PreparedCompaction { + /// Merge bounded selected ingress roots from the authoritative base. Caller + /// indices select from that base; raw input/output descriptors are not accepted. + pub async fn prepare( + root: &Path, + budget: DiskBudget, + base: Arc, + selected: &[usize], + limits: CompactionLimits, + ) -> Result { + limits.validate()?; + let (_, deadline) = base.live_lease()?; + let (directory, _) = base.catalog_parts(); + directory.validate()?; + if selected.len() < 2 || selected.len() > LEVEL_ZERO_ROOTS { + return Err(MetadataError::Limit.into()); + } + let mut inputs = Vec::with_capacity(selected.len()); + for &at in selected { + let root = *directory + .level_zero + .get(at) + .ok_or(CatalogPreparationError::Integrity)?; + if inputs.contains(&root) { + return Err(CatalogPreparationError::Integrity); + } + inputs.push(root); + } + // Canonical selection order makes replay/binding independent of caller + // index order and never materializes the indexed run inventory. + inputs.sort_by_key(|root| (root.operation, root.artifact.digest)); + let root = root.to_owned(); + timeout_at(deadline, async { + check_admin(&base).await?; + Self::prepare_inner(root, budget, base, inputs, limits).await + }) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } + async fn prepare_inner( + root: std::path::PathBuf, + budget: DiskBudget, + base: Arc, + selected: Vec, + limits: CompactionLimits, + ) -> Result { + let context = base.context(); + let mut builder = new_builder(root, budget.clone(), &base, limits.spool).await?; + let mut binding = BoundedEncoder::new(16 << 10)?; + binding.write_bytes(b"canopy.ingress-compaction.v1\0")?; + binding.write_bytes(&context.repository)?; + binding.write_u8(context.format.bytes() as u8)?; + binding.write_count(selected.len())?; + let indexes = base.indexes(); + let mut input_count = 0_u64; + let mut input_bytes = 0_u64; + for &root in &selected { + reference(&mut binding, root)?; + let mut runs = 0_u64; + let mut objects = 0_u64; + let mut first = None; + let mut last = None; + let mut cursor = indexes.ranges().cursor(Some(root), None)?; + while let Some(stored) = cursor.next().await? { + runs = runs.checked_add(1).ok_or(MetadataError::Limit)?; + objects = objects + .checked_add(stored.coverage.object_count) + .ok_or(MetadataError::Limit)?; + first.get_or_insert(stored.coverage.first_oid); + last = Some(stored.coverage.last_oid); + input_count = input_count.checked_add(1).ok_or(MetadataError::Limit)?; + input_bytes = input_bytes + .checked_add(stored.run.size) + .ok_or(MetadataError::Limit)?; + if input_count > u64::from(limits.input_runs) || input_bytes > limits.input_bytes { + return Err(MetadataError::Limit.into()); + } + builder = copy_run(builder, &base, stored).await?; + base.live_lease()?; + } + // Summaries cover one disjoint run set, not the union of overlapping + // roots. Per-file canonical folding and the merged spool establish + // that union independently. + if runs != root.record_count + || objects != root.object_count + || first != Some(root.first_key) + || last != Some(root.last_key) + { + return Err(CatalogPreparationError::Integrity); + } + } + let (output, descriptor, edge_count) = + finish_output(builder, budget, &base, limits.output).await?; + let selected = Selection::Ingress(selected); + let catalog = replacement(&base, &selected, output).await?; + base.live_lease()?; + Ok(Self { + base, + selected, + output, + catalog, + object_count: descriptor.object_count, + edge_count, + input_count, + inputs_digest: *blake3::hash(&binding.finish()).as_bytes(), + inventory_digest: descriptor.inventory_digest, + }) + } +} + +pub(super) async fn check_admin( + base: &PreparationBaseResolver, +) -> Result<(), CatalogPreparationError> { + let (client, target, check) = base.capability(); + let sql = cellule_runtime::primitives::sql::SqlCell::::new( + client.clone(), + target.clone(), + ) + .map_err(|_| PreparationBaseError::Context)?; + let role = sql + .query( + None, + SqlBatch { + statements: vec![access_statement(&check.actor)], + }, + ) + .await + .map_err(|_| PreparationBaseError::Inactive)?; + if !decode_access(&role.output) + .map_err(|_| PreparationBaseError::Context)? + .is_some_and(|role| role >= TokenScope::Admin) + { + return Err(PreparationBaseError::Inactive.into()); + } + Ok(()) +} +pub(super) async fn new_builder( + root: std::path::PathBuf, + budget: DiskBudget, + base: &PreparationBaseResolver, + limits: MetadataLimits, +) -> Result { + let context = base.context(); + tokio::task::spawn_blocking(move || { + let workspace = Arc::new( + tempfile::Builder::new() + .prefix("canopy-compaction-") + .tempdir_in(root)?, + ); + let mut builder = DirectoryBuilder::new( + workspace.path(), + budget, + context.repository, + context.operation, + context.format, + limits, + )?; + builder.retain_workspace(workspace); + Ok::<_, MetadataError>(builder) + }) + .await? + .map_err(Into::into) +} +pub(super) async fn copy_run( + builder: DirectoryBuilder, + base: &PreparationBaseResolver, + stored: crate::packs::directory::StoredRun, +) -> Result { + let run = RunLoader::load(&*base.files(), stored).await?; + tokio::task::spawn_blocking(move || { + let mut builder = builder; + if run.descriptor() != stored.run { + return Err(MetadataError::Integrity); + } + builder.add_coverage(&run, stored.coverage)?; + Ok::<_, MetadataError>(builder) + }) + .await? + .map_err(Into::into) +} +pub(super) async fn finish_output( + builder: DirectoryBuilder, + budget: DiskBudget, + base: &PreparationBaseResolver, + limits: MetadataLimits, +) -> Result<(NodeRef, crate::packs::directory::RunDescriptor, u64), CatalogPreparationError> { + let indexes = base.indexes(); + let context = base.context(); + let (mut partitioner, descriptor, edge_count) = tokio::task::spawn_blocking(move || { + let (run, edges) = builder.seal_with_edges()?; + let descriptor = run.descriptor(); + Ok::<_, MetadataError>(( + DirectoryPartitioner::new(Arc::new(run), budget, limits)?, + descriptor, + edges, + )) + }) + .await??; + let store = indexes.store(); + let mut output = None; + loop { + let (next, retained) = tokio::task::spawn_blocking(move || { + let next = partitioner.next_run()?; + Ok::<_, MetadataError>((next, partitioner)) + }) + .await??; + partitioner = retained; + let Some(run) = next else { + break; + }; + output = Some( + indexes + .ranges() + .insert(output, context.operation, run.upload(&store).await?) + .await?, + ); + base.live_lease()?; + } + let output = output.ok_or(CatalogPreparationError::Integrity)?; + if output.object_count != descriptor.object_count + || output.first_key != descriptor.first_oid + || output.last_key != descriptor.last_oid + { + return Err(CatalogPreparationError::Integrity); + } + Ok((output, descriptor, edge_count)) +} diff --git a/crates/canopy-server/src/packs/publication/compaction/publish.rs b/crates/canopy-server/src/packs/publication/compaction/publish.rs new file mode 100644 index 0000000..3ac6d35 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/compaction/publish.rs @@ -0,0 +1,261 @@ +use super::*; +use crate::packs::publication::{ + commands::{authorized, check_pin, fact, load, matched}, + publish::{authenticate, changed, checkpoint, retention_matches}, + sql::*, +}; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct PublishedCompaction { + pub generation: u64, + pub certificate_digest: [u8; 32], +} +impl WireValue for PublishedCompaction { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + if self.generation == 0 || self.generation > i64::MAX as u64 { + return Err(CodecError::Invalid("invalid compaction generation")); + } + e.write_u64(self.generation)?; + e.write_bytes(&self.certificate_digest) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + generation: d.read_u64()?, + certificate_digest: crate::packs::directory::index::codec::fixed(d)?, + }; + let mut e = BoundedEncoder::new(128)?; + value.encode(&mut e)?; + Ok(value) + } +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum CompactionReply { + Published(PublishedCompaction), + Denied(PreparationDenial), +} +impl WireValue for CompactionReply { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + match self { + Self::Published(value) => { + e.write_u8(0)?; + value.encode(e) + } + Self::Denied(reason) => PreparationReply::Denied(*reason).encode(e), + } + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let tag = d.read_u8()?; + if tag == 0 { + return Ok(Self::Published(PublishedCompaction::decode(d)?)); + } + let reason = match tag { + 1 => PreparationDenial::Unauthorized, + 2 => PreparationDenial::Conflict, + 3 => PreparationDenial::Stale, + 4 => PreparationDenial::Expired, + 5 => PreparationDenial::Capacity, + 6 => PreparationDenial::Missing, + _ => return Err(CodecError::Invalid("invalid compaction reply")), + }; + Ok(Self::Denied(reason)) + } +} +fn denied(reason: PreparationDenial) -> CommandResult { + CommandResult::Rejected(CompactionReply::Denied(reason)) +} +fn verification(data: &certificate::CertificateData) -> [u8; 32] { + let mut hash = blake3::Hasher::new(); + hash.update(b"canopy.compaction-verification.v1\0"); + hash.update(&data.inputs_digest); + hash.update(&data.inventory_digest); + for count in [data.object_count, data.edge_count, data.input_count] { + hash.update(&count.to_be_bytes()); + } + *hash.finalize().as_bytes() +} +fn saved_result(bytes: &[u8]) -> Result { + let mut d = BoundedDecoder::new(bytes, 128)?; + let value = PublishedCompaction::decode(&mut d)?; + d.finish()?; + Ok(value) +} +pub struct PublishCatalogCompaction; +impl Command for PublishCatalogCompaction { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 22; + const CODEC_VERSION: u32 = 1; + type Input = CatalogCertificate; + type Output = CompactionReply; + fn execute( + context: &mut CommandContext<'_, '_>, + certificate: Self::Input, + ) -> cellule_runtime::Result> { + let Some((data, key)) = authenticate(context, &certificate, None, None)? else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + if !data.compaction || data.object_count == 0 || data.input_count == 0 { + return Ok(denied(PreparationDenial::Unauthorized)); + } + let Some(format) = authorized( + context, + data.token.repository, + &data.actor, + TokenScope::Admin, + )? + else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + let saved = context.sql(&statement("SELECT actor,request_digest,verification_digest,result FROM catalog_compactions WHERE id=?1", vec![blob(data.token.operation)]))?; + if let Some(row) = rows(&saved)?.first() { + let [ + SqlValue::Text(actor), + digest, + binding, + SqlValue::Blob(result), + ] = row.as_slice() + else { + return Err(Error::Command("invalid compaction outcome")); + }; + if *actor != data.actor + || fixed::<32>(digest)? != data.token.request_digest + || fixed::<32>(binding)? != verification(&data) + { + return Ok(denied(PreparationDenial::Conflict)); + } + return Ok(CommandResult::Success(CompactionReply::Published( + saved_result(result)?, + ))); + } + if !rows(&context.sql(&statement( + "SELECT id FROM pushes WHERE id=?1", + vec![blob(data.token.operation)], + ))?)? + .is_empty() + { + return Ok(denied(PreparationDenial::Conflict)); + } + if data.token.owner != context.owner_fence() { + return Ok(denied(PreparationDenial::Stale)); + } + let Some(row) = load(context, data.token)? else { + return Ok(denied(PreparationDenial::Missing)); + }; + if !matched( + &row, + &LeaseCheck { + token: data.token, + actor: data.actor.clone(), + }, + ) { + return Ok(denied(PreparationDenial::Stale)); + } + if row.expires <= now(context.now_ms())? { + return Ok(denied(PreparationDenial::Expired)); + } + check_pin(context, &row)?; + if format != data.catalog.format + || !retention_matches(context, &data, row.generation, format)? + || fact(context, data.token.repository, format, None)? != data.base + { + return Ok(denied(PreparationDenial::Conflict)); + } + let counts = context.sql(&statement( + "SELECT count(*) FROM (SELECT generation FROM catalog_generations LIMIT ?1)", + vec![number(MAX_RETAINED_GENERATIONS)?], + ))?; + let Some([count]) = rows(&counts)?.first().map(Vec::as_slice) else { + return Err(Error::Command("missing compaction generation count")); + }; + if unsigned(count)? >= MAX_RETAINED_GENERATIONS || data.base.generation >= i64::MAX as u64 { + return Ok(denied(PreparationDenial::Capacity)); + } + let Some(missing) = checkpoint(context, &data, &key)? else { + return Ok(denied(PreparationDenial::Conflict)); + }; + let bytes = certificate.bytes()?; + let digest = *blake3::hash(&bytes).as_bytes(); + if row.expires <= now(context.now_ms())? { + return Ok(denied(PreparationDenial::Expired)); + } + // No rejected result after writes. Every later error aborts the entire + // Cell transaction; the SDK's selected durability gate controls ACK. + if missing { + changed(context.sql(&statement("UPDATE catalog_operations SET attestation=?1,attestation_digest=?2 WHERE id=?3 AND attestation IS NULL", vec![blob(&bytes),blob(digest),blob(data.token.operation)]))?)?; + changed(context.sql(&statement("UPDATE catalog_leases SET attestation=?1,attestation_digest=?2 WHERE incarnation=?3 AND admission_sequence=?4 AND attestation IS NULL", vec![blob(&bytes),blob(digest),blob(data.token.owner.incarnation.as_bytes()),number(data.token.attempt)?]))?)?; + } + let generation = data.base.generation + 1; + let mut encoded = BoundedEncoder::new(256)?; + data.catalog.encode(&mut encoded)?; + let mut refs = BoundedEncoder::new(128)?; + if let Some(root) = data.base.refs { + root.encode(&mut refs)?; + } + changed(context.sql(&statement( + "INSERT INTO catalog_generations(generation,catalog,certificate,refs) VALUES(?1,?2,?3,?4)", + vec![ + number(generation)?, blob(encoded.finish()), blob(digest), + data.base.refs.map_or(SqlValue::Null, |_| blob(refs.finish())), + ], + ))?)?; + changed(context.sql(&statement( + "UPDATE catalog_state SET generation=?1 WHERE singleton=1 AND generation=?2", + vec![number(generation)?, number(data.base.generation)?], + ))?)?; + let result = PublishedCompaction { + generation, + certificate_digest: digest, + }; + let mut encoded = BoundedEncoder::new(128)?; + result.encode(&mut encoded)?; + changed(context.sql(&statement("INSERT INTO catalog_compactions(id,actor,request_digest,verification_digest,result) VALUES(?1,?2,?3,?4,?5)", vec![blob(data.token.operation),SqlValue::Text(data.actor.clone()),blob(data.token.request_digest),blob(verification(&data)),blob(encoded.finish())]))?)?; + changed(context.sql(&statement( + "DELETE FROM catalog_operations WHERE id=?1", + vec![blob(data.token.operation)], + ))?)?; + Ok(CommandResult::Success(CompactionReply::Published(result))) + } +} +/// Read-only logical recovery/preflight. This cannot allocate a namespace, +/// return a new mutation receipt, or grant retention/deletion authority. +pub struct CheckCompletedCompaction; +impl Query for CheckCompletedCompaction { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 23; + const CODEC_VERSION: u32 = 1; + type Input = BeginRequest; + type Output = Option; + fn execute( + context: &mut QueryContext<'_>, + input: Self::Input, + ) -> cellule_runtime::Result { + validate_component(&input.actor)?; + if !decode_access(&context.sql(&SqlBatch { + statements: vec![access_statement(&input.actor)], + })?)? + .is_some_and(|role| role >= TokenScope::Admin) + || identity( + &context.sql(&statement(IDENTITY, vec![]))?, + input.repository, + )? + .is_none() + { + return Ok(Some(CompactionReply::Denied( + PreparationDenial::Unauthorized, + ))); + } + let saved = context.sql(&statement( + "SELECT actor,request_digest,result FROM catalog_compactions WHERE id=?1", + vec![blob(input.operation)], + ))?; + let Some(row) = rows(&saved)?.first() else { + return Ok(None); + }; + let [SqlValue::Text(actor), digest, SqlValue::Blob(result)] = row.as_slice() else { + return Err(Error::Command("invalid compaction lookup")); + }; + if *actor != input.actor || fixed::<32>(digest)? != input.request_digest { + return Ok(Some(CompactionReply::Denied(PreparationDenial::Conflict))); + } + Ok(Some(CompactionReply::Published(saved_result(result)?))) + } +} diff --git a/crates/canopy-server/src/packs/publication/compaction/range.rs b/crates/canopy-server/src/packs/publication/compaction/range.rs new file mode 100644 index 0000000..e9a9842 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/compaction/range.rs @@ -0,0 +1,338 @@ +use super::*; +use crate::{ + ObjectId, + packs::directory::{ + StoredRun, + index::{IndexError, RangeIndex, codec::write_run}, + snapshot::{DirectorySnapshot, MAX_LEVELS, RunLoader}, + }, +}; + +/// Caller chooses a catalog position, never supplies authoritative descriptors. +/// Ingress is promoted to level 0; a level is promoted to its adjacent successor. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum CompactionSource { + Ingress(usize), + Level(usize), +} +#[derive(Clone)] +enum Source { + Ingress(NodeRef), + Level(usize), +} +#[derive(Clone)] +pub(super) struct RangeSelection { + source: Source, + run: StoredRun, + portion: StoredRun, + remainder: Option, + target: usize, + overlaps: Vec, +} +impl PreparedCompaction { + /// Move one source run plus every intersecting target run into the next + /// level. If overlaps exceed the input budget, move a verified source prefix + /// and retain its exact verified suffix in the same physical file. `after` + /// is the exclusive last OID of a previously processed source portion; None starts at the first run. None output means the source is empty + /// or exhausted. A source/target file exceeding the physical input budget + /// rejects; selected target windows never truncate coverage silently. + /// Unchanged disjoint files are verified and promoted without rewriting. + pub async fn prepare_range( + root: &Path, + budget: DiskBudget, + base: Arc, + source: CompactionSource, + after: Option, + limits: CompactionLimits, + ) -> Result, CatalogPreparationError> { + limits.validate()?; + let (_, deadline) = base.live_lease()?; + let root = root.to_owned(); + timeout_at(deadline, async { + prepare::check_admin(&base).await?; + let (directory, _) = base.catalog_parts(); + directory.validate()?; + let (source, source_root, target) = match source { + CompactionSource::Ingress(at) => { + let selected = *directory.level_zero.get(at).ok_or(IndexError::Stale)?; + (Source::Ingress(selected), Some(selected), 0) + } + CompactionSource::Level(at) if at < MAX_LEVELS - 1 => ( + Source::Level(at), + directory.levels.get(at).copied().flatten(), + at + 1, + ), + CompactionSource::Level(_) => return Err(IndexError::Limit.into()), + }; + let indexes = base.indexes(); + let mut cursor = indexes.ranges().cursor(source_root, after)?; + let Some(run) = cursor.next().await? else { + return Ok(None); + }; + // Include the source in both budgets before selecting targets. The + // interval seek includes target ranges enclosing either endpoint. + if run.run.size > limits.input_bytes { + return Err(MetadataError::Limit.into()); + } + let (overlaps, through) = select_window( + indexes.ranges(), + directory.levels.get(target).copied().flatten(), + run, + limits, + ) + .await?; + let file = base.files().load(run).await?; + let (portion, remainder, portion_edges) = tokio::task::spawn_blocking(move || { + if file.descriptor() != run.run { + return Err(MetadataError::Integrity); + } + let (left, right, edges) = file.split_coverage(run.coverage, through)?; + let portion = StoredRun { + coverage: left.ok_or(MetadataError::Integrity)?, + ..run + }; + let remainder = right.map(|coverage| StoredRun { coverage, ..run }); + Ok::<_, MetadataError>((portion, remainder, edges)) + }) + .await??; + let selection = RangeSelection { + source, + run, + portion, + remainder, + target, + overlaps, + }; + let input_count = selection.overlaps.len() as u64 + 1; + let inputs_digest = + selection.digest(base.context().repository, base.context().format)?; + let (output, descriptor, edge_count) = + if selection.overlaps.is_empty() && run.run.size <= limits.output.max_file_bytes { + let output = indexes + .ranges() + .insert(None, base.context().operation, portion) + .await?; + (output, portion.coverage, portion_edges) + } else { + let mut builder = + prepare::new_builder(root, budget.clone(), &base, limits.spool).await?; + for input in std::iter::once(&portion).chain(selection.overlaps.iter()) { + builder = prepare::copy_run(builder, &base, *input).await?; + base.live_lease()?; + } + let (output, descriptor, edges) = + prepare::finish_output(builder, budget, &base, limits.output).await?; + (output, descriptor.coverage(), edges) + }; + let selected = Selection::Range(Box::new(selection)); + let catalog = replacement(&base, &selected, output).await?; + base.live_lease()?; + Ok(Some(Self { + base, + selected, + output, + catalog, + object_count: descriptor.object_count, + edge_count, + input_count, + inputs_digest, + inventory_digest: descriptor.inventory_digest, + })) + }) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } +} +impl RangeSelection { + pub(super) fn last_moved(&self) -> ObjectId { + self.portion.coverage.last_oid + } + fn digest(&self, repository: [u8; 16], format: ObjectFormat) -> Result<[u8; 32], CodecError> { + let mut hash = blake3::Hasher::new(); + let mut e = BoundedEncoder::new(1024)?; + e.write_bytes(b"canopy.range-compaction.v2\0")?; + e.write_bytes(&repository)?; + e.write_u8(format.bytes() as u8)?; + match self.source { + Source::Ingress(root) => { + e.write_u8(0)?; + reference(&mut e, root)?; + } + Source::Level(at) => { + e.write_u8(1)?; + e.write_u8(at as u8)?; + } + } + e.write_u8(self.target as u8)?; + e.write_count(self.overlaps.len())?; + write_run(&mut e, self.run)?; + hash.update(&e.finish()); + let mut e = BoundedEncoder::new(1024)?; + e.write_bool(self.remainder.is_some())?; + if let Some(remainder) = self.remainder { + write_run(&mut e, remainder)?; + } + hash.update(&e.finish()); + for run in std::iter::once(&self.portion).chain(self.overlaps.iter()) { + let mut e = BoundedEncoder::new(1024)?; + write_run(&mut e, *run)?; + hash.update(&e.finish()); + } + Ok(*hash.finalize().as_bytes()) + } + pub(super) async fn replace( + &self, + directory: &mut DirectorySnapshot, + index: &RangeIndex, + operation: [u8; 16], + output: NodeRef, + ) -> Result<(), IndexError> { + directory.validate()?; + let source_at = match self.source { + Source::Ingress(root) => directory + .level_zero + .iter() + .position(|candidate| *candidate == root) + .ok_or(IndexError::Stale)?, + Source::Level(at) => at, + }; + let source_root = match self.source { + Source::Ingress(root) => Some(root), + Source::Level(at) => directory.levels.get(at).copied().flatten(), + }; + if index.find(source_root, self.run.coverage.first_oid).await? != Some(self.run) { + return Err(IndexError::Stale); + } + let first = self + .overlaps + .iter() + .map(|run| run.coverage.first_oid) + .chain(std::iter::once(self.portion.coverage.first_oid)) + .min() + .ok_or(IndexError::Integrity)?; + let last = self + .overlaps + .iter() + .map(|run| run.coverage.last_oid) + .chain(std::iter::once(self.portion.coverage.last_oid)) + .max() + .ok_or(IndexError::Integrity)?; + if output.first_key != first || output.last_key != last { + return Err(IndexError::Integrity); + } + index.validate_root(output).await?; + let target_root = directory.levels.get(self.target).copied().flatten(); + let current = index + .overlapping(target_root, first, last, self.overlaps.len()) + .await; + // Any newly inserted overlapping target, changed preferred placement or + // missing selected target invalidates reuse. Out-of-range path updates + // remain eligible; never rebuild a level from an old root. + match current { + Ok(runs) if runs == self.overlaps => {} + Ok(_) | Err(IndexError::Limit) => return Err(IndexError::Stale), + Err(error) => return Err(error), + } + let mut remaining = index.remove(source_root, operation, self.run).await?; + if let Some(remainder) = self.remainder { + remaining = Some(index.insert(remaining, operation, remainder).await?); + } + match self.source { + Source::Ingress(_) => { + if let Some(root) = remaining { + directory.level_zero[source_at] = root; + } else { + directory.level_zero.remove(source_at); + } + } + Source::Level(_) => directory.levels[source_at] = remaining, + } + let mut target = target_root; + for run in &self.overlaps { + target = index.remove(target, operation, *run).await?; + } + if target.is_none() { + // The complete verified output tree already has the right context. + // Reuse it rather than reuploading identical intermediate nodes. + target = Some(output); + } else { + let mut cursor = index.cursor(Some(output), None)?; + while let Some(run) = cursor.next().await? { + target = Some(index.insert(target, operation, run).await?); + } + } + if directory.levels.len() <= self.target { + directory.levels.resize(self.target + 1, None); + } + directory.levels[self.target] = target; + directory.validate() + } +} + +/// A bounded consecutive target prefix. A later target outside the returned +/// interval is never read/copied or removed by this operation. +async fn select_window( + index: &RangeIndex, + target: Option, + source: StoredRun, + limits: CompactionLimits, +) -> Result<(Vec, ObjectId), CatalogPreparationError> { + let mut selected = Vec::new(); + let mut bytes = source.run.size; + let mut physical = std::collections::BTreeMap::new(); + physical.insert( + (source.run.operation, source.artifact.digest), + (source.run, source.artifact), + ); + let mut next = index.successor(target, source.coverage.first_oid).await?; + let mut cursor = if let Some(first) = next { + Some(index.cursor(target, Some(first.coverage.first_oid))?) + } else { + None + }; + while let Some(run) = next { + if run.coverage.first_oid > source.coverage.last_oid { + break; + } + let key = (run.run.operation, run.artifact.digest); + let extra = match physical.get(&key) { + Some(descriptor) if *descriptor == (run.run, run.artifact) => 0, + Some(_) => return Err(MetadataError::Integrity.into()), + None => run.run.size, + }; + let total = bytes.checked_add(extra).ok_or(MetadataError::Limit)?; + if selected.len() + 1 >= limits.input_runs as usize || total > limits.input_bytes { + let through = if let Some(last) = selected.last() { + let last: &StoredRun = last; + last.coverage.last_oid.min(source.coverage.last_oid) + } else if run.coverage.first_oid > source.coverage.first_oid { + predecessor(run.coverage.first_oid)? + } else { + return Err(MetadataError::Limit.into()); + }; + return Ok((selected, through)); + } + bytes = total; + physical.insert(key, (run.run, run.artifact)); + selected.push(run); + if run.coverage.last_oid >= source.coverage.last_oid { + break; + } + next = match &mut cursor { + Some(cursor) => cursor.next().await?, + None => None, + }; + } + Ok((selected, source.coverage.last_oid)) +} +fn predecessor(oid: ObjectId) -> Result { + let mut bytes = oid.to_vec(); + for byte in bytes.iter_mut().rev() { + if *byte != 0 { + *byte -= 1; + return bytes.try_into().map_err(|_| MetadataError::Integrity); + } + *byte = 255; + } + Err(MetadataError::Integrity) +} diff --git a/crates/canopy-server/src/packs/publication/compaction/schedule.rs b/crates/canopy-server/src/packs/publication/compaction/schedule.rs new file mode 100644 index 0000000..b1a4e95 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/compaction/schedule.rs @@ -0,0 +1,224 @@ +//! Bounded advisory selection. Catalog-derived preparation and publication +//! remain the authority; local rotation state grants no retention/delete rights. +use super::*; +use crate::{ + ObjectId, + packs::directory::{ + index::IndexError, + snapshot::{DirectorySnapshot, MAX_LEVELS}, + }, +}; + +#[derive(Clone, Copy, Debug)] +pub struct CompactionPolicy { + /// Logical object target for the first nonoverlapping level. + pub base_objects: u64, + pub level_ratio: u32, + /// Begin urgent ingress dispatch at this many overlapping roots. + pub ingress_high_water: usize, + /// At most this many urgent jobs before an eligible higher-level job. + pub urgent_burst: u8, +} +impl Default for CompactionPolicy { + fn default() -> Self { + Self { + base_objects: 1 << 18, + level_ratio: 4, + ingress_high_water: 8, + urgent_burst: 3, + } + } +} +impl CompactionPolicy { + fn targets(self) -> Result<[u64; MAX_LEVELS], IndexError> { + if self.base_objects == 0 + || self.base_objects > i64::MAX as u64 + || !(2..=16).contains(&self.level_ratio) + || !(1..=LEVEL_ZERO_ROOTS).contains(&self.ingress_high_water) + || !(1..=32).contains(&self.urgent_burst) + { + return Err(IndexError::Limit); + } + let mut result = [0; MAX_LEVELS]; + let mut next = self.base_objects; + for target in &mut result { + *target = next; + next = next + .saturating_mul(u64::from(self.level_ratio)) + .min(i64::MAX as u64); + } + Ok(result) + } + /// Logical counts are exact within a disjoint level, but may overlap counts + /// in other levels. They are scheduling pressure, never a global inventory. + pub fn pressure(self, directory: &DirectorySnapshot) -> Result { + directory.validate()?; + let targets = self.targets()?; + let mut objects = [0; MAX_LEVELS]; + for (at, root) in directory.levels.iter().enumerate() { + objects[at] = root.map_or(0, |root| root.object_count); + } + // The final level has no adjacent successor. Never silently select a + // nonexistent level or claim that this pressure has been drained. + if objects[MAX_LEVELS - 1] > targets[MAX_LEVELS - 1] { + return Err(IndexError::Limit); + } + Ok(CompactionPressure { + ingress_roots: directory.level_zero.len(), + level_objects: objects, + level_targets: targets, + }) + } +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct CompactionPressure { + pub ingress_roots: usize, + pub level_objects: [u64; MAX_LEVELS], + pub level_targets: [u64; MAX_LEVELS], +} + +/// One repository's local advisory maintenance cursor. O(MAX_LEVELS) state; +/// restart resets rotation, without changing authority or dropping catalog work. +/// Selection progresses after successful private preparation, not a durable ACK. +/// The caller must retain/recover uncertain publication before releasing inputs. +pub struct CompactionPlanner { + policy: CompactionPolicy, + context: Option<([u8; 16], ObjectFormat)>, + next_class: usize, + next_ingress: usize, + urgent_streak: u8, + after: [Option; MAX_LEVELS - 1], +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +struct Choice { + source: CompactionSource, + class: usize, + urgent_streak: u8, +} +impl CompactionPlanner { + pub fn new(policy: CompactionPolicy) -> Result { + policy.targets()?; + Ok(Self { + policy, + context: None, + next_class: 0, + next_ingress: 0, + urgent_streak: 0, + after: [None; MAX_LEVELS - 1], + }) + } + fn choose(&self, directory: &DirectorySnapshot) -> Result, IndexError> { + let pressure = self.policy.pressure(directory)?; + if self + .context + .is_some_and(|context| context != (directory.repository, directory.format)) + { + return Err(IndexError::Integrity); + } + let level_ready = |at: usize| pressure.level_objects[at] > pressure.level_targets[at]; + let higher_ready = (0..MAX_LEVELS - 1).any(level_ready); + let ingress = || CompactionSource::Ingress(self.next_ingress % pressure.ingress_roots); + if pressure.ingress_roots >= self.policy.ingress_high_water + && (!higher_ready || self.urgent_streak < self.policy.urgent_burst) + { + return Ok(Some(Choice { + source: ingress(), + class: 0, + urgent_streak: self.urgent_streak.saturating_add(1), + })); + } + // Class zero is ingress; classes 1..MAX_LEVELS are promotable levels. + // After a full urgent burst, give one eligible level a turn even if + // ingress occupies the current rotation position. + for offset in 0..MAX_LEVELS { + let class = (self.next_class + offset) % MAX_LEVELS; + let source = if class == 0 { + if pressure.ingress_roots == 0 + || (higher_ready && self.urgent_streak >= self.policy.urgent_burst) + { + continue; + } + ingress() + } else if level_ready(class - 1) { + CompactionSource::Level(class - 1) + } else { + continue; + }; + return Ok(Some(Choice { + source, + class, + urgent_streak: 0, + })); + } + Ok(None) + } + fn advance(&mut self, directory: &DirectorySnapshot, choice: Choice, through: ObjectId) { + self.context = Some((directory.repository, directory.format)); + if choice.urgent_streak == 0 { + self.next_class = (choice.class + 1) % MAX_LEVELS; + } + self.urgent_streak = choice.urgent_streak; + match choice.source { + CompactionSource::Ingress(at) => self.next_ingress = at + 1, + CompactionSource::Level(at) => self.after[at] = Some(through), + } + } + /// Select geometric pressure from this queried base and prepare one verified + /// bounded window. No history scan or raw caller-supplied run descriptor. + /// Tail ingress is eligible even below the urgent high-water mark. + pub async fn prepare_next( + &mut self, + root: &Path, + budget: DiskBudget, + base: Arc, + limits: CompactionLimits, + ) -> Result, CatalogPreparationError> { + limits.validate()?; + let (_, deadline) = base.live_lease()?; + timeout_at(deadline, async { + let (directory, _) = base.catalog_parts(); + let Some(choice) = self.choose(&directory)? else { + prepare::check_admin(&base).await?; + return Ok(None); + }; + let after = match choice.source { + CompactionSource::Ingress(_) => None, + CompactionSource::Level(at) => self.after[at], + }; + let mut prepared = PreparedCompaction::prepare_range( + root, + budget.clone(), + Arc::clone(&base), + choice.source, + after, + limits, + ) + .await?; + if prepared.is_none() && after.is_some() { + // Wrap only after indexed exhaustion, so newly inserted lower + // ranges and retained prefixes cannot be skipped indefinitely. + prepared = PreparedCompaction::prepare_range( + root, + budget, + base, + choice.source, + None, + limits, + ) + .await?; + } + let prepared = prepared.ok_or(IndexError::Integrity)?; + let Selection::Range(selection) = &prepared.selected else { + return Err(IndexError::Integrity.into()); + }; + self.advance(&directory, choice, selection.last_moved()); + Ok(Some(prepared)) + }) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/canopy-server/src/packs/publication/compaction/schedule/tests.rs b/crates/canopy-server/src/packs/publication/compaction/schedule/tests.rs new file mode 100644 index 0000000..78a4a98 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/compaction/schedule/tests.rs @@ -0,0 +1,164 @@ +use super::*; +use canopy_object_storage::artifact::ArtifactDescriptor; + +fn root(format: ObjectFormat, operation: u8, objects: u64) -> NodeRef { + let mut first = vec![0; format.bytes()]; + first[0] = 1; + let mut last = first.clone(); + last[0] = 2; + NodeRef { + operation: [operation; 16], + artifact: ArtifactDescriptor { + size: 1024, + digest: [operation; 32], + manifest_digest: [3; 32], + }, + height: 0, + first_key: first.try_into().unwrap(), + last_key: last.try_into().unwrap(), + record_count: 1, + object_count: objects, + } +} + +#[test] +fn geometric_targets_reject_invalid_profiles_and_final_level_overflow() { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let policy = CompactionPolicy { + base_objects: 2, + level_ratio: 2, + ..CompactionPolicy::default() + }; + let mut directory = DirectorySnapshot::empty([1; 16], format); + directory.levels = vec![Some(root(format, 1, 2)), Some(root(format, 2, 4))]; + let pressure = policy.pressure(&directory).unwrap(); + assert_eq!(&pressure.level_targets[..4], &[2, 4, 8, 16]); + assert!( + CompactionPlanner::new(policy) + .unwrap() + .choose(&directory) + .unwrap() + .is_none() + ); + directory.levels.resize(MAX_LEVELS, None); + directory.levels[MAX_LEVELS - 1] = Some(root(format, 3, (2 << (MAX_LEVELS - 1)) + 1)); + assert!(matches!( + policy.pressure(&directory), + Err(IndexError::Limit) + )); + for bad in [ + CompactionPolicy { + base_objects: 0, + ..policy + }, + CompactionPolicy { + base_objects: u64::MAX, + ..policy + }, + CompactionPolicy { + level_ratio: 1, + ..policy + }, + CompactionPolicy { + level_ratio: 17, + ..policy + }, + CompactionPolicy { + ingress_high_water: 0, + ..policy + }, + CompactionPolicy { + ingress_high_water: LEVEL_ZERO_ROOTS + 1, + ..policy + }, + CompactionPolicy { + urgent_burst: 0, + ..policy + }, + CompactionPolicy { + urgent_burst: 33, + ..policy + }, + ] { + assert!(CompactionPlanner::new(bad).is_err()); + } + let saturated = CompactionPolicy { + base_objects: i64::MAX as u64 / 2, + level_ratio: 16, + ..policy + } + .targets() + .unwrap(); + assert!( + saturated[1..] + .iter() + .all(|target| *target == i64::MAX as u64) + ); + } +} + +#[test] +fn urgent_ingress_cannot_starve_any_pressured_level_and_roots_rotate() { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let policy = CompactionPolicy { + base_objects: 1, + level_ratio: 2, + ingress_high_water: 2, + urgent_burst: 3, + }; + let mut planner = CompactionPlanner::new(policy).unwrap(); + let mut directory = DirectorySnapshot::empty([1; 16], format); + directory.level_zero = (1..=5) + .map(|operation| root(format, operation, 10)) + .collect(); + directory.levels = (0..MAX_LEVELS - 1) + .map(|at| Some(root(format, at as u8 + 10, (1 << at) + 1))) + .collect(); + let through = directory.level_zero[0].first_key; + let mut levels = Vec::new(); + let mut ingress = Vec::new(); + let mut streak = 0; + for _ in 0..4 * (MAX_LEVELS - 1) { + let choice = planner.choose(&directory).unwrap().unwrap(); + match choice.source { + CompactionSource::Ingress(at) => { + ingress.push(at); + streak += 1; + assert!(streak <= 3); + } + CompactionSource::Level(at) => { + levels.push(at); + streak = 0; + } + } + planner.advance(&directory, choice, through); + } + assert_eq!(levels, (0..MAX_LEVELS - 1).collect::>()); + assert!(ingress.iter().enumerate().all(|(n, at)| *at == n % 5)); + let mut foreign = directory.clone(); + foreign.repository = [2; 16]; + assert!(matches!( + planner.choose(&foreign), + Err(IndexError::Integrity) + )); + foreign.repository = directory.repository; + foreign.format = if format == ObjectFormat::Sha1 { + ObjectFormat::Sha256 + } else { + ObjectFormat::Sha1 + }; + // Use a structurally valid foreign-format snapshot, so this assertion + // tests planner context binding rather than malformed OID widths. + foreign.level_zero = vec![root(foreign.format, 1, 10)]; + foreign.levels.clear(); + foreign.validate().unwrap(); + assert!(planner.choose(&foreign).is_err()); + // Tail ingress is not stranded below the urgent watermark. + directory.level_zero.truncate(1); + directory.levels.clear(); + assert_eq!( + planner.choose(&directory).unwrap().unwrap().source, + CompactionSource::Ingress(0) + ); + } +} diff --git a/crates/canopy-server/src/packs/publication/completion.rs b/crates/canopy-server/src/packs/publication/completion.rs new file mode 100644 index 0000000..5a060b5 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/completion.rs @@ -0,0 +1,846 @@ +//! Exact native outcome and catalog/ref publication in one durable command. +//! Decoded payloads are untrusted; the privately prepared factory signs their +//! complete binding. Large payloads still need the immutable-root transport. +use super::*; +use super::{ + commands::{check_pin, load, matched}, + publish::{authenticate, changed}, + sql::*, +}; +use crate::{ + PushPlan, + git_http::GitHttpResponse, + push::{CHUNK_BYTES, MAX_RESPONSE_BYTES, VerifiedPushCertificate}, +}; +use cellule_ltx::DiskBudget; +use cellule_runtime::{ + CellClient, CellTarget, Committed, InvocationError, MutationIdentity, Receipt, + primitives::sql::SqlCell, +}; +use sha2::{Digest as _, Sha256}; +use std::path::Path; +use tokio::time::timeout_at; + +const JSON_BYTES: usize = 64 << 10; + +/// Input from the native preparation service. Signed bytes can only arrive via +/// the gateway's opaque native-verified witness, not a decoded annotation DTO. +pub struct PushCompletionRequest { + pub plan: Option, + pub response: GitHttpResponse, + pub options: Vec, + pub certificate: Option, +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum CompletionCatalogProof { + Refs(RefPublicationProof), + OutcomeOnly(OutcomeCertificate), +} +impl CompletionCatalogProof { + fn refs_digest(&self) -> Result, CodecError> { + match self { + Self::Refs(proof) => Ok(Some(super::ref_proof::binding( + &proof.plan, + &proof.ancestry, + )?)), + Self::OutcomeOnly(_) => Ok(None), + } + } +} +/// Transport annotation only. Raw bytes cannot be converted into a trusted +/// native witness; edits invalidate the completion certificate. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct SignedPushAnnotation> { + pub body: B, + pub signer: String, + pub key: String, +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct CatalogPushCompletion { + pub proof: CompletionCatalogProof, + pub response_id: [u8; 16], + pub response: GitHttpResponse, + pub options: Vec, + pub signed: Option, +} +#[derive(Debug, thiserror::Error)] +pub enum PushCompletionProofError { + #[error("push completion ref proof failed")] + Refs(#[from] RefProofError), + #[error("push completion lease is inactive")] + Base(#[from] PreparationBaseError), + #[error("push completion attestation failed")] + Attestation(#[from] CatalogAttestationError), + #[error("push completion payload is invalid")] + Codec(#[from] CodecError), + #[error("push completion command failed")] + Command(#[source] Box>), + #[error("native report does not match the completion plan")] + Report(#[from] crate::push::PushError), +} +impl PreparedCatalog { + /// The command is the acknowledgement boundary. Do not put a local lease + /// timeout around it: cancellation after admission can be an uncertain + /// outcome, which must retain the logical ID and resolve through replay. + pub async fn complete_push( + &self, + identity: MutationIdentity, + request: PushCompletionRequest, + root: &Path, + budget: DiskBudget, + limits: crate::packs::metadata::MetadataLimits, + ) -> Result, PushCompletionProofError> { + let input = Box::pin(self.push_completion(request, root, budget, limits)).await?; + self.ensure_live()?; + let (client, target, _) = self.base.capability(); + Box::pin(client.command::(target, identity, input)) + .await + .map_err(|error| PushCompletionProofError::Command(Box::new(error))) + } + + pub async fn push_completion( + &self, + request: PushCompletionRequest, + root: &Path, + budget: DiskBudget, + limits: crate::packs::metadata::MetadataLimits, + ) -> Result { + if request.plan.is_none() { + return self.base.session.push_outcome(request).await; + } + let (_, deadline) = self.base.live_lease()?; + timeout_at(deadline, async { + let signed = signed_annotation(&self.base.session, request.certificate)?; + crate::push::report::publication_matches(&request.response, request.plan.as_ref())?; + let plan = request + .plan + .ok_or(CodecError::Invalid("missing ref plan"))?; + let (plan, ancestry) = self.ref_evidence(plan, root, budget, limits).await?; + let response_id = uuid::Uuid::new_v4().into_bytes(); + let refs_digest = Some(super::ref_proof::binding(&plan, &ancestry)?); + let binding = payload_binding( + Some(&plan), + &response_id, + &request.response, + &request.options, + signed.as_ref(), + )?; + let certificate = self.issue_certificate(refs_digest, Some(binding)).await?; + let proof = CompletionCatalogProof::Refs(RefPublicationProof { + plan, + ancestry, + certificate, + }); + let input = CatalogPushCompletion { + proof, + response_id, + response: request.response, + options: request.options, + signed, + }; + self.ensure_live()?; + Ok(input) + }) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } +} + +pub(super) fn signed_annotation( + session: &PreparationSession, + certificate: Option, +) -> Result, CodecError> { + let (_, target, check) = session.capability(); + scoped_signed_annotation(target, check, certificate) +} +pub(super) fn scoped_signed_annotation( + target: &CellTarget, + check: &LeaseCheck, + certificate: Option, +) -> Result, CodecError> { + if certificate.as_ref().is_some_and(|certificate| { + certificate.target != *target + || certificate.request_digest != check.token.request_digest + || certificate.signer != check.actor + }) { + return Err(CodecError::Invalid("signed push witness context differs")); + } + Ok(certificate.map(|certificate| SignedPushAnnotation { + body: certificate.body, + signer: certificate.signer, + key: certificate.key, + })) +} + +#[derive(Debug, thiserror::Error)] +pub enum CatalogPushResponseError { + #[error("completed push response is incomplete, corrupt or foreign")] + Invalid, + #[error("completed push lookup was denied: {0:?}")] + Denied(PreparationDenial), + #[error("completed push lookup failed")] + Lookup(#[source] Box>>), + #[error("completed push response SQL capability failed")] + Capability(#[from] Error), + #[error("completed push response read failed")] + Query(#[source] Box>>), +} +impl PreparedCatalog { + /// Read the exact stored wire outcome at the command's receipt, including + /// refusals. The logical identity and actor must match this private session. + /// This intentionally does not invoke the legacy rejection rewriter. + pub async fn completed_push_response( + &self, + completed: &Committed, + ) -> Result { + self.base.session.completed_push_response(completed).await + } +} +/// Current authorized lookup and exact replay after reconnect/restart. It does +/// not allocate a preparation, namespace or durable command. Owner queries are +/// FIFO behind preceding publication; every subsequent read carries that floor. +pub async fn replay_push_response( + client: &CellClient, + target: &CellTarget, + request: BeginRequest, + minimum: Option, +) -> Result, CatalogPushResponseError> { + if target + != &crate::repository_target(target.tenant(), target.application(), request.repository)? + { + return Err(CatalogPushResponseError::Invalid); + } + let found = client + .query::(target, minimum, request.clone()) + .await + .map_err(|error| CatalogPushResponseError::Lookup(Box::new(error)))?; + match found.output { + None => Ok(None), + Some(CatalogCompletionReply::Denied(reason)) => { + Err(CatalogPushResponseError::Denied(reason)) + } + Some(CatalogCompletionReply::Completed(output)) => Ok(Some( + load_response(client, target, &request, found.receipt, output).await?, + )), + } +} +pub(super) async fn load_response( + client: &CellClient, + target: &CellTarget, + request: &BeginRequest, + minimum: Receipt, + output: CompletedCatalogPush, +) -> Result { + let sql = SqlCell::::new(client.clone(), target.clone())?; + let result = sql.query(Some(minimum), statement( + "SELECT r.status,r.headers,r.size,r.digest,count(c.part),coalesce(sum(length(c.body)),0),coalesce(min(c.part),0),coalesce(max(c.part),-1) FROM pushes p JOIN push_responses r ON p.response_id=r.id AND r.push_id=p.id LEFT JOIN push_response_chunks c ON c.response_id=r.id WHERE p.id=?1 AND p.actor=?2 AND p.request_digest=?3 AND p.response_id=?4 AND p.completion_digest IS NOT NULL GROUP BY r.id", + vec![blob(request.operation),SqlValue::Text(request.actor.clone()),blob(request.request_digest),blob(output.response_id)], + )).await.map_err(|error|CatalogPushResponseError::Query(Box::new(error)))?; + let Some( + [ + SqlValue::Integer(status), + SqlValue::Text(headers), + size, + digest, + count, + total, + first, + last, + ], + ) = rows(&result.output)?.first().map(Vec::as_slice) + else { + return Err(CatalogPushResponseError::Invalid); + }; + let size = usize::try_from(unsigned(size)?).map_err(|_| CatalogPushResponseError::Invalid)?; + let parts = size.div_ceil(CHUNK_BYTES); + if size > MAX_RESPONSE_BYTES + || unsigned(count)? != parts as u64 + || unsigned(total)? != size as u64 + || *first != SqlValue::Integer(0) + || *last != SqlValue::Integer(parts as i64 - 1) + { + return Err(CatalogPushResponseError::Invalid); + } + let headers: Vec<(String, String)> = + serde_json::from_str(headers).map_err(|_| CatalogPushResponseError::Invalid)?; + let digest = fixed::<32>(digest)?; + let mut body = Vec::with_capacity(size); + for part in 0..parts { + let result = sql + .query( + Some(minimum), + statement( + "SELECT body FROM push_response_chunks WHERE response_id=?1 AND part=?2", + vec![blob(output.response_id), number(part as u64)?], + ), + ) + .await + .map_err(|error| CatalogPushResponseError::Query(Box::new(error)))?; + let Some([SqlValue::Blob(bytes)]) = rows(&result.output)?.first().map(Vec::as_slice) else { + return Err(CatalogPushResponseError::Invalid); + }; + if bytes.len() != (size - body.len()).min(CHUNK_BYTES) { + return Err(CatalogPushResponseError::Invalid); + } + body.extend_from_slice(bytes); + } + if body.len() != size || blake3::hash(&body).as_bytes() != &digest { + return Err(CatalogPushResponseError::Invalid); + } + Ok(GitHttpResponse { + status: u16::try_from(*status).map_err(|_| CatalogPushResponseError::Invalid)?, + headers, + body, + }) +} + +fn json(value: &T) -> Result, CodecError> { + let bytes = serde_json::to_vec(value).map_err(|_| CodecError::Invalid("completion JSON"))?; + if bytes.len() > JSON_BYTES { + return Err(CodecError::Invalid("completion JSON size")); + } + Ok(bytes) +} +pub(super) fn validate_payload( + response: &GitHttpResponse, + options: &[String], + signed: Option<&SignedPushAnnotation>, +) -> Result<(), CodecError> { + if !(100..=599).contains(&response.status) + || response.body.len() > MAX_RESPONSE_BYTES + || !crate::push::valid_options(options) + || response.headers.iter().any(|(name, value)| { + axum::http::HeaderName::from_bytes(name.as_bytes()).is_err() + || axum::http::HeaderValue::from_str(value).is_err() + || (name.eq_ignore_ascii_case("Content-Length") + && value.parse::().ok() != Some(response.body.len())) + }) + || signed.as_ref().is_some_and(|signed| { + signed.body.is_empty() + || signed.body.len() > MAX_RESPONSE_BYTES + || validate_component(&signed.signer).is_err() + || signed.key.is_empty() + || signed.key.len() > 4096 + || signed.key.chars().any(char::is_control) + }) + { + return Err(CodecError::Invalid("invalid push completion payload")); + } + json(&response.headers)?; + json(&options)?; + Ok(()) +} +pub(super) fn payload_binding( + plan: Option<&PushPlan>, + response_id: &[u8; 16], + response: &GitHttpResponse, + options: &[String], + signed: Option<&SignedPushAnnotation>, +) -> Result<[u8; 32], CodecError> { + validate_payload(response, options, signed)?; + let mut hash = blake3::Hasher::new(); + hash.update(b"canopy.push-completion.v1\0"); + hash.update(&[u8::from(plan.is_some())]); + if let Some(plan) = plan { + hash.update(&super::ref_proof::plan_digest(plan)?); + } + fn bytes(hash: &mut blake3::Hasher, bytes: &[u8]) -> Result<(), CodecError> { + let mut size = BoundedEncoder::new(4)?; + size.write_count(bytes.len())?; + hash.update(&size.finish()); + hash.update(bytes); + Ok(()) + } + bytes(&mut hash, response_id)?; + hash.update(&u32::from(response.status).to_le_bytes()); + bytes(&mut hash, &json(&response.headers)?)?; + bytes(&mut hash, &response.body)?; + bytes(&mut hash, &json(&options)?)?; + hash.update(&[u8::from(signed.is_some())]); + if let Some(signed) = signed { + bytes(&mut hash, signed.signer.as_bytes())?; + bytes(&mut hash, signed.key.as_bytes())?; + bytes(&mut hash, &signed.body)?; + } + Ok(*hash.finalize().as_bytes()) +} +impl CatalogPushCompletion { + fn validate(&self) -> Result<(), CodecError> { + validate_payload(&self.response, &self.options, self.signed.as_ref()) + } + fn binding(&self) -> Result<[u8; 32], CodecError> { + let plan = match &self.proof { + CompletionCatalogProof::Refs(proof) => Some(&proof.plan), + CompletionCatalogProof::OutcomeOnly(_) => None, + }; + payload_binding( + plan, + &self.response_id, + &self.response, + &self.options, + self.signed.as_ref(), + ) + } +} +impl WireValue for CatalogPushCompletion { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.validate()?; + match &self.proof { + CompletionCatalogProof::Refs(proof) => { + e.write_u8(0)?; + proof.encode(e)?; + } + CompletionCatalogProof::OutcomeOnly(certificate) => { + e.write_u8(1)?; + certificate.encode(e)?; + } + } + e.write_bytes(&self.response_id)?; + e.write_u32(u32::from(self.response.status))?; + e.write_bytes(&json(&self.response.headers)?)?; + e.write_bytes(&self.response.body)?; + e.write_bytes(&json(&self.options)?)?; + e.write_bool(self.signed.is_some())?; + if let Some(signed) = &self.signed { + e.write_text(&signed.signer)?; + e.write_text(&signed.key)?; + e.write_bytes(&signed.body)?; + } + Ok(()) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let proof = match d.read_u8()? { + 0 => CompletionCatalogProof::Refs(RefPublicationProof::decode(d)?), + 1 => CompletionCatalogProof::OutcomeOnly(OutcomeCertificate::decode(d)?), + _ => return Err(CodecError::Invalid("completion proof kind")), + }; + let response_id = crate::packs::directory::index::codec::fixed(d)?; + let status = + u16::try_from(d.read_u32()?).map_err(|_| CodecError::Invalid("completion status"))?; + fn decode_json(bytes: &[u8]) -> Result<&[u8], CodecError> { + if bytes.len() > JSON_BYTES { + Err(CodecError::Invalid("completion JSON size")) + } else { + Ok(bytes) + } + } + let headers = serde_json::from_slice(decode_json(d.read_bytes()?)?) + .map_err(|_| CodecError::Invalid("completion headers"))?; + let body = d.read_bytes()?; + if body.len() > MAX_RESPONSE_BYTES { + return Err(CodecError::Invalid("completion body size")); + } + let body = body.to_vec(); + let options = serde_json::from_slice(decode_json(d.read_bytes()?)?) + .map_err(|_| CodecError::Invalid("completion options"))?; + let signed = if d.read_bool()? { + let signer = d.read_text()?.to_owned(); + let key = d.read_text()?.to_owned(); + let body = d.read_bytes()?; + if body.len() > MAX_RESPONSE_BYTES { + return Err(CodecError::Invalid("signed push size")); + } + Some(SignedPushAnnotation { + signer, + key, + body: body.to_vec(), + }) + } else { + None + }; + let value = Self { + proof, + response_id, + response: GitHttpResponse { + status, + headers, + body, + }, + options, + signed, + }; + value.validate()?; + Ok(value) + } +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct CompletedCatalogPush { + pub response_id: [u8; 16], + pub rejected: bool, + pub publication: Option, +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum CatalogCompletionReply { + Completed(CompletedCatalogPush), + Denied(PreparationDenial), +} +impl WireValue for CatalogCompletionReply { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + match self { + Self::Denied(reason) => PreparationReply::Denied(*reason).encode(e), + Self::Completed(value) => { + if value.rejected && value.publication.is_some() { + return Err(CodecError::Invalid("rejected push published refs")); + } + e.write_u8(0)?; + e.write_bytes(&value.response_id)?; + e.write_bool(value.rejected)?; + e.write_bool(value.publication.is_some())?; + if let Some(publication) = value.publication { + publication.encode(e)?; + } + Ok(()) + } + } + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = match d.read_u8()? { + 0 => Self::Completed(CompletedCatalogPush { + response_id: crate::packs::directory::index::codec::fixed(d)?, + rejected: d.read_bool()?, + publication: if d.read_bool()? { + Some(PublishedRefs::decode(d)?) + } else { + None + }, + }), + 1 => Self::Denied(PreparationDenial::Unauthorized), + 2 => Self::Denied(PreparationDenial::Conflict), + 3 => Self::Denied(PreparationDenial::Stale), + 4 => Self::Denied(PreparationDenial::Expired), + 5 => Self::Denied(PreparationDenial::Capacity), + 6 => Self::Denied(PreparationDenial::Missing), + _ => return Err(CodecError::Invalid("completion reply")), + }; + value.encode(&mut BoundedEncoder::new(128)?)?; + Ok(value) + } +} +fn denied(reason: PreparationDenial) -> CommandResult { + CommandResult::Rejected(CatalogCompletionReply::Denied(reason)) +} +/// Read-only preflight, using the same logical identity as Begin. A completed +/// request must return its saved response before repeating native preparation. +/// lease_ms is encoded by the reused BeginRequest but grants no lease here. +pub struct CheckCompletedPush; +impl Query for CheckCompletedPush { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 20; + const CODEC_VERSION: u32 = 1; + type Input = BeginRequest; + type Output = Option; + fn execute( + context: &mut QueryContext<'_>, + input: BeginRequest, + ) -> cellule_runtime::Result { + validate_component(&input.actor)?; + if !decode_access(&context.sql(&SqlBatch { + statements: vec![access_statement(&input.actor)], + })?)? + .is_some_and(|role| role >= TokenScope::Read) + || identity( + &context.sql(&statement(IDENTITY, vec![]))?, + input.repository, + )? + .is_none() + { + return Ok(Some(CatalogCompletionReply::Denied( + PreparationDenial::Unauthorized, + ))); + } + let saved = context.sql(&statement( + "SELECT actor,request_digest,response_id,rejected,publication FROM pushes WHERE id=?1", + vec![blob(input.operation)], + ))?; + let Some(row) = rows(&saved)?.first() else { + return Ok(None); + }; + let [ + SqlValue::Text(actor), + digest, + response, + rejected, + publication, + ] = row.as_slice() + else { + return Err(Error::Command("invalid completed push lookup")); + }; + if *actor != input.actor || fixed::<32>(digest)? != input.request_digest { + return Ok(Some(CatalogCompletionReply::Denied( + PreparationDenial::Conflict, + ))); + } + let response_id = match response { + SqlValue::Blob(response) => fixed::<16>(&SqlValue::Blob(response.clone()))?, + SqlValue::Null if *publication == SqlValue::Null => return Ok(None), + _ => { + return Ok(Some(CatalogCompletionReply::Denied( + PreparationDenial::Conflict, + ))); + } + }; + let SqlValue::Integer(rejected) = rejected else { + return Err(Error::Command("invalid saved push refusal")); + }; + let publication = match publication { + SqlValue::Null => None, + SqlValue::Blob(bytes) => { + let mut d = BoundedDecoder::new(bytes, 128)?; + let value = PublishedRefs::decode(&mut d)?; + d.finish()?; + Some(value) + } + _ => return Err(Error::Command("invalid saved publication outcome")), + }; + Ok(Some(CatalogCompletionReply::Completed( + CompletedCatalogPush { + response_id, + rejected: *rejected == 1, + publication, + }, + ))) + } +} +pub struct CompleteCatalogPush; +impl Command for CompleteCatalogPush { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 19; + const CODEC_VERSION: u32 = 1; + type Input = CatalogPushCompletion; + type Output = CatalogCompletionReply; + fn execute( + context: &mut CommandContext<'_, '_>, + input: Self::Input, + ) -> cellule_runtime::Result> { + let completion_digest = input.binding()?; + let (check, key, catalog_data, outcome_data) = match &input.proof { + CompletionCatalogProof::Refs(proof) => { + let Some((data, key)) = authenticate( + context, + &proof.certificate, + input.proof.refs_digest()?, + Some(completion_digest), + )? + else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + ( + LeaseCheck { + token: data.token, + actor: data.actor.clone(), + }, + key, + Some(data), + None, + ) + } + CompletionCatalogProof::OutcomeOnly(certificate) => { + let Some((data, key)) = + super::outcome::authenticate(context, certificate, completion_digest)? + else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + (data.check.clone(), key, None, Some(data)) + } + }; + if input + .signed + .as_ref() + .is_some_and(|certificate| certificate.signer != check.actor) + { + return Ok(denied(PreparationDenial::Unauthorized)); + } + let saved = context.sql(&statement("SELECT actor,request_digest,response_id,completion_digest,rejected,publication FROM pushes WHERE id=?1", vec![blob(check.token.operation)]))?; + let mut pending = false; + if let Some(row) = rows(&saved)?.first() { + let [ + SqlValue::Text(actor), + digest, + response, + completion, + rejected, + publication, + ] = row.as_slice() + else { + return Err(Error::Command("invalid completed push identity")); + }; + if *actor != check.actor || fixed::<32>(digest)? != check.token.request_digest { + return Ok(denied(PreparationDenial::Conflict)); + } + match response { + SqlValue::Blob(response) => { + if fixed::<32>(completion)? != completion_digest { + return Ok(denied(PreparationDenial::Conflict)); + } + let publication = match publication { + SqlValue::Null => None, + SqlValue::Blob(bytes) => { + let mut d = BoundedDecoder::new(bytes, 128)?; + let value = PublishedRefs::decode(&mut d)?; + d.finish()?; + Some(value) + } + _ => return Err(Error::Command("invalid saved push publication")), + }; + let SqlValue::Integer(rejected) = rejected else { + return Err(Error::Command("invalid saved push refusal")); + }; + return Ok(CommandResult::Success(CatalogCompletionReply::Completed( + CompletedCatalogPush { + response_id: fixed::<16>(&SqlValue::Blob(response.clone()))?, + rejected: *rejected == 1, + publication, + }, + ))); + } + SqlValue::Null if *publication == SqlValue::Null => pending = true, + _ => return Ok(denied(PreparationDenial::Conflict)), + } + } + if check.token.owner != context.owner_fence() { + return Ok(denied(PreparationDenial::Stale)); + } + let Some(row) = load(context, check.token)? else { + return Ok(denied(PreparationDenial::Missing)); + }; + if !matched( + &row, + &LeaseCheck { + token: check.token, + actor: check.actor.clone(), + }, + ) { + return Ok(denied(PreparationDenial::Stale)); + } + check_pin(context, &row)?; + if row.expires <= now(context.now_ms())? { + return Ok(denied(PreparationDenial::Expired)); + } + let signed_digest: Option<[u8; 32]> = input + .signed + .as_ref() + .map(|certificate| Sha256::digest(&certificate.body).into()); + let replay = if let Some(digest) = signed_digest { + !rows(&context.sql(&statement( + "SELECT push_id FROM push_certificates WHERE digest=?1", + vec![blob(digest)], + ))?)? + .is_empty() + } else { + false + }; + if let Some(data) = &outcome_data + && !super::outcome::current_authority(context, data, row.generation)? + { + return Ok(denied(PreparationDenial::Unauthorized)); + } + // A moving catalog requires reconciliation only when publishing refs. + if !replay + && let Some(data) = &catalog_data + && (!super::publish::retention_matches( + context, + data, + row.generation, + data.catalog.format, + )? || super::commands::fact( + context, + check.token.repository, + data.catalog.format, + None, + )? != data.base) + { + return Ok(denied(PreparationDenial::Conflict)); + } + let mut publication = None; + let mut rejected = replay; + if !replay && let CompletionCatalogProof::Refs(proof) = &input.proof { + match super::publish::publish_authenticated( + context, + proof, + catalog_data + .clone() + .ok_or(Error::Command("missing catalog completion proof"))?, + key, + )? { + CommandResult::Success(PublicationReply::Published(value)) => { + publication = Some(value); + pending = true; + } + CommandResult::Rejected(PublicationReply::Denied( + PreparationDenial::Conflict | PreparationDenial::Unauthorized, + )) => rejected = true, + CommandResult::Rejected(PublicationReply::Denied(reason)) => { + return Ok(denied(reason)); + } + _ => return Err(Error::Command("invalid catalog publication decision")), + } + } + let reason = if replay { + "Canopy signed push certificate was already used" + } else { + crate::push::report::REJECTED + }; + let response = if rejected { + match crate::push::report::rejected_report(&input.response, reason) { + Ok(response) => std::borrow::Cow::Owned(response), + Err(_) if matches!(input.proof, CompletionCatalogProof::OutcomeOnly(_)) => { + std::borrow::Cow::Borrowed(&input.response) + } + Err(_) => return Err(Error::Command("cannot encode push refusal")), + } + } else { + std::borrow::Cow::Borrowed(&input.response) + }; + if publication.is_none() && row.expires <= now(context.now_ms())? { + return Ok(denied(PreparationDenial::Expired)); + } + // No rejection below this point. Any failure rolls back refs, catalog, + // signed-certificate ownership, chunks and completed response together. + if !pending { + changed(context.sql(&statement( + "INSERT INTO pushes(id,actor,request_digest) VALUES(?1,?2,?3)", + vec![ + blob(check.token.operation), + SqlValue::Text(check.actor.clone()), + blob(check.token.request_digest), + ], + ))?)?; + } + if let Some(certificate) = &input.signed + && !replay + { + changed(context.sql(&statement("INSERT INTO push_certificates(digest,push_id,actor,signer,key,size,recorded_at_ms) VALUES(?1,?2,?3,?4,?5,?6,?7)", vec![blob(signed_digest.ok_or(Error::Command("missing signed push digest"))?),blob(check.token.operation),SqlValue::Text(check.actor.clone()),SqlValue::Text(certificate.signer.clone()),SqlValue::Text(certificate.key.clone()),number(certificate.body.len() as u64)?,SqlValue::Integer(context.now_ms())]))?)?; + for (part, body) in certificate.body.chunks(CHUNK_BYTES).enumerate() { + changed(context.sql(&statement( + "INSERT INTO push_certificate_chunks(push_id,part,body) VALUES(?1,?2,?3)", + vec![ + blob(check.token.operation), + number(part as u64)?, + blob(body), + ], + ))?)?; + } + } + changed(context.sql(&statement("INSERT INTO push_responses(id,push_id,status,headers,size,digest) VALUES(?1,?2,?3,?4,?5,?6)", vec![blob(input.response_id),blob(check.token.operation),SqlValue::Integer(i64::from(response.status)),SqlValue::Text(serde_json::to_string(&response.headers).map_err(|_| Error::Command("invalid push response headers"))?),number(response.body.len() as u64)?,blob(blake3::hash(&response.body).as_bytes())]))?)?; + for (part, body) in response.body.chunks(CHUNK_BYTES).enumerate() { + changed(context.sql(&statement( + "INSERT INTO push_response_chunks(response_id,part,body) VALUES(?1,?2,?3)", + vec![blob(input.response_id), number(part as u64)?, blob(body)], + ))?)?; + } + changed(context.sql(&statement("UPDATE pushes SET response_id=?1,rejected=?2,options=?3,rejection_reason=?4,completion_digest=?5 WHERE id=?6 AND response_id IS NULL", vec![blob(input.response_id),SqlValue::Integer(i64::from(rejected)),SqlValue::Text(serde_json::to_string(&input.options).map_err(|_| Error::Command("invalid push options"))?),if rejected {SqlValue::Text(reason.into())} else {SqlValue::Null},blob(completion_digest),blob(check.token.operation)]))?)?; + if publication.is_none() { + changed(context.sql(&statement( + "DELETE FROM catalog_operations WHERE id=?1", + vec![blob(check.token.operation)], + ))?)?; + } + Ok(CommandResult::Success(CatalogCompletionReply::Completed( + CompletedCatalogPush { + response_id: input.response_id, + rejected, + publication, + }, + ))) + } +} diff --git a/crates/canopy-server/src/packs/publication/coordinator.rs b/crates/canopy-server/src/packs/publication/coordinator.rs new file mode 100644 index 0000000..f682bb6 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/coordinator.rs @@ -0,0 +1,997 @@ +//! Bounded, account-fair dispatch of already prepared final commands. +//! Catalog reconciliation, uploads and ref proof construction happen before +//! dispatch. This primitive does not guarantee progress against a moving root; +//! frontier pipelining and production orchestration are separate requirements. +use super::*; +use crate::{git_http::GitHttpResponse, packs::metadata::MetadataLimits}; +use cellule_ltx::DiskBudget; +use cellule_runtime::{ + CellClient, CellTarget, Committed, InvocationError, MutationIdentity, PreparedCommand, +}; +use std::{ + collections::{HashMap, VecDeque}, + path::Path, + sync::Arc, +}; +use tokio::{ + sync::{Mutex, Notify, watch}, + time::Instant, +}; + +/// Two encoded copies: retained exact retry and the in-flight transport. +/// Scratch/native work remains independently charged to its DiskBudget. +const COMMAND_RESERVATION: u64 = 8 << 20; +const INLINE_BYTES: u32 = 4 << 20; +mod inputs; +pub use inputs::{NativeInputReadyError, ReadyNativeInputs, RegisteredNativeInputs}; +mod preparation; +pub use preparation::{ + PreparationCommandKind, PreparationCommandOutcome, PreparationReadyError, ReadyPreparation, +}; +mod work; +use work::MAINTENANCE_RESERVATION; +pub use work::{ + CompactionReadyError, PublicationClass, PublicationError, PublicationOutcome, + ReadyCatalogCompaction, ReadyPublication, +}; + +#[derive(Clone, Copy, Debug)] +pub struct PublicationLimits { + /// Includes queued, executing and uncertain operations. + pub operations: usize, + pub per_actor: usize, + pub command_bytes: u64, + /// Commands awaiting durable outcome, not concurrent Cell transactions. + pub in_flight: usize, + /// Reserved admitted slots, including uncertain maintenance commands. + pub maintenance_operations: usize, + /// Maintenance cannot consume every concurrent durability wait slot. + pub maintenance_in_flight: usize, + pub foreground_burst: u8, +} +impl Default for PublicationLimits { + fn default() -> Self { + Self { + operations: 32, + per_actor: 8, + command_bytes: 256 << 20, + in_flight: 8, + maintenance_operations: 4, + maintenance_in_flight: 2, + foreground_burst: 3, + } + } +} +impl PublicationLimits { + fn validate(self) -> Result<(), PublicationScheduleError> { + if self.operations < 3 + || self.operations > MAX_OPERATIONS as usize + || self.maintenance_operations == 0 + || self.maintenance_operations >= self.operations + || self.per_actor == 0 + || self.per_actor >= self.operations - self.maintenance_operations + || self + .command_bytes + .saturating_sub(self.maintenance_operations as u64 * MAINTENANCE_RESERVATION) + / COMMAND_RESERVATION + <= self.per_actor as u64 + || self.in_flight == 0 + || self.in_flight > self.operations + || self.maintenance_in_flight == 0 + || self.maintenance_in_flight > self.in_flight + || (self.in_flight > 1 && self.maintenance_in_flight == self.in_flight) + || !(1..=32).contains(&self.foreground_burst) + { + return Err(PublicationScheduleError::InvalidLimits); + } + Ok(()) + } +} + +/// Private factory output. Keep this exact command on admission failure; +/// rebuilding a completion allocates a different response identity. +#[must_use] +pub struct ReadyCatalogPush { + owner: PushPreparation, + command: PreparedCommand, +} +#[derive(Clone)] +enum PushPreparation { + Catalog(Arc), + Outcome(Arc), +} +impl PushPreparation { + fn session(&self) -> &PreparationSession { + match self { + Self::Catalog(prepared) => &prepared.base.session, + Self::Outcome(session) => session, + } + } + fn capability(&self) -> (&CellClient, &CellTarget, &LeaseCheck) { + self.session().capability() + } +} +impl PreparationSession { + /// Retain the exact outcome command in the same bounded dispatch/recovery + /// path as ref publications. No artifacts or catalog upload are needed. + pub async fn ready_outcome( + self: &Arc, + identity: MutationIdentity, + request: PushCompletionRequest, + ) -> Result { + let input = self.push_outcome(request).await?; + input.encode(&mut BoundedEncoder::new(INLINE_BYTES)?)?; + self.live_lease()?; + let command = self + .client + .prepare_command::(&self.target, identity, input) + .await + .map_err(|error| PushCompletionProofError::Command(Box::new(error)))?; + self.live_lease()?; + Ok(ReadyCatalogPush { + owner: PushPreparation::Outcome(Arc::clone(self)), + command, + }) + } +} +impl PreparedCatalog { + pub async fn ready_push( + self: &Arc, + identity: MutationIdentity, + request: PushCompletionRequest, + root: &Path, + budget: DiskBudget, + limits: MetadataLimits, + ) -> Result { + let outcome_only = request.plan.is_none(); + let input = Box::pin(self.push_completion(request, root, budget, limits)).await?; + // The reservation has a fixed upper bound even if a registry is later + // configured with a larger command envelope. Large plans need roots. + input.encode(&mut BoundedEncoder::new(INLINE_BYTES)?)?; + self.ensure_live()?; + let (client, target, _) = self.base.capability(); + let command = client + .prepare_command::(target, identity, input) + .await + .map_err(|error| PushCompletionProofError::Command(Box::new(error)))?; + self.ensure_live()?; + Ok(ReadyCatalogPush { + owner: if outcome_only { + PushPreparation::Outcome(Arc::new(self.base.session.clone())) + } else { + PushPreparation::Catalog(Arc::clone(self)) + }, + command, + }) + } +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq, thiserror::Error)] +pub enum PublicationScheduleError { + #[error("invalid publication admission limits")] + InvalidLimits, + #[error("publication coordinator is closed")] + Closed, + #[error("publication admission capacity exceeded")] + Capacity, + #[error("publication belongs to another repository coordinator")] + Foreign, + #[error("logical publication already admitted")] + Duplicate, + #[error("publication does not have an unresolved outcome")] + NotUncertain, + #[error("publication is no longer held without execution")] + NotHeld, +} +pub struct PublicationAdmissionFailure { + pub reason: PublicationScheduleError, + pub ready: ReadyPublication, +} +impl std::fmt::Debug for PublicationAdmissionFailure { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("PublicationAdmissionFailure") + .field("reason", &self.reason) + .finish_non_exhaustive() + } +} +impl std::fmt::Display for PublicationAdmissionFailure { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + self.reason.fmt(f) + } +} +impl std::error::Error for PublicationAdmissionFailure {} + +#[derive(Clone, Debug)] +pub enum PublicationState { + /// Admitted and charged, but cannot execute until explicitly activated. + Held, + Queued, + Running, + /// Retained in the coordinator and charged until explicit resolution. + Uncertain(Arc), + /// Rejected carries its durable receipt; NotStarted never acknowledges. + Finished(Result>), + /// Held proof dropped before returning credits; no command was dispatched. + Discarded, +} +impl PublicationState { + fn observed(&self) -> bool { + matches!( + self, + Self::Uncertain(_) | Self::Finished(_) | Self::Discarded + ) + } +} +struct ReadContext { + client: CellClient, + target: CellTarget, + request: BeginRequest, +} +struct Job { + operation: [u8; 16], + actor: String, + // Removed before terminal notification; tickets never retain command + // payloads or local inventory after the admission charge is released. + ready: Mutex>, + class: PublicationClass, + reservation: u64, + status: watch::Sender, + read: ReadContext, + admitted: Instant, +} +struct Work { + job: Arc, + recover: bool, + queued: Instant, +} +struct FairQueue { + actors: VecDeque, + queues: HashMap>, +} +impl Default for FairQueue { + fn default() -> Self { + Self { + actors: VecDeque::new(), + queues: HashMap::new(), + } + } +} +impl FairQueue { + fn push(&mut self, actor: String, work: T) { + let queue = self.queues.entry(actor.clone()).or_default(); + if queue.is_empty() { + self.actors.push_back(actor); + } + queue.push_back(work); + } + fn pop(&mut self) -> Option { + let actor = self.actors.pop_front()?; + let queue = self.queues.get_mut(&actor)?; + let work = queue.pop_front(); + if queue.is_empty() { + self.queues.remove(&actor); + } else { + self.actors.push_back(actor); + } + work + } +} +struct ClassQueue { + foreground: FairQueue, + maintenance: FairQueue, + foreground_streak: u8, +} +impl Default for ClassQueue { + fn default() -> Self { + Self { + foreground: FairQueue::default(), + maintenance: FairQueue::default(), + foreground_streak: 0, + } + } +} +impl ClassQueue { + fn push(&mut self, class: PublicationClass, actor: String, work: T) { + match class { + PublicationClass::Foreground => self.foreground.push(actor, work), + PublicationClass::Maintenance => self.maintenance.push(actor, work), + } + } + fn pop(&mut self, maintenance_ready: bool, burst: u8) -> Option { + if maintenance_ready + && self.foreground_streak >= burst + && let Some(work) = self.maintenance.pop() + { + self.foreground_streak = 0; + return Some(work); + } + if let Some(work) = self.foreground.pop() { + self.foreground_streak = self.foreground_streak.saturating_add(1); + return Some(work); + } + if maintenance_ready && let Some(work) = self.maintenance.pop() { + self.foreground_streak = 0; + return Some(work); + } + None + } +} +#[derive(Default)] +struct State { + jobs: HashMap<[u8; 16], Arc>, + actors: HashMap, + queue: ClassQueue, + counts: [usize; 2], + bytes: [u64; 2], + worker: bool, + closed: bool, +} +struct Inner { + target: CellTarget, + limits: PublicationLimits, + state: Mutex, + drained: Notify, + changed: Notify, + #[cfg(test)] + gate: Mutex>, + #[cfg(test)] + fault: std::sync::atomic::AtomicU8, +} +#[cfg(test)] +struct TestGate { + entered: tokio::sync::oneshot::Sender<()>, + release: tokio::sync::oneshot::Receiver<()>, +} + +/// Keep one service-owned instance per repository. Dropping a waiter does not +/// cancel an admitted command. Close/drain on shutdown and retain the returned +/// unresolved tickets for outcome recovery; this is not a durable local outbox. +#[derive(Clone)] +pub struct PublicationCoordinator { + inner: Arc, +} +#[must_use] +#[derive(Clone)] +pub struct PublicationTicket { + inner: Arc, + job: Arc, +} +/// Bounded service snapshot, including uncertain work that is still charged. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct PublicationStats { + pub admitted: usize, + pub accounts: usize, + pub held: usize, + pub queued: usize, + pub in_flight: usize, + pub uncertain: usize, + pub command_bytes: u64, + pub closed: bool, + pub foreground: usize, + pub maintenance: usize, +} +impl PublicationCoordinator { + pub fn new( + target: CellTarget, + limits: PublicationLimits, + ) -> Result { + limits.validate()?; + Ok(Self { + inner: Arc::new(Inner { + target, + limits, + state: Mutex::new(State::default()), + drained: Notify::new(), + changed: Notify::new(), + #[cfg(test)] + gate: Mutex::new(None), + #[cfg(test)] + fault: std::sync::atomic::AtomicU8::new(0), + }), + }) + } + /// Non-waiting admission. All unbounded/native work precedes this call. + /// The account key comes from the private lease, never a request label. + pub async fn submit( + &self, + ready: impl Into, + ) -> Result> { + let ready = ready.into(); + let mut state = self.inner.state.lock().await; + self.admit(&mut state, ready, false) + } + /// Synchronous ownership handoff without dispatch. A lifecycle can retain + /// this ticket before yielding, so cancellation cannot strand an unowned + /// admission. Contended admission returns the original ready value. + pub fn try_reserve( + &self, + ready: impl Into, + ) -> Result> { + let ready = ready.into(); + let Ok(mut state) = self.inner.state.try_lock() else { + return Err(Box::new(PublicationAdmissionFailure { + reason: PublicationScheduleError::Capacity, + ready, + })); + }; + self.admit(&mut state, ready, true) + } + fn admit( + &self, + state: &mut State, + ready: ReadyPublication, + held: bool, + ) -> Result> { + let class = ready.class(); + let reservation = ready.reservation(); + let at = class.index(); + let (client, target, check) = ready.capability(); + let limits = self.inner.limits; + let (operation_limit, byte_limit) = match class { + PublicationClass::Foreground => ( + limits.operations - limits.maintenance_operations, + limits.command_bytes + - limits.maintenance_operations as u64 * MAINTENANCE_RESERVATION, + ), + PublicationClass::Maintenance => ( + limits.maintenance_operations, + limits.maintenance_operations as u64 * MAINTENANCE_RESERVATION, + ), + }; + let reason = if target != &self.inner.target { + Some(PublicationScheduleError::Foreign) + } else if state.closed { + Some(PublicationScheduleError::Closed) + } else if state.jobs.contains_key(&check.token.operation) { + Some(PublicationScheduleError::Duplicate) + } else if state.counts[at] >= operation_limit + || state + .actors + .get(&check.actor) + .map_or(0, |counts| counts[at]) + >= limits.per_actor + || state.bytes[at] > byte_limit - reservation + { + Some(PublicationScheduleError::Capacity) + } else { + None + }; + if let Some(reason) = reason { + return Err(Box::new(PublicationAdmissionFailure { reason, ready })); + } + let read = ReadContext { + client: client.clone(), + target: target.clone(), + request: BeginRequest { + repository: check.token.repository, + operation: check.token.operation, + request_digest: check.token.request_digest, + actor: check.actor.clone(), + lease_ms: DEFAULT_LEASE_MS, + }, + }; + let job = Arc::new(Job { + operation: check.token.operation, + actor: check.actor.clone(), + class, + reservation, + ready: Mutex::new(Some(ready)), + status: watch::channel(if held { + PublicationState::Held + } else { + PublicationState::Queued + }) + .0, + read, + admitted: Instant::now(), + }); + state.actors.entry(job.actor.clone()).or_default()[at] += 1; + state.counts[at] += 1; + state.bytes[at] += job.reservation; + state.jobs.insert(job.operation, Arc::clone(&job)); + if !held { + enqueue(state, &job, false); + self.start(state); + self.inner.changed.notify_one(); + } + Ok(PublicationTicket { + inner: Arc::clone(&self.inner), + job, + }) + } + /// Recover only the retained exact command. A fresh response/proof factory + /// cannot replace it while acceptance is unknown. Recovery joins the fair + /// queue; an unknown/expired/unreachable resolution retains its reservation. + pub async fn recover( + &self, + ticket: &PublicationTicket, + ) -> Result<(), PublicationScheduleError> { + if !Arc::ptr_eq(&self.inner, &ticket.inner) { + return Err(PublicationScheduleError::Foreign); + } + let mut state = self.inner.state.lock().await; + if !state + .jobs + .get(&ticket.job.operation) + .is_some_and(|job| Arc::ptr_eq(job, &ticket.job)) + || !matches!(*ticket.job.status.borrow(), PublicationState::Uncertain(_)) + { + return Err(PublicationScheduleError::NotUncertain); + } + ticket.job.status.send_replace(PublicationState::Queued); + enqueue(&mut state, &ticket.job, true); + self.start(&mut state); + self.inner.changed.notify_one(); + Ok(()) + } + fn start(&self, state: &mut State) { + if !state.worker { + state.worker = true; + tokio::spawn(supervise(Arc::clone(&self.inner))); + } + } + /// Stop new work and wait for dispatched commands (without cancellation). + /// Held activation/discard and recovery remain available after closing. + /// Close the producer lifecycle first: this does not activate held proofs. + /// Every returned ticket remains + /// charged and owns the original command, identity and verification input. + pub async fn close_and_drain(&self) -> Vec { + loop { + let wake = self.inner.drained.notified(); + tokio::pin!(wake); + wake.as_mut().enable(); + { + let mut state = self.inner.state.lock().await; + state.closed = true; + if !state.worker { + return state + .jobs + .values() + .map(|job| PublicationTicket { + inner: Arc::clone(&self.inner), + job: Arc::clone(job), + }) + .collect(); + } + } + wake.await; + } + } + /// Service-internal lookup after its caller loses a ticket. This is not an + /// externally authorized product query; use completed-request replay there. + pub async fn pending(&self, operation: [u8; 16]) -> Option { + self.inner + .state + .lock() + .await + .jobs + .get(&operation) + .map(|job| PublicationTicket { + inner: Arc::clone(&self.inner), + job: Arc::clone(job), + }) + } + pub async fn stats(&self) -> PublicationStats { + let state = self.inner.state.lock().await; + let mut stats = PublicationStats { + admitted: state.jobs.len(), + accounts: state.actors.len(), + held: 0, + queued: 0, + in_flight: 0, + uncertain: 0, + command_bytes: state.bytes.iter().sum(), + closed: state.closed, + foreground: state.counts[0], + maintenance: state.counts[1], + }; + for job in state.jobs.values() { + match *job.status.borrow() { + PublicationState::Held => stats.held += 1, + PublicationState::Queued => stats.queued += 1, + PublicationState::Running => stats.in_flight += 1, + PublicationState::Uncertain(_) => stats.uncertain += 1, + PublicationState::Finished(_) | PublicationState::Discarded => {} + } + } + stats + } + #[cfg(test)] + pub(super) async fn pause_for_test( + &self, + ) -> ( + tokio::sync::oneshot::Sender<()>, + tokio::sync::oneshot::Receiver<()>, + ) { + let (release, wait) = tokio::sync::oneshot::channel(); + let (entered, start) = tokio::sync::oneshot::channel(); + *self.inner.gate.lock().await = Some(TestGate { + entered, + release: wait, + }); + (release, start) + } + #[cfg(test)] + pub(super) fn fault_for_test(&self, fault: u8) { + self.inner + .fault + .store(fault, std::sync::atomic::Ordering::Release); + } + #[cfg(test)] + pub(super) async fn reservations_for_test(&self) -> (usize, u64, usize) { + let state = self.inner.state.lock().await; + ( + state.jobs.len(), + state.bytes.iter().sum(), + state.actors.len(), + ) + } + #[cfg(test)] + pub(super) async fn with_admission_for_test(&self, inspect: impl FnOnce() -> T) -> T { + let _state = self.inner.state.lock().await; + inspect() + } +} +impl PublicationTicket { + /// Join the existing fair queue once. Idempotence lets a recovering + /// lifecycle observe an already activated command without recreating it. + /// Existing admission can activate after coordinator close. + pub async fn activate(&self) -> Result<(), PublicationScheduleError> { + let mut state = self.inner.state.lock().await; + let held = matches!(self.state(), PublicationState::Held); + if matches!(self.state(), PublicationState::Discarded) { + return Err(PublicationScheduleError::NotHeld); + } + if !held { + return Ok(()); + } + self.job.status.send_replace(PublicationState::Queued); + enqueue(&mut state, &self.job, false); + PublicationCoordinator { + inner: Arc::clone(&self.inner), + } + .start(&mut state); + self.inner.changed.notify_one(); + Ok(()) + } + /// Only a proven unexecuted held command can be discarded. Holding the + /// coordinator state lock serializes this with activation; drop private + /// proof resources before returning their admission credits. + pub async fn discard_held(&self) -> Result<(), PublicationScheduleError> { + let mut state = self.inner.state.lock().await; + if !matches!(self.state(), PublicationState::Held) { + return Err(PublicationScheduleError::NotHeld); + } + self.job.ready.lock().await.take(); + release(&mut state, &self.job); + self.job.status.send_replace(PublicationState::Discarded); + self.inner.drained.notify_waiters(); + Ok(()) + } + pub async fn recover(&self) -> Result<(), PublicationScheduleError> { + PublicationCoordinator { + inner: Arc::clone(&self.inner), + } + .recover(self) + .await + } + pub(super) async fn wait_recovered(&self) { + let mut status = self.job.status.subscribe(); + while matches!(*status.borrow_and_update(), PublicationState::Uncertain(_)) { + if status.changed().await.is_err() { + return; + } + } + } + pub fn class(&self) -> PublicationClass { + self.job.class + } + pub fn state(&self) -> PublicationState { + self.job.status.borrow().clone() + } + /// Cancellation drops only this observation, never admitted execution. + pub async fn wait(&self) -> PublicationState { + let mut status = self.job.status.subscribe(); + loop { + let observed = status.borrow_and_update().clone(); + if observed.observed() { + return observed; + } + if status.changed().await.is_err() { + return status.borrow().clone(); + } + } + } + pub async fn response(&self) -> Result { + let completed = match self.state() { + PublicationState::Finished(Ok(PublicationOutcome::Push(completed))) => completed, + _ => return Err(CatalogPushResponseError::Invalid), + }; + let CatalogCompletionReply::Completed(output) = completed.output else { + return Err(CatalogPushResponseError::Invalid); + }; + let read = &self.job.read; + super::completion::load_response( + &read.client, + &read.target, + &read.request, + completed.receipt, + output, + ) + .await + } +} + +fn enqueue(state: &mut State, job: &Arc, recover: bool) { + state.queue.push( + job.class, + job.actor.clone(), + Work { + job: Arc::clone(job), + recover, + queued: Instant::now(), + }, + ); +} +fn release(state: &mut State, job: &Job) { + state.jobs.remove(&job.operation); + let count = state + .actors + .get_mut(&job.actor) + .expect("admitted actor count"); + let at = job.class.index(); + count[at] -= 1; + if *count == [0, 0] { + state.actors.remove(&job.actor); + } + state.counts[at] -= 1; + state.bytes[at] -= job.reservation; +} + +async fn supervise(inner: Arc) { + loop { + if tokio::spawn(run(Arc::clone(&inner))).await.is_ok() { + return; + } + // A panicking dispatch task is an uncertain outcome, never evidence + // that the transport did not accept the mutation. Preserve its exact + // command, notify observers and keep other ready accounts progressing. + let state = inner.state.lock().await; + for job in state.jobs.values() { + // Release watch's read guard before awaiting or replacing status. + let running = matches!(*job.status.borrow(), PublicationState::Running); + if running && let Some(ready) = job.ready.lock().await.as_ref() { + job.status + .send_replace(PublicationState::Uncertain(Arc::new(ready.pending()))); + } + } + drop(state); + } +} +type DispatchResult = Result; +async fn run(inner: Arc) { + let mut tasks = tokio::task::JoinSet::new(); + let mut active: HashMap> = HashMap::new(); + loop { + let wake = inner.changed.notified(); + tokio::pin!(wake); + wake.as_mut().enable(); + { + let mut state = inner.state.lock().await; + while tasks.len() < inner.limits.in_flight { + let maintenance_active = active + .values() + .filter(|job| job.class == PublicationClass::Maintenance) + .count(); + let Some(work) = state.queue.pop( + maintenance_active < inner.limits.maintenance_in_flight, + inner.limits.foreground_burst, + ) else { + break; + }; + work.job.status.send_replace(PublicationState::Running); + tracing::debug!(target: "canopy::publication", event = "dispatch", + operation = ?work.job.operation, class = ?work.job.class, recovery = work.recover, + queue_wait_us = work.queued.elapsed().as_micros()); + let job = Arc::clone(&work.job); + let handle = tasks.spawn(dispatch(Arc::clone(&inner), work)); + active.insert(handle.id(), job); + } + if tasks.is_empty() { + state.worker = false; + inner.drained.notify_waiters(); + return; + } + } + // Multiple commands can await durable publication. The Cell's own + // transaction order and CAS remain authoritative; dispatch order does + // not promise network arrival order or overlapping push success. + tokio::select! { + result = tasks.join_next_with_id() => { + match result.expect("active publication task") { + Ok((id, outcome)) => { + let job = active.remove(&id).expect("active publication identity"); + finish(&inner, &job, outcome).await; + } + Err(error) => { + let job = active.remove(&error.id()).expect("failed publication identity"); + let error = job.ready.lock().await.as_ref().expect("failed command retained").pending(); + finish(&inner, &job, Err(error)).await; + } + } + } + _ = wake => {} + } + } +} +async fn dispatch(inner: Arc, work: Work) -> DispatchResult { + #[cfg(test)] + { + // Do not hold the hook's mutex across a wait: other dispatched jobs + // must be able to progress while the first transport is suspended. + let gate = inner.gate.lock().await.take(); + if let Some(gate) = gate { + let _ = gate.entered.send(()); + let _ = gate.release.await; + } + } + #[cfg(not(test))] + drop(inner); + let ready = work + .job + .ready + .lock() + .await + .as_ref() + .expect("admitted publication retains its command") + .dispatch_copy(); + #[cfg(test)] + let fault = inner.fault.swap(0, std::sync::atomic::Ordering::AcqRel); + #[cfg(not(test))] + let fault = 0; + ready.dispatch(work.recover, fault).await +} +async fn finish(inner: &Inner, job: &Job, outcome: DispatchResult) { + let disposition = outcome + .as_ref() + .err() + .map_or("committed", PublicationError::disposition); + tracing::debug!(target: "canopy::publication", event = "observed", operation = ?job.operation, + class = ?job.class, disposition, residence_us = job.admitted.elapsed().as_micros()); + let uncertain = outcome + .as_ref() + .err() + .is_some_and(PublicationError::uncertain); + if uncertain { + job.status + .send_replace(PublicationState::Uncertain(Arc::new(outcome.unwrap_err()))); + } else { + // Drop large resources before making their admission reusable. + job.ready.lock().await.take(); + let mut state = inner.state.lock().await; + release(&mut state, job); + job.status + .send_replace(PublicationState::Finished(outcome.map_err(Arc::new))); + } +} + +#[cfg(test)] +mod fairness { + use super::*; + #[test] + fn limits_leave_room_for_another_account_and_bound_dispatch_concurrency() { + assert!(PublicationLimits::default().validate().is_ok()); + for limits in [ + PublicationLimits { + operations: 0, + ..PublicationLimits::default() + }, + PublicationLimits { + per_actor: 32, + ..PublicationLimits::default() + }, + PublicationLimits { + command_bytes: 8 << 20, + ..PublicationLimits::default() + }, + PublicationLimits { + command_bytes: 64 << 20, + ..PublicationLimits::default() + }, + PublicationLimits { + in_flight: 0, + ..PublicationLimits::default() + }, + PublicationLimits { + in_flight: 33, + ..PublicationLimits::default() + }, + PublicationLimits { + maintenance_operations: 0, + ..PublicationLimits::default() + }, + PublicationLimits { + maintenance_operations: 32, + ..PublicationLimits::default() + }, + PublicationLimits { + maintenance_in_flight: 0, + ..PublicationLimits::default() + }, + PublicationLimits { + maintenance_in_flight: 8, + ..PublicationLimits::default() + }, + PublicationLimits { + foreground_burst: 0, + ..PublicationLimits::default() + }, + PublicationLimits { + foreground_burst: 33, + ..PublicationLimits::default() + }, + ] { + assert_eq!( + limits.validate(), + Err(PublicationScheduleError::InvalidLimits) + ); + } + } + #[test] + fn ready_accounts_rotate_and_each_account_preserves_fifo() { + let mut queue = FairQueue::default(); + for n in 0..8 { + queue.push("busy".into(), n); + } + queue.push("other".into(), 100); + queue.push("third".into(), 200); + assert_eq!(queue.pop(), Some(0)); + // Busy traffic cannot move itself ahead of already ready accounts. + queue.push("busy".into(), 8); + queue.push("other".into(), 101); + assert_eq!( + (queue.pop(), queue.pop(), queue.pop(), queue.pop()), + (Some(100), Some(200), Some(1), Some(101)) + ); + for n in 2..9 { + assert_eq!(queue.pop(), Some(n)); + } + assert_eq!(queue.pop(), None); + assert!(queue.actors.is_empty() && queue.queues.is_empty()); + queue.push("busy".into(), 9); + assert_eq!(queue.pop(), Some(9)); + } + #[test] + fn ready_classes_bound_foreground_bursts_and_preserve_blocked_maintenance() { + let mut queue = ClassQueue::default(); + for n in 0..12 { + queue.push(PublicationClass::Foreground, "busy".into(), n); + } + queue.push(PublicationClass::Maintenance, "admin-a".into(), 100); + queue.push(PublicationClass::Maintenance, "admin-b".into(), 200); + assert_eq!( + (queue.pop(true, 3), queue.pop(true, 3), queue.pop(true, 3)), + (Some(0), Some(1), Some(2)) + ); + // A running maintenance job at its concurrency cap must not block ready + // foreground work or remove the waiting maintenance command. + assert_eq!(queue.pop(false, 3), Some(3)); + assert_eq!(queue.pop(true, 3), Some(100)); + assert_eq!( + ( + queue.pop(true, 3), + queue.pop(true, 3), + queue.pop(true, 3), + queue.pop(true, 3) + ), + (Some(4), Some(5), Some(6), Some(200)) + ); + for n in 7..12 { + assert_eq!(queue.pop(true, 3), Some(n)); + } + assert_eq!(queue.pop(true, 3), None); + queue.push(PublicationClass::Maintenance, "admin-a".into(), 101); + assert_eq!(queue.pop(false, 3), None); + assert_eq!(queue.pop(true, 3), Some(101)); + } +} diff --git a/crates/canopy-server/src/packs/publication/coordinator/inputs.rs b/crates/canopy-server/src/packs/publication/coordinator/inputs.rs new file mode 100644 index 0000000..bdbea9b --- /dev/null +++ b/crates/canopy-server/src/packs/publication/coordinator/inputs.rs @@ -0,0 +1,100 @@ +//! Exact bound input checkpoint dispatch reuses publication admission and recovery. +use super::*; + +pub(super) const INPUT_RESERVATION: u64 = 8192; +const INPUT_INLINE_BYTES: u32 = 4096; + +#[derive(Debug, thiserror::Error)] +pub enum NativeInputReadyError { + #[error("bound input preparation inactive")] + Base(#[from] PreparationBaseError), + #[error("bound input checkpoint encoding failed")] + Codec(#[from] CodecError), + #[error("bound input checkpoint command preparation failed")] + Command(#[source] Box>), +} +#[must_use] +pub struct ReadyNativeInputs { + pub(super) session: Arc, + pub(super) command: PreparedCommand, + pub(super) digest: [u8; 32], +} +/// Original durable outcome plus a fresh custody observation. Neither the +/// recorded lease nor a successful observation grants future authority; every +/// proof factory still checks the shared session and final commands recheck SQL. +#[derive(Clone, Debug)] +pub struct RegisteredNativeInputs { + pub registration: Committed, + pub custody: Result<(), Arc>, +} +impl PreparationSession { + /// Prepare, but do not execute, the exact adopted-input checkpoint command. + /// Submit through the repository's existing PublicationCoordinator. Failed + /// admission returns the same command, identity and session for later use. + pub async fn ready_inputs( + self: &Arc, + identity: MutationIdentity, + proof: NativeInputCertificate, + ) -> Result { + self.live_lease()?; + let digest = proof.bound_digest(self)?; + proof.encode(&mut BoundedEncoder::new(INPUT_INLINE_BYTES)?)?; + let command = self + .client + .prepare_command::(&self.target, identity, proof) + .await + .map_err(|e| NativeInputReadyError::Command(Box::new(e)))?; + self.live_lease()?; + Ok(ReadyNativeInputs { + session: self.clone(), + command, + digest, + }) + } +} +impl ReadyNativeInputs { + pub(super) async fn dispatch(self, recover: bool, fault: u8) -> DispatchResult { + let result = super::super::exact::invoke( + &self.session.client, + self.command, + recover, + INPUT_INLINE_BYTES, + fault, + ) + .await; + match result { + Ok(registration) => { + // Keep the known original outcome across fresh probes. Failure + // revokes local use, never converts committed recovery to denial. + let matched = matches!(®istration.output, StagingReply::Granted(lease) + if lease.token == self.session.lease.token && lease.format == self.session.lease.format); + let custody = if matched { + super::super::inputs::observe_bound_registration( + &self.session, + self.digest, + registration.receipt, + ) + .await + } else { + Err(PreparationBaseError::Context.into()) + }; + if custody.is_err() { + self.session.fence(); + } + Ok(PublicationOutcome::Inputs(RegisteredNativeInputs { + registration, + custody: custody.map_err(Arc::new), + })) + } + Err(error) => { + if !matches!( + &error, + InvocationError::Pending(_) | InvocationError::InvalidPublishedResult { .. } + ) { + self.session.fence(); + } + Err(PublicationError::Inputs(error)) + } + } + } +} diff --git a/crates/canopy-server/src/packs/publication/coordinator/preparation.rs b/crates/canopy-server/src/packs/publication/coordinator/preparation.rs new file mode 100644 index 0000000..9ae7861 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/coordinator/preparation.rs @@ -0,0 +1,224 @@ +//! Bound attempt commands share exact publication ownership and admission. +use super::*; + +const INLINE_BYTES: u32 = 4096; +pub(super) const RESERVATION: u64 = 2 * INLINE_BYTES as u64; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum PreparationCommandKind { + Claim, + Renew, +} +#[derive(Debug, thiserror::Error)] +pub enum PreparationReadyError { + #[error("preparation inactive or context differs")] + Base(#[from] PreparationBaseError), + #[error("preparation command encoding failed")] + Codec(#[from] CodecError), + #[error("preparation command preparation failed")] + Command(#[source] Box>), +} +#[derive(Clone)] +enum ExactPreparation { + Claim(PreparedCommand), + Renew { + command: PreparedCommand, + session: Arc, + }, +} +#[must_use] +pub struct ReadyPreparation { + inner: Box, +} +#[derive(Clone)] +struct PreparationRequest { + client: CellClient, + target: CellTarget, + check: LeaseCheck, + exact: ExactPreparation, +} +/// Recorded outcomes remain recoverable even when fresh custody is unavailable. +/// Only a freshly queried session can authorize subsequent private factories. +#[derive(Clone)] +pub struct PreparationCommandOutcome { + pub kind: PreparationCommandKind, + pub committed: Committed, + pub session: Result, Arc>, +} +impl std::fmt::Debug for PreparationCommandOutcome { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("PreparationCommandOutcome") + .field("kind", &self.kind) + .field("committed", &self.committed) + .field("session", &self.session.as_ref().map(|s| s.lease.token)) + .finish() + } +} +fn validate(target: &CellTarget, request: &LeaseRequest) -> Result<(), PreparationReadyError> { + if crate::repository_target( + target.tenant(), + target.application(), + request.check.token.repository, + ) + .map_err(|_| PreparationBaseError::Context)? + != *target + || request.lease_ms == 0 + || request.lease_ms > MAX_LEASE_MS + { + return Err(PreparationBaseError::Context.into()); + } + request.encode(&mut BoundedEncoder::new(INLINE_BYTES)?)?; + Ok(()) +} +impl ReadyPreparation { + /// Claim may recover an expired or previous-owner attempt. Do not require a + /// local live session; authoritative execution checks the exact old token. + pub async fn claim( + client: CellClient, + target: CellTarget, + request: LeaseRequest, + identity: MutationIdentity, + ) -> Result { + validate(&target, &request)?; + let check = request.check.clone(); + let command = client + .prepare_command::(&target, identity, request) + .await + .map_err(|e| PreparationReadyError::Command(Box::new(e)))?; + Ok(Self { + inner: Box::new(PreparationRequest { + client, + target, + check, + exact: ExactPreparation::Claim(command), + }), + }) + } + pub(super) fn dispatch_copy(&self) -> Self { + Self { + inner: self.inner.clone(), + } + } + pub(super) fn capability(&self) -> (&CellClient, &CellTarget, &LeaseCheck) { + (&self.inner.client, &self.inner.target, &self.inner.check) + } + pub(super) fn pending(&self) -> PublicationError { + let evidence = match &self.inner.exact { + ExactPreparation::Claim(command) => command.evidence(), + ExactPreparation::Renew { command, .. } => command.evidence(), + }; + PublicationError::Preparation(InvocationError::Pending(Box::new(evidence.clone()))) + } + pub(super) async fn dispatch(self, recover: bool, fault: u8) -> DispatchResult { + let inner = *self.inner; + let (kind, result, existing) = match inner.exact { + ExactPreparation::Claim(command) => ( + PreparationCommandKind::Claim, + super::super::exact::invoke(&inner.client, command, recover, INLINE_BYTES, fault) + .await, + None, + ), + ExactPreparation::Renew { command, session } => ( + PreparationCommandKind::Renew, + super::super::exact::invoke(&inner.client, command, recover, INLINE_BYTES, fault) + .await, + Some(session), + ), + }; + let committed = match result { + Ok(committed) => committed, + Err(error) => { + if !matches!( + &error, + InvocationError::Pending(_) | InvocationError::InvalidPublishedResult { .. } + ) && let Some(session) = existing + { + session.fence(); + } + return Err(PublicationError::Preparation(error)); + } + }; + let custody = async { + let PreparationReply::Granted(lease) = &committed.output else { + return Err(PreparationBaseError::Context); + }; + if let Some(session) = &existing { + if lease.token != session.lease.token + || lease.base != session.lease.base + || lease.format != session.lease.format + { + return Err(PreparationBaseError::Context); + } + session.refresh(committed.receipt).await?; + Ok(session.clone()) + } else { + if lease.token.repository != inner.check.token.repository + || lease.token.operation != inner.check.token.operation + || lease.token.request_digest != inner.check.token.request_digest + || lease.token == inner.check.token + { + return Err(PreparationBaseError::Context); + } + let session = PreparationSession::open( + inner.client.clone(), + inner.target.clone(), + LeaseCheck { + token: lease.token, + actor: inner.check.actor.clone(), + }, + Some(committed.receipt), + ) + .await?; + if session.lease.base != lease.base || session.lease.format != lease.format { + return Err(PreparationBaseError::Context); + } + Ok(Arc::new(session)) + } + } + .await; + if custody.is_err() + && let Some(session) = existing + { + session.fence(); + } + Ok(PublicationOutcome::Preparation(PreparationCommandOutcome { + kind, + committed, + session: custody.map_err(Arc::new), + })) + } +} +impl PreparationSession { + /// Prepare an exact renewal for service dispatch; a refused admission keeps + /// the same identity. An ambiguous renewal keeps the previously observed + /// deadline until resolved, and never grants custody from a recorded clock. + pub async fn ready_renew( + self: &Arc, + identity: MutationIdentity, + lease_ms: u64, + ) -> Result { + self.live_lease()?; + let request = LeaseRequest { + check: self.check.clone(), + lease_ms, + }; + validate(&self.target, &request)?; + let command = self + .client + .prepare_command::(&self.target, identity, request) + .await + .map_err(|e| PreparationReadyError::Command(Box::new(e)))?; + self.live_lease()?; + Ok(ReadyPreparation { + inner: Box::new(PreparationRequest { + client: self.client.clone(), + target: self.target.clone(), + check: self.check.clone(), + exact: ExactPreparation::Renew { + command, + session: self.clone(), + }, + }), + }) + } +} diff --git a/crates/canopy-server/src/packs/publication/coordinator/work.rs b/crates/canopy-server/src/packs/publication/coordinator/work.rs new file mode 100644 index 0000000..5ba2252 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/coordinator/work.rs @@ -0,0 +1,257 @@ +use super::*; + +const MAINTENANCE_INLINE_BYTES: u32 = 4096; +pub(super) const MAINTENANCE_RESERVATION: u64 = 2 * MAINTENANCE_INLINE_BYTES as u64; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum PublicationClass { + Foreground, + Maintenance, +} +impl PublicationClass { + pub(super) fn index(self) -> usize { + match self { + Self::Foreground => 0, + Self::Maintenance => 1, + } + } + pub(super) fn reservation(self) -> u64 { + match self { + Self::Foreground => COMMAND_RESERVATION, + Self::Maintenance => MAINTENANCE_RESERVATION, + } + } +} + +#[derive(Debug, thiserror::Error)] +pub enum CompactionReadyError { + #[error("compaction attestation failed")] + Attestation(#[from] CatalogAttestationError), + #[error("compaction preparation inactive")] + Base(#[from] PreparationBaseError), + #[error("compaction command encoding failed")] + Codec(#[from] CodecError), + #[error("compaction command preparation failed")] + Command(#[source] Box>), +} + +#[must_use] +pub struct ReadyCatalogCompaction { + prepared: Arc, + command: PreparedCommand, +} +impl PreparedCompaction { + /// Retain the private verified input and exact SDK command through dispatch + /// and uncertain-outcome recovery. No alternate proof/identity on retry. + pub async fn ready_compaction( + self: &Arc, + identity: MutationIdentity, + ) -> Result { + let input = self.certificate().await?; + input.encode(&mut BoundedEncoder::new(MAINTENANCE_INLINE_BYTES)?)?; + let base = self.preparation_base(); + base.live_lease()?; + let (client, target, _) = base.capability(); + let command = client + .prepare_command::(target, identity, input) + .await + .map_err(|error| CompactionReadyError::Command(Box::new(error)))?; + base.live_lease()?; + Ok(ReadyCatalogCompaction { + prepared: Arc::clone(self), + command, + }) + } +} + +/// Every variant is privately prepared. Wrapping one never constructs a proof. +#[must_use] +pub enum ReadyPublication { + Push(ReadyCatalogPush), + Compaction(ReadyCatalogCompaction), + Inputs(ReadyNativeInputs), + Preparation(ReadyPreparation), +} +impl From for ReadyPublication { + fn from(ready: ReadyCatalogPush) -> Self { + Self::Push(ready) + } +} +impl From for ReadyPublication { + fn from(ready: ReadyCatalogCompaction) -> Self { + Self::Compaction(ready) + } +} +impl From for ReadyPublication { + fn from(ready: ReadyNativeInputs) -> Self { + Self::Inputs(ready) + } +} +impl From for ReadyPublication { + fn from(ready: ReadyPreparation) -> Self { + Self::Preparation(ready) + } +} +impl ReadyPublication { + /// Final work must share the lifecycle's exact session fence and clock. + /// Equal SQL tokens from an independently opened session are insufficient. + pub(in crate::packs::publication) fn belongs_to(&self, session: &PreparationSession) -> bool { + let source = match self { + Self::Push(ready) => ready.owner.session(), + Self::Compaction(ready) => &ready.prepared.preparation_base().session, + Self::Inputs(_) | Self::Preparation(_) => return false, + }; + source.target == session.target + && source.check == session.check + && source.ceiling == session.ceiling + && Arc::ptr_eq(&source.deadline, &session.deadline) + && Arc::ptr_eq(&source.fenced, &session.fenced) + } + pub(super) fn reservation(&self) -> u64 { + match self { + Self::Inputs(_) => inputs::INPUT_RESERVATION, + Self::Preparation(_) => preparation::RESERVATION, + _ => self.class().reservation(), + } + } + pub(super) fn dispatch_copy(&self) -> Self { + match self { + Self::Preparation(ready) => Self::Preparation(ready.dispatch_copy()), + Self::Push(ready) => Self::Push(ReadyCatalogPush { + owner: ready.owner.clone(), + command: ready.command.clone(), + }), + Self::Compaction(ready) => Self::Compaction(ReadyCatalogCompaction { + prepared: Arc::clone(&ready.prepared), + command: ready.command.clone(), + }), + Self::Inputs(ready) => Self::Inputs(ReadyNativeInputs { + session: ready.session.clone(), + command: ready.command.clone(), + digest: ready.digest, + }), + } + } + pub(super) fn class(&self) -> PublicationClass { + match self { + Self::Push(_) | Self::Inputs(_) | Self::Preparation(_) => PublicationClass::Foreground, + Self::Compaction(_) => PublicationClass::Maintenance, + } + } + pub(super) fn capability(&self) -> (&CellClient, &CellTarget, &LeaseCheck) { + match self { + Self::Push(ready) => ready.owner.capability(), + Self::Inputs(ready) => ready.session.capability(), + Self::Preparation(ready) => ready.capability(), + Self::Compaction(ready) => ready.prepared.preparation_base().capability(), + } + } + pub(super) fn pending(&self) -> PublicationError { + match self { + Self::Preparation(ready) => ready.pending(), + Self::Push(ready) => PublicationError::Push(InvocationError::Pending(Box::new( + ready.command.evidence().clone(), + ))), + Self::Compaction(ready) => PublicationError::Compaction(InvocationError::Pending( + Box::new(ready.command.evidence().clone()), + )), + Self::Inputs(ready) => PublicationError::Inputs(InvocationError::Pending(Box::new( + ready.command.evidence().clone(), + ))), + } + } + pub(super) async fn dispatch(self, recover: bool, fault: u8) -> DispatchResult { + let client = self.capability().0.clone(); + match self { + Self::Inputs(ready) => ready.dispatch(recover, fault).await, + Self::Preparation(ready) => ready.dispatch(recover, fault).await, + Self::Push(ready) => super::super::exact::invoke_guarded( + &client, + ready.command, + recover, + 128, + fault, + move || { + ready + .owner + .session() + .live_lease() + .map(|_| ()) + .map_err(|_| Error::Command("inactive final preparation")) + }, + ) + .await + .map(PublicationOutcome::Push) + .map_err(PublicationError::Push), + Self::Compaction(ready) => super::super::exact::invoke_guarded( + &client, + ready.command, + recover, + 128, + fault, + move || { + ready + .prepared + .preparation_base() + .live_lease() + .map(|_| ()) + .map_err(|_| Error::Command("inactive final preparation")) + }, + ) + .await + .map(PublicationOutcome::Compaction) + .map_err(PublicationError::Compaction), + } + } +} + +#[derive(Clone, Debug)] +pub enum PublicationOutcome { + Push(Committed), + Compaction(Committed), + Inputs(RegisteredNativeInputs), + Preparation(PreparationCommandOutcome), +} +#[derive(Debug, thiserror::Error)] +pub enum PublicationError { + #[error("bound preparation command: {0}")] + Preparation(#[source] InvocationError), + #[error("bound input checkpoint: {0}")] + Inputs(#[source] InvocationError), + #[error("push publication: {0}")] + Push(#[source] InvocationError), + #[error("compaction publication: {0}")] + Compaction(#[source] InvocationError), +} +impl PublicationError { + pub(super) fn disposition(&self) -> &'static str { + fn kind(error: &InvocationError) -> &'static str { + match error { + InvocationError::Rejected(_) => "rejected", + InvocationError::NotStarted(_) => "not_started", + InvocationError::Pending(_) => "pending", + InvocationError::InvalidPublishedResult { .. } => "invalid_published_result", + } + } + match self { + Self::Push(error) => kind(error), + Self::Preparation(error) => kind(error), + Self::Inputs(error) => kind(error), + Self::Compaction(error) => kind(error), + } + } + pub(super) fn uncertain(&self) -> bool { + fn unknown(error: &InvocationError) -> bool { + matches!( + error, + InvocationError::Pending(_) | InvocationError::InvalidPublishedResult { .. } + ) + } + match self { + Self::Push(error) => unknown(error), + Self::Preparation(error) => unknown(error), + Self::Inputs(error) => unknown(error), + Self::Compaction(error) => unknown(error), + } + } +} diff --git a/crates/canopy-server/src/packs/publication/exact.rs b/crates/canopy-server/src/packs/publication/exact.rs new file mode 100644 index 0000000..056f76e --- /dev/null +++ b/crates/canopy-server/src/packs/publication/exact.rs @@ -0,0 +1,86 @@ +//! Shared exact-command resolution for final publication and custody commands. +use super::*; +use cellule_runtime::{ + CellClient, Committed, InvocationError, PreparedCommand, Receipt, Resolution, + cell::executor::StoredOutcome, +}; + +pub(super) async fn resolve( + client: &CellClient, + command: PreparedCommand, + output_limit: u32, + before_execute: impl FnOnce() -> Result<(), Error> + Send, +) -> Result, InvocationError> { + let evidence = command.evidence().clone(); + match client.resolve(&evidence).await { + Ok(Resolution::Absent) => { + before_execute().map_err(InvocationError::NotStarted)?; + Box::pin(command.execute()).await + } + Ok(Resolution::Committed(outcome)) => { + let receipt = Receipt { + cell: evidence.target().cell_id(), + incarnation: evidence.incarnation(), + commit_sequence: outcome.commit_sequence(), + }; + let decoded = (|| { + let mut decoder = BoundedDecoder::new(outcome.result(), output_limit)?; + let output = C::Output::decode(&mut decoder)?; + decoder.finish()?; + Ok::<_, CodecError>(Committed { output, receipt }) + })() + .map_err(|source| InvocationError::InvalidPublishedResult { + receipt, + source: Box::new(source.into()), + })?; + match outcome { + StoredOutcome::Success { .. } => Ok(decoded), + StoredOutcome::Rejected { .. } => Err(InvocationError::Rejected(Box::new(decoded))), + } + } + // Unknown, expiration and changed incarnation never prove that an + // earlier submission failed. Keep exact evidence for logical recovery. + Ok(Resolution::Unknown | Resolution::Expired) | Err(_) => { + Err(InvocationError::Pending(Box::new(evidence))) + } + } +} + +/// Execute one retained transport copy. Only authoritative absence can cause an +/// exact recovery execution; unknown/expired evidence remains charged upstream. +pub(super) async fn invoke( + client: &CellClient, + command: PreparedCommand, + recover: bool, + output_limit: u32, + fault: u8, +) -> Result, InvocationError> { + invoke_guarded(client, command, recover, output_limit, fault, || Ok(())).await +} + +/// A local custody guard applies only before initial submission or proven +/// absence. Resolve known outcomes first, even after local custody is fenced. +pub(super) async fn invoke_guarded( + client: &CellClient, + command: PreparedCommand, + recover: bool, + output_limit: u32, + fault: u8, + before_execute: impl FnOnce() -> Result<(), Error> + Send, +) -> Result, InvocationError> { + let evidence = command.evidence().clone(); + let outcome = if fault == 1 { + Err(InvocationError::Pending(Box::new(evidence.clone()))) + } else if recover { + resolve(client, command, output_limit, before_execute).await + } else { + before_execute().map_err(InvocationError::NotStarted)?; + Box::pin(command.execute()).await + }; + if fault == 2 { + Err(InvocationError::Pending(Box::new(evidence))) + } else { + assert_ne!(fault, 3, "injected exact command panic after execution"); + outcome + } +} diff --git a/crates/canopy-server/src/packs/publication/initialization.rs b/crates/canopy-server/src/packs/publication/initialization.rs new file mode 100644 index 0000000..528758f --- /dev/null +++ b/crates/canopy-server/src/packs/publication/initialization.rs @@ -0,0 +1,175 @@ +//! Fresh empty catalog/ref publication. No SQL-ref conversion or decoded-root +//! signing adapter exists: only a privately assembled empty catalog can mint it. +use super::*; +use crate::packs::ref_state::{RefSnapshotError, RefStateSnapshot}; +use cellule_runtime::{InvocationError, primitives::sql::SqlCell}; +use tokio::time::timeout_at; + +mod publish; +pub use publish::{CheckInitializedCatalog, InitializeCatalogRefs}; +pub const INITIALIZATION_BYTES: u32 = 2048; +const INITIAL_HEAD: &str = "refs/heads/main"; + +#[derive(Debug, thiserror::Error)] +pub enum InitializationPreparationError { + #[error("initialization preparation is inactive")] + Base(#[from] PreparationBaseError), + #[error("initialization snapshot failed")] + Snapshot(#[from] RefSnapshotError), + #[error("initialization certificate failed")] + Certificate(#[from] CatalogAttestationError), + #[error("initialization capability failed")] + Capability(#[from] Error), + #[error("initialization authorization query failed")] + Query(#[source] Box>>), + #[error("initialization encoding failed")] + Codec(#[from] CodecError), + #[error("initialization requires an empty generation-zero catalog")] + Ineligible, +} + +/// Public fields permit transport only. The final command authenticates the +/// purpose-separated MAC binding; raw descriptors cannot mint this proof. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct InitialRefProof { + pub certificate: CatalogCertificate, + pub refs: RefStateSnapshotRoot, +} +fn binding(refs: RefStateSnapshotRoot) -> Result<[u8; 32], CodecError> { + let mut e = BoundedEncoder::new(128)?; + refs.encode(&mut e)?; + let mut hash = blake3::Hasher::new(); + hash.update(b"canopy.ref-initialization.v1\0"); + hash.update(&e.finish()); + Ok(*hash.finalize().as_bytes()) +} +impl InitialRefProof { + fn shape(&self) -> Result<(), CodecError> { + let data = self.certificate.data()?; + if data.compaction + || data.base.generation != 0 + || data.base.refs.is_some() + || data.object_count != 0 + || data.edge_count != 0 + || data.input_count != 0 + || data.input_checkpoint_digest.is_some() + || data.completion_digest.is_some() + || data.refs_digest != Some(binding(self.refs)?) + || self.refs.operation() != data.token.artifact_operation + { + return Err(CodecError::Invalid("invalid fresh initialization proof")); + } + Ok(()) + } +} +impl WireValue for InitialRefProof { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.shape()?; + self.certificate.encode(e)?; + self.refs.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + certificate: CatalogCertificate::decode(d)?, + refs: RefStateSnapshotRoot::decode(d)?, + }; + value.shape()?; + Ok(value) + } +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum InitializationReply { + Initialized(Box), + Denied(PreparationDenial), +} +fn initial_fact(fact: &GenerationFact) -> Result<(), CodecError> { + fact.validate()?; + if fact.generation != 1 || fact.refs.is_none() { + return Err(CodecError::Invalid("invalid initialized generation")); + } + Ok(()) +} +impl WireValue for InitializationReply { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + match self { + Self::Initialized(fact) => { + initial_fact(fact)?; + e.write_u8(0)?; + fact.encode(e) + } + Self::Denied(reason) => PreparationReply::Denied(*reason).encode(e), + } + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + match d.read_u8()? { + 0 => { + let fact = GenerationFact::decode(d)?; + initial_fact(&fact)?; + Ok(Self::Initialized(Box::new(fact))) + } + 1 => Ok(Self::Denied(PreparationDenial::Unauthorized)), + 2 => Ok(Self::Denied(PreparationDenial::Conflict)), + 3 => Ok(Self::Denied(PreparationDenial::Stale)), + 4 => Ok(Self::Denied(PreparationDenial::Expired)), + 5 => Ok(Self::Denied(PreparationDenial::Capacity)), + 6 => Ok(Self::Denied(PreparationDenial::Missing)), + _ => Err(CodecError::Invalid("invalid initialization reply")), + } + } +} + +impl PreparedCatalog { + /// Mint from the held catalog and its own store, never from caller roots. + /// The final command independently refuses non-pristine repository state. + pub async fn empty_ref_initialization( + &self, + ) -> Result { + let (_, deadline) = self.base.live_lease()?; + timeout_at(deadline, async { + if self.base().generation != 0 + || self.object_count() != 0 + || self.edge_count() != 0 + || self.input_count() != 0 + || self.input_checkpoint_digest.is_some() + { + return Err(InitializationPreparationError::Ineligible); + } + let (client, target, check) = self.base.capability(); + let sql = SqlCell::::new(client.clone(), target.clone())?; + let access = sql + .query( + None, + SqlBatch { + statements: vec![access_statement(&check.actor)], + }, + ) + .await + .map_err(|e| InitializationPreparationError::Query(Box::new(e)))?; + if !decode_access(&access.output)?.is_some_and(|role| role >= TokenScope::Admin) { + return Err(PreparationBaseError::Inactive.into()); + } + self.ensure_live()?; + let store = self.base.indexes().store(); + let refs = RefStateSnapshotRoot::upload( + &store, + self.token().artifact_operation, + RefStateSnapshot { + repository: self.token().repository, + format: self.catalog().format, + generation: 0, + default_branch: INITIAL_HEAD.into(), + root: None, + }, + ) + .await?; + let certificate = self.issue_certificate(Some(binding(refs)?), None).await?; + self.ensure_live()?; + let value = InitialRefProof { certificate, refs }; + value.shape()?; + Ok(value) + }) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } +} diff --git a/crates/canopy-server/src/packs/publication/initialization/publish.rs b/crates/canopy-server/src/packs/publication/initialization/publish.rs new file mode 100644 index 0000000..1b66a08 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/initialization/publish.rs @@ -0,0 +1,239 @@ +use super::*; +use crate::packs::publication::{ + commands::{authorized, check_pin, fact, load, matched}, + publish::{authenticate, changed, checkpoint, retention_matches}, + sql::*, +}; + +fn denied(reason: PreparationDenial) -> CommandResult { + CommandResult::Rejected(InitializationReply::Denied(reason)) +} +fn verification( + catalog: StoredCatalog, + refs: RefStateSnapshotRoot, +) -> Result<[u8; 32], CodecError> { + let mut e = BoundedEncoder::new(512)?; + catalog.encode(&mut e)?; + refs.encode(&mut e)?; + let mut hash = blake3::Hasher::new(); + hash.update(b"canopy.initialization-outcome.v1\0"); + hash.update(&e.finish()); + Ok(*hash.finalize().as_bytes()) +} +fn saved( + bytes: &[u8], + repository: [u8; 16], + format: ObjectFormat, + digest: [u8; 32], +) -> Result { + let mut d = BoundedDecoder::new(bytes, 512)?; + let fact = GenerationFact::decode(&mut d)?; + d.finish()?; + initial_fact(&fact)?; + if fact + .catalog + .is_none_or(|root| root.repository != repository || root.format != format) + { + return Err(CodecError::Invalid("invalid initialization result")); + } + if verification( + fact.catalog + .ok_or(CodecError::Invalid("missing initial catalog"))?, + fact.refs + .ok_or(CodecError::Invalid("missing initial refs"))?, + )? != digest + { + return Err(CodecError::Invalid( + "initialization roots differ from outcome", + )); + } + Ok(fact) +} +pub struct InitializeCatalogRefs; +impl Command for InitializeCatalogRefs { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 31; + const CODEC_VERSION: u32 = 1; + type Input = InitialRefProof; + type Output = InitializationReply; + fn execute( + context: &mut CommandContext<'_, '_>, + input: Self::Input, + ) -> cellule_runtime::Result> { + input.shape()?; + let Some((data, key)) = authenticate( + context, + &input.certificate, + Some(binding(input.refs)?), + None, + )? + else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + let logical = context.sql(&statement("SELECT actor,request_digest,verification_digest,result FROM catalog_initialization WHERE id=?1", vec![blob(data.token.operation)]))?; + if let Some(row) = rows(&logical)?.first() { + let [ + SqlValue::Text(actor), + request, + digest, + SqlValue::Blob(bytes), + ] = row.as_slice() + else { + return Err(Error::Command("invalid initialization outcome")); + }; + if *actor != data.actor + || fixed::<32>(request)? != data.token.request_digest + || fixed::<32>(digest)? != verification(data.catalog, input.refs)? + { + return Ok(denied(PreparationDenial::Conflict)); + } + return Ok(CommandResult::Success(InitializationReply::Initialized( + Box::new(saved( + bytes, + data.token.repository, + data.catalog.format, + fixed(digest)?, + )?), + ))); + } + // Exact recorded replay above grants no write. New initialization must + // satisfy the current actual fence, admin role, pin and pristine state. + if data.token.owner != context.owner_fence() { + return Ok(denied(PreparationDenial::Stale)); + } + let Some(format) = authorized( + context, + data.token.repository, + &data.actor, + TokenScope::Admin, + )? + else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + let Some(row) = load(context, data.token)? else { + return Ok(denied(PreparationDenial::Missing)); + }; + if !matched( + &row, + &LeaseCheck { + token: data.token, + actor: data.actor.clone(), + }, + ) { + return Ok(denied(PreparationDenial::Stale)); + } + if row.expires <= now(context.now_ms())? { + return Ok(denied(PreparationDenial::Expired)); + } + check_pin(context, &row)?; + if format != data.catalog.format + || !retention_matches(context, &data, row.generation, format)? + || fact(context, data.token.repository, format, None)? != data.base + { + return Ok(denied(PreparationDenial::Conflict)); + } + let pristine = context.sql(&statement("SELECT generation=0 AND default_branch=?1 AND NOT EXISTS(SELECT 1 FROM refs) AND NOT EXISTS(SELECT 1 FROM pushes) AND NOT EXISTS(SELECT 1 FROM catalog_compactions) AND NOT EXISTS(SELECT 1 FROM catalog_initialization) AND NOT EXISTS(SELECT 1 FROM catalog_generations WHERE generation>0) FROM ref_generation WHERE singleton=1", vec![SqlValue::Text(INITIAL_HEAD.into())]))?; + match rows(&pristine)?.first().map(Vec::as_slice) { + Some([SqlValue::Integer(1)]) => {} + Some([SqlValue::Integer(0)]) => return Ok(denied(PreparationDenial::Conflict)), + _ => return Err(Error::Command("missing initialization state")), + } + let Some(missing) = checkpoint(context, &data, &key)? else { + return Ok(denied(PreparationDenial::Conflict)); + }; + let verified_roots = verification(data.catalog, input.refs)?; + let bytes = input.certificate.bytes()?; + let digest = *blake3::hash(&bytes).as_bytes(); + let result = GenerationFact { + generation: 1, + catalog: Some(data.catalog), + refs: Some(input.refs), + certificate: Some(digest), + }; + let mut encoded = BoundedEncoder::new(512)?; + result.encode(&mut encoded)?; + let mut catalog = BoundedEncoder::new(256)?; + data.catalog.encode(&mut catalog)?; + let mut refs = BoundedEncoder::new(128)?; + input.refs.encode(&mut refs)?; + if row.expires <= now(context.now_ms())? { + return Ok(denied(PreparationDenial::Expired)); + } + // No rejection after the first write: later failures abort all roots, + // the durable outcome and checkpoint together at the Cell ACK gate. + if missing { + changed(context.sql(&statement("UPDATE catalog_operations SET attestation=?1,attestation_digest=?2 WHERE id=?3 AND attestation IS NULL", vec![blob(&bytes),blob(digest),blob(data.token.operation)]))?)?; + changed(context.sql(&statement("UPDATE catalog_leases SET attestation=?1,attestation_digest=?2 WHERE incarnation=?3 AND admission_sequence=?4 AND attestation IS NULL", vec![blob(&bytes),blob(digest),blob(data.token.owner.incarnation.as_bytes()),number(data.token.attempt)?]))?)?; + } + changed(context.sql(&statement("INSERT INTO catalog_generations(generation,catalog,certificate,refs) VALUES(1,?1,?2,?3)", vec![blob(catalog.finish()),blob(digest),blob(refs.finish())]))?)?; + changed(context.sql(&statement( + "UPDATE catalog_state SET generation=1 WHERE singleton=1 AND generation=0", + vec![], + ))?)?; + changed(context.sql(&statement("INSERT INTO catalog_initialization(singleton,id,actor,request_digest,verification_digest,result) VALUES(1,?1,?2,?3,?4,?5)", vec![blob(data.token.operation),SqlValue::Text(data.actor),blob(data.token.request_digest),blob(verified_roots),blob(encoded.finish())]))?)?; + changed(context.sql(&statement( + "DELETE FROM catalog_operations WHERE id=?1", + vec![blob(data.token.operation)], + ))?)?; + Ok(CommandResult::Success(InitializationReply::Initialized( + Box::new(result), + ))) + } +} + +/// Fresh read authorization and exact logical identity; no namespace allocation +/// or synthetic mutation receipt. The one retained result roots empty metadata. +pub struct CheckInitializedCatalog; +impl Query for CheckInitializedCatalog { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 32; + const CODEC_VERSION: u32 = 1; + type Input = BeginRequest; + type Output = Option; + fn execute( + context: &mut QueryContext<'_>, + input: Self::Input, + ) -> cellule_runtime::Result { + validate_component(&input.actor)?; + // CellClient routes this query through an already-scoped target. Check + // its persisted repository identity and current admin role here. + if !decode_access(&context.sql(&SqlBatch { + statements: vec![access_statement(&input.actor)], + })?)? + .is_some_and(|role| role >= TokenScope::Admin) + { + return Ok(None); + } + let Some(format) = identity( + &context.sql(&statement(IDENTITY, vec![]))?, + input.repository, + )? + else { + return Ok(None); + }; + let outcome = context.sql(&statement("SELECT actor,request_digest,verification_digest,result FROM catalog_initialization WHERE id=?1", vec![blob(input.operation)]))?; + let Some( + [ + SqlValue::Text(actor), + request, + digest, + SqlValue::Blob(bytes), + ], + ) = rows(&outcome)?.first().map(Vec::as_slice) + else { + if rows(&outcome)?.is_empty() { + return Ok(None); + } + return Err(Error::Command("invalid initialization lookup")); + }; + if *actor != input.actor || fixed::<32>(request)? != input.request_digest { + return Ok(None); + } + Ok(Some(saved( + bytes, + input.repository, + format, + fixed(digest)?, + )?)) + } +} diff --git a/crates/canopy-server/src/packs/publication/inputs.rs b/crates/canopy-server/src/packs/publication/inputs.rs new file mode 100644 index 0000000..c66f4ab --- /dev/null +++ b/crates/canopy-server/src/packs/publication/inputs.rs @@ -0,0 +1,962 @@ +//! Durable creating-input custody. These proofs authorize retaining/reopening +//! descriptors only; physical/canonical/closure/ref proof remains independent. +use super::*; +use super::{certificate::CertificateEnvelope, commands::*, sql::*}; +use crate::packs::{ + directory::index::{ + IndexError, + codec::{fixed as wire_fixed, read_reference, reference}, + }, + sources::{NativeInputIndex, NativeInputRoot, NativePackDescriptor}, +}; +use crate::{git_gateway::preflight::SavedPushRequest, packs::wire_request::WireRequestRoot}; +use canopy_object_storage::artifact::ArtifactStore; +use cellule_runtime::{CellClient, CellTarget, InvocationError, primitives::sql::SqlCell}; +use std::sync::Arc; + +mod custody; +#[cfg(test)] +mod limits_tests; +pub(in crate::packs) use custody::RetainedNativeInput; +pub(super) use custody::verify_digest; + +const DOMAIN: &[u8] = b"canopy.staged-native-inputs.v3\0"; +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct NativeInputCertificate(CertificateEnvelope); +#[derive(Clone, Debug, PartialEq, Eq)] +struct Inputs { + tenant: [u8; 16], + application: [u8; 16], + token: PreparationToken, + actor: String, + format: ObjectFormat, + root: Option, + wire_request: Option, + native_result: Option, + source: Option<(PreparationToken, [u8; 32])>, + previous: Option<[u8; 32]>, +} +impl Inputs { + fn validate(&self) -> Result<(), CodecError> { + validate_component(&self.actor).map_err(|_| CodecError::Invalid("input actor"))?; + if let Some(root) = self.root { + root.validate(self.format) + .map_err(|_| CodecError::Invalid("input root"))?; + codec::artifact_valid(root.operation)?; + if self.source.is_none() && root.operation != self.token.artifact_operation { + return Err(CodecError::Invalid("input root namespace")); + } + } + if let Some(wire) = self.wire_request { + wire.validate()?; + if self.source.is_none() && wire.operation() != self.token.artifact_operation { + return Err(CodecError::Invalid("wire request namespace")); + } + } + if let Some(native) = self.native_result { + native.validate()?; + if self.wire_request.is_none() + || self.source.is_none() && native.operation() != self.token.artifact_operation + { + return Err(CodecError::Invalid("native result namespace")); + } + } + if self.source.is_some_and(|(source, _)| { + source.repository != self.token.repository + || source.operation != self.token.operation + || source.request_digest != self.token.request_digest + }) { + return Err(CodecError::Invalid("input adoption source")); + } + if self.previous == Some([0; 32]) { + return Err(CodecError::Invalid("input predecessor")); + } + Ok(()) + } +} +impl WireValue for Inputs { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.validate()?; + e.write_bytes(DOMAIN)?; + e.write_bytes(&self.tenant)?; + e.write_bytes(&self.application)?; + self.token.encode(e)?; + e.write_text(&self.actor)?; + e.write_u8(self.format.bytes() as u8)?; + e.write_bool(self.root.is_some())?; + if let Some(root) = self.root { + reference(e, root)?; + } + e.write_bool(self.wire_request.is_some())?; + if let Some(root) = self.wire_request { + root.encode(e)?; + } + e.write_bool(self.native_result.is_some())?; + if let Some(root) = self.native_result { + root.encode(e)?; + } + e.write_bool(self.source.is_some())?; + if let Some((token, digest)) = self.source { + token.encode(e)?; + e.write_bytes(&digest)?; + } + e.write_bool(self.previous.is_some())?; + if let Some(previous) = self.previous { + e.write_bytes(&previous)?; + } + Ok(()) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + if d.read_bytes()? != DOMAIN { + return Err(CodecError::Invalid("input proof domain")); + } + let tenant = wire_fixed(d)?; + let application = wire_fixed(d)?; + let token = PreparationToken::decode(d)?; + let actor = d.read_text()?.into(); + let format = match d.read_u8()? { + 20 => ObjectFormat::Sha1, + 32 => ObjectFormat::Sha256, + _ => return Err(CodecError::Invalid("input format")), + }; + let root = if d.read_bool()? { + Some(read_reference(d, format)?) + } else { + None + }; + let wire_request = if d.read_bool()? { + Some(WireRequestRoot::decode(d)?) + } else { + None + }; + let native_result = if d.read_bool()? { + Some(NativeResultRoot::decode(d)?) + } else { + None + }; + let source = if d.read_bool()? { + Some((PreparationToken::decode(d)?, wire_fixed(d)?)) + } else { + None + }; + let previous = if d.read_bool()? { + Some(wire_fixed(d)?) + } else { + None + }; + let value = Self { + tenant, + application, + token, + actor, + format, + root, + wire_request, + native_result, + source, + previous, + }; + value.validate()?; + Ok(value) + } +} +impl NativeInputCertificate { + pub(super) fn scoped_check(&self, target: &CellTarget) -> Result { + let data: Inputs = self.0.data()?; + if data.tenant != *target.tenant().as_bytes() + || data.application != *target.application().as_bytes() + || crate::repository_target( + target.tenant(), + target.application(), + data.token.repository, + ) + .map_err(|_| CodecError::Invalid("input target"))? + != *target + { + return Err(CodecError::Invalid("input target")); + } + Ok(LeaseCheck { + token: data.token, + actor: data.actor, + }) + } + pub(super) fn bound_digest( + &self, + session: &PreparationSession, + ) -> Result<[u8; 32], CodecError> { + let data: Inputs = self.0.data()?; + if self.scoped_check(&session.target)? != session.check + || data.format != session.lease.format + || data.source.is_none() + { + return Err(CodecError::Invalid("bound input checkpoint context")); + } + Ok(*blake3::hash(&self.bytes()?).as_bytes()) + } + pub fn token(&self) -> Result { + Ok(self.0.data::()?.token) + } + pub fn root(&self) -> Result, CodecError> { + Ok(self.0.data::()?.root) + } + pub fn wire_request(&self) -> Result, CodecError> { + Ok(self.0.data::()?.wire_request) + } + pub fn native_result(&self) -> Result, CodecError> { + Ok(self.0.data::()?.native_result) + } + pub(super) fn checkpoint_lineage(&self) -> Result<([u8; 32], Option<[u8; 32]>), CodecError> { + Ok(( + *blake3::hash(&self.bytes()?).as_bytes(), + self.0.data::()?.previous, + )) + } + fn bytes(&self) -> Result, CodecError> { + let mut e = BoundedEncoder::new(CERTIFICATE_BYTES)?; + self.encode(&mut e)?; + Ok(e.finish()) + } + fn from_bytes(bytes: &[u8]) -> Result { + let mut d = BoundedDecoder::new(bytes, CERTIFICATE_BYTES)?; + let value = Self::decode(&mut d)?; + d.finish()?; + Ok(value) + } +} +impl WireValue for NativeInputCertificate { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.0.data::()?; + self.0.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self(CertificateEnvelope::decode(d)?); + value.0.data::()?; + Ok(value) + } +} +#[derive(Debug, thiserror::Error)] +pub enum InputCheckpointError { + #[error("input custody failed")] + Staging(#[from] StagingError), + #[error("bound input custody failed")] + Preparation(#[from] PreparationBaseError), + #[error("bound input custody query failed")] + BoundCustody(#[source] Box>>), + #[error("input index failed")] + Index(#[from] IndexError), + #[error("input proof failed")] + Codec(#[from] CodecError), + #[error("input issuer capability failed")] + Capability(#[from] Error), + #[error("input issuer query failed")] + Query(#[source] Box>>), + #[error("input custody query failed")] + Custody(#[source] Box>>), + #[error("retained input query failed")] + Retained(#[source] Box>>), +} +impl StagingContext { + pub(crate) async fn check_push_identity( + &self, + expected: &CellTarget, + input: &BeginRequest, + format: ObjectFormat, + ) -> Result<(), InputCheckpointError> { + self.ensure_live()?; + let (client, target, check) = self.capability(); + if expected != target + || input.repository != check.token.repository + || input.operation != check.token.operation + || input.request_digest != check.token.request_digest + || input.actor != check.actor + || format != self.format() + { + return Err(StagingError::Context.into()); + } + let live = client + .query::(target, None, check.clone()) + .await + .map_err(|e| InputCheckpointError::Custody(Box::new(e)))? + .output + .ok_or(StagingError::Inactive)?; + if live.token != check.token || live.format != format { + return Err(StagingError::Context.into()); + } + self.ensure_live()?; + Ok(()) + } + pub(crate) async fn push_checkpoint( + &self, + ) -> Result<(NativeInputCertificate, CellTarget, LeaseCheck, ObjectFormat), InputCheckpointError> + { + self.ensure_live()?; + let (client, target, check) = self.capability(); + let proof = client + .query::(target, None, check.clone()) + .await + .map_err(|e| InputCheckpointError::Retained(Box::new(e)))?; + let value = proof.output.ok_or(StagingError::Inactive)?; + let inputs: Inputs = value.0.data()?; + if value.scoped_check(target)? != check || inputs.format != self.format() { + return Err(StagingError::Context.into()); + } + let live = client + .query::(target, Some(proof.receipt), check.clone()) + .await + .map_err(|e| InputCheckpointError::Custody(Box::new(e)))? + .output + .ok_or(StagingError::Inactive)?; + if live.token != check.token || live.format != self.format() { + return Err(StagingError::Context.into()); + } + self.ensure_live()?; + Ok((value, target.clone(), check, self.format())) + } + /// Seals an authenticated descriptor inventory in the creating namespace. + /// The iterator is consumed incrementally; it need not own a complete list. + /// Pair existence/decoding is independently established by PhysicalVerifier. + pub async fn seal_native_inputs( + &self, + store: Arc, + inputs: I, + ) -> Result + where + I: IntoIterator, + I::IntoIter: Send, + { + self.seal_inputs(store, inputs, None).await + } + /// Retains the original encoded push alongside the native inventory in the + /// same immutable checkpoint. Neither input establishes publication proof. + pub async fn seal_push_inputs( + &self, + store: Arc, + inputs: I, + request: SavedPushRequest, + ) -> Result + where + I: IntoIterator, + I::IntoIter: Send, + { + let (_, target, check) = self.capability(); + let wire = request.scoped_root(target, &check, self.format())?; + self.seal_inputs(store, inputs, Some(wire)).await + } + async fn seal_inputs( + &self, + store: Arc, + inputs: I, + wire_request: Option, + ) -> Result + where + I: IntoIterator, + I::IntoIter: Send, + { + let token = self.token()?; + if store.repository() != token.repository { + return Err(StagingError::Context.into()); + } + let index = NativeInputIndex::new(store, self.format()); + let mut root = None; + for native in inputs { + self.ensure_live()?; + if native.operation != token.artifact_operation { + return Err(StagingError::Context.into()); + } + codec::artifact_valid(native.operation)?; + root = Some(index.insert(root, token.artifact_operation, native).await?); + } + self.sign_inputs(root, wire_request, None, None, None).await + } + /// Retains an exact immutable input root under the successor pin. Adoption + /// does not copy an input inventory or recertify native bodies. + pub async fn adopt_native_inputs( + &self, + store: Arc, + prior: &NativeInputCertificate, + ) -> Result { + self.ensure_live()?; + let (client, target, check) = self.capability(); + let (root, wire_request, native_result, source) = + adoption(client, target, &check, self.format(), store, prior).await?; + self.sign_inputs(root, wire_request, native_result, Some(source), None) + .await + } + /// Path-copy the registered input tree, preserving its request and every + /// old descriptor. Command 29 compares the exact prior checkpoint digest. + pub async fn append_native_inputs( + &self, + store: Arc, + prior: &NativeInputCertificate, + inputs: I, + ) -> Result + where + I: IntoIterator, + I::IntoIter: Send, + { + self.append_inputs(store, prior, inputs, None).await + } + pub async fn append_native_result( + &self, + store: Arc, + prior: &NativeInputCertificate, + inputs: I, + result: SavedNativeResult, + ) -> Result + where + I: IntoIterator, + I::IntoIter: Send, + { + self.append_inputs(store, prior, inputs, Some(result)).await + } + async fn append_inputs( + &self, + store: Arc, + prior: &NativeInputCertificate, + inputs: I, + result: Option, + ) -> Result + where + I: IntoIterator, + I::IntoIter: Send, + { + let (current, target, check, format) = self.push_checkpoint().await?; + if ¤t != prior || store.repository() != check.token.repository { + return Err(StagingError::Context.into()); + } + let data: Inputs = prior.0.data()?; + let native_result = if let Some(result) = result { + if data.native_result.is_some() { + return Err(StagingError::Context.into()); + } + Some(result.scoped_root( + &target, + &check, + format, + data.wire_request.ok_or(StagingError::Context)?, + prior.checkpoint_lineage()?.0, + )?) + } else { + data.native_result + }; + let index = NativeInputIndex::new(store, format); + let mut root = data.root; + for native in inputs { + self.ensure_live()?; + if native.operation != check.token.artifact_operation { + return Err(StagingError::Context.into()); + } + root = Some( + index + .insert(root, check.token.artifact_operation, native) + .await?, + ); + } + let (current, _, _, _) = self.push_checkpoint().await?; + if ¤t != prior || prior.scoped_check(&target)? != check { + return Err(StagingError::Context.into()); + } + if data.native_result.is_some() && root != data.root { + return Err(StagingError::Context.into()); + } + if root == data.root && native_result == data.native_result { + return Ok(prior.clone()); + } + self.sign_inputs( + root, + data.wire_request, + native_result, + data.source, + Some(*blake3::hash(&prior.bytes()?).as_bytes()), + ) + .await + } + async fn sign_inputs( + &self, + root: Option, + wire_request: Option, + native_result: Option, + source: Option<(PreparationToken, [u8; 32])>, + previous: Option<[u8; 32]>, + ) -> Result { + let token = self.token()?; + let (client, target, check) = self.capability(); + let live = client + .query::(target, None, check.clone()) + .await + .map_err(|e| InputCheckpointError::Custody(Box::new(e)))? + .output + .ok_or(StagingError::Inactive)?; + if live.token != token || live.format != self.format() { + return Err(StagingError::Context.into()); + } + let proof = issue_inputs( + client, + target, + Inputs { + tenant: *target.tenant().as_bytes(), + application: *target.application().as_bytes(), + token: check.token, + actor: check.actor, + format: self.format(), + root, + wire_request, + native_result, + source, + previous, + }, + ) + .await?; + self.ensure_live()?; + Ok(proof) + } +} +impl PreparationSession { + pub(crate) async fn push_checkpoint( + &self, + ) -> Result<(NativeInputCertificate, CellTarget, LeaseCheck, ObjectFormat), InputCheckpointError> + { + let (lease, _) = self.live_lease()?; + let (client, target, check) = self.capability(); + let proof = client + .query::(target, None, check.clone()) + .await + .map_err(|e| InputCheckpointError::Retained(Box::new(e)))?; + let value = proof.output.ok_or(PreparationBaseError::Inactive)?; + let inputs: Inputs = value.0.data()?; + if value.scoped_check(target)? != *check || inputs.format != lease.format { + return Err(PreparationBaseError::Context.into()); + } + self.refresh(proof.receipt).await?; + Ok((value, target.clone(), check.clone(), lease.format)) + } + /// Recover registered native inputs after a bound preparation Claim. Its + /// immutable generation floor and input custody are independent facts. + pub async fn adopt_native_inputs( + &self, + store: Arc, + prior: &NativeInputCertificate, + ) -> Result { + let (lease, _) = self.live_lease()?; + let (client, target, check) = self.capability(); + let (root, wire_request, native_result, source) = + adoption(client, target, check, lease.format, store, prior).await?; + let current = client + .query::(target, None, check.clone()) + .await + .map_err(|e| InputCheckpointError::BoundCustody(Box::new(e)))? + .output + .ok_or(PreparationBaseError::Inactive)?; + if current.token != lease.token + || current.base != lease.base + || current.format != lease.format + { + return Err(PreparationBaseError::Context.into()); + } + let proof = issue_inputs( + client, + target, + Inputs { + tenant: *target.tenant().as_bytes(), + application: *target.application().as_bytes(), + token: check.token, + actor: check.actor.clone(), + format: lease.format, + root, + wire_request, + native_result, + source: Some(source), + previous: None, + }, + ) + .await?; + self.live_lease()?; + Ok(proof) + } +} +async fn adoption( + client: &CellClient, + target: &CellTarget, + check: &LeaseCheck, + format: ObjectFormat, + store: Arc, + prior: &NativeInputCertificate, +) -> Result< + ( + Option, + Option, + Option, + (PreparationToken, [u8; 32]), + ), + InputCheckpointError, +> { + let data: Inputs = prior.0.data()?; + if store.repository() != check.token.repository + || data.token.repository != check.token.repository + || data.token.operation != check.token.operation + || data.token.request_digest != check.token.request_digest + || data.actor != check.actor + || data.format != format + || data.tenant != *target.tenant().as_bytes() + || data.application != *target.application().as_bytes() + { + return Err(StagingError::Context.into()); + } + let retained = client + .query::( + target, + None, + LeaseCheck { + token: data.token, + actor: check.actor.clone(), + }, + ) + .await + .map_err(|e| InputCheckpointError::Retained(Box::new(e)))? + .output; + if retained.as_ref() != Some(prior) { + return Err(StagingError::Inactive.into()); + } + if let Some(root) = data.root { + NativeInputIndex::new(Arc::clone(&store), format) + .validate_root(root) + .await?; + } + if let Some(root) = data.wire_request { + let record = root + .read(&store) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + if !record.matches(target, check, format) { + return Err(StagingError::Context.into()); + } + } + if let Some(root) = data.native_result { + root.check_request( + &store, + data.wire_request.ok_or(StagingError::Context)?, + target, + check, + format, + ) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + } + // The checkpoint's exact immutable root is retained by the source pin and + // checked again in the destination write. No historical leaf scan/copy is + // needed to transfer custody. Physical reconstruction must exhaust it. + Ok(( + data.root, + data.wire_request, + data.native_result, + (data.token, *blake3::hash(&prior.bytes()?).as_bytes()), + )) +} +async fn issue_inputs( + client: &CellClient, + target: &CellTarget, + data: Inputs, +) -> Result { + let result=SqlCell::::new(client.clone(),target.clone())?.query(None,statement("SELECT push_cert_seed FROM repository_identity WHERE singleton=1 AND repository_id=?1 AND object_format=?2",vec![blob(data.token.repository),SqlValue::Text(data.format.as_str().into())])).await.map_err(|e|InputCheckpointError::Query(Box::new(e)))?; + let seed = attestation::seed(&result.output)?; + Ok(NativeInputCertificate(CertificateEnvelope::seal( + &data, &seed, + )?)) +} +fn denied(reason: PreparationDenial) -> CommandResult { + CommandResult::Rejected(StagingReply::Denied(reason)) +} +pub struct RegisterStagedInputs; +impl Command for RegisterStagedInputs { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 29; + const CODEC_VERSION: u32 = 1; + type Input = NativeInputCertificate; + type Output = StagingReply; + fn execute( + context: &mut CommandContext<'_, '_>, + proof: NativeInputCertificate, + ) -> cellule_runtime::Result> { + let data: Inputs = proof.0.data()?; + if data.tenant != *context.target().tenant().as_bytes() + || data.application != *context.target().application().as_bytes() + || data.token.owner != context.owner_fence() + { + return Ok(denied(PreparationDenial::Stale)); + } + let Some(format) = authorized( + context, + data.token.repository, + &data.actor, + TokenScope::Write, + )? + else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + let seed = attestation::seed(&context.sql(&statement( + "SELECT push_cert_seed FROM repository_identity WHERE singleton=1", + vec![], + ))?)?; + if format != data.format || !proof.0.authenticated(&seed) { + return Ok(denied(PreparationDenial::Conflict)); + } + let Some(row) = load(context, data.token)? else { + return Ok(denied(PreparationDenial::Missing)); + }; + if !matched( + &row, + &LeaseCheck { + token: data.token, + actor: data.actor, + }, + ) { + return Ok(denied(PreparationDenial::Stale)); + } + if row.generation.is_some() && data.source.is_none() { + return Ok(denied(PreparationDenial::Conflict)); + } + check_pin(context, &row)?; + let now = now(context.now_ms())?; + if row.expires <= now { + return Ok(denied(PreparationDenial::Expired)); + } + let bytes = proof.bytes()?; + let existing = context.sql(&checkpoint(data.token)?)?; + let Some([old, digest, SqlValue::Integer(_)]) = rows(&existing)?.first().map(Vec::as_slice) + else { + return Err(Error::Command("input pin is absent")); + }; + match (old, digest) { + (SqlValue::Null, SqlValue::Null) => { + if data.previous.is_some() { + return Ok(denied(PreparationDenial::Conflict)); + } + if let Some((source, expected)) = data.source { + let retained = context.sql(&checkpoint(source)?)?; + let Some( + [ + SqlValue::Blob(source_bytes), + source_digest, + SqlValue::Integer(expires), + ], + ) = rows(&retained)?.first().map(Vec::as_slice) + else { + return Ok(denied(PreparationDenial::Missing)); + }; + if *expires <= now { + return Ok(denied(PreparationDenial::Expired)); + } + if fixed::<32>(source_digest)? != expected + || *blake3::hash(source_bytes).as_bytes() != expected + { + return Ok(denied(PreparationDenial::Conflict)); + } + let source_proof = NativeInputCertificate::from_bytes(source_bytes)?; + let previous: Inputs = source_proof.0.data()?; + if !source_proof.0.authenticated(&seed) + || previous.token != source + || previous.actor != row.actor + || previous.format != format + || previous.tenant != data.tenant + || previous.application != data.application + || previous.root != data.root + || previous.wire_request != data.wire_request + || previous.native_result != data.native_result + { + return Ok(denied(PreparationDenial::Conflict)); + } + } + context.sql(&statement("UPDATE catalog_leases SET input_checkpoint=?1,input_checkpoint_digest=?2 WHERE incarnation=?3 AND admission_sequence=?4", vec![blob(&bytes),blob(blake3::hash(&bytes).as_bytes()),blob(data.token.owner.incarnation.as_bytes()),number(data.token.attempt)?]))?; + } + (SqlValue::Blob(old), SqlValue::Blob(digest)) + if old == &bytes && digest.as_slice() == blake3::hash(&bytes).as_bytes() => {} + (SqlValue::Blob(old), SqlValue::Blob(digest)) + if data.previous == Some(*blake3::hash(old).as_bytes()) + && digest.as_slice() == blake3::hash(old).as_bytes() => + { + // Only the private append issuer can attest complete preservation + // of the old tree. SQL still compares the exact predecessor and + // rejects updates after Bind or beyond the bounded revision cap. + let prior = NativeInputCertificate::from_bytes(old)?; + let prior_data: Inputs = prior.0.data()?; + if row.generation.is_some() + || !prior.0.authenticated(&seed) + || prior_data.token != data.token + || prior_data.actor != row.actor + || prior_data.tenant != data.tenant + || prior_data.application != data.application + || prior_data.format != data.format + || prior_data.source != data.source + || prior_data.wire_request != data.wire_request + || prior_data.native_result.is_some() + || (data.root == prior_data.root + && data.native_result == prior_data.native_result) + || (data.root.is_none() && prior_data.root.is_some()) + || data + .native_result + .is_some_and(|root| root.operation() != data.token.artifact_operation) + || prior_data.root.is_some_and(|old| { + data.root.is_none_or(|new| { + new.record_count < old.record_count + || new.object_count < old.object_count + }) + }) + { + return Ok(denied(PreparationDenial::Conflict)); + } + let changed=context.sql(&statement("UPDATE catalog_leases SET input_checkpoint=?1,input_checkpoint_digest=?2,input_checkpoint_previous_digest=?3,input_checkpoint_revision=input_checkpoint_revision+1 WHERE incarnation=?4 AND admission_sequence=?5 AND input_checkpoint_digest=?3 AND input_checkpoint_revision<256",vec![blob(&bytes),blob(blake3::hash(&bytes).as_bytes()),blob(digest),blob(data.token.owner.incarnation.as_bytes()),number(data.token.attempt)?]))?; + if changed.first().is_none_or(|set| set.rows_affected != 1) { + return Ok(denied(PreparationDenial::Capacity)); + } + } + _ => return Ok(denied(PreparationDenial::Conflict)), + } + Ok(CommandResult::Success(StagingReply::Granted(Box::new( + StagingLease { + token: row.token, + format, + observed_at_ms: now, + expires_at_ms: row.expires, + }, + )))) + } +} +fn checkpoint(token: PreparationToken) -> cellule_runtime::Result { + Ok(statement( + "SELECT input_checkpoint,input_checkpoint_digest,expires_at_ms FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2 AND operation=?3 AND owner_epoch=?4 AND artifact_operation=?5", + vec![ + blob(token.owner.incarnation.as_bytes()), + number(token.attempt)?, + blob(token.operation), + blob(token.owner.epoch.to_be_bytes()), + blob(token.artifact_operation), + ], + )) +} +pub struct CheckStagedInputs; +impl Query for CheckStagedInputs { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 30; + const CODEC_VERSION: u32 = 1; + type Input = LeaseCheck; + type Output = Option; + fn execute( + context: &mut QueryContext<'_>, + check: LeaseCheck, + ) -> cellule_runtime::Result { + validate_component(&check.actor)?; + if !decode_access(&context.sql(&SqlBatch { + statements: vec![access_statement(&check.actor)], + })?)? + .is_some_and(|r| r >= TokenScope::Write) + { + return Ok(None); + } + let Some(format) = identity( + &context.sql(&statement(IDENTITY, vec![]))?, + check.token.repository, + )? + else { + return Ok(None); + }; + let Some(row) = operation( + &context.sql(&statement(OPERATION, vec![blob(check.token.operation)]))?, + check.token.repository, + check.token.operation, + )? + else { + return Ok(None); + }; + if row.actor != check.actor || row.token.request_digest != check.token.request_digest { + return Ok(None); + } + let results = context.sql(&checkpoint(check.token)?)?; + let Some([SqlValue::Blob(bytes), digest, SqlValue::Integer(expires)]) = + rows(&results)?.first().map(Vec::as_slice) + else { + return Ok(None); + }; + if *expires <= now(context.now_ms())? + || fixed::<32>(digest)? != *blake3::hash(bytes).as_bytes() + { + return Ok(None); + } + let proof = NativeInputCertificate::from_bytes(bytes)?; + let data: Inputs = proof.0.data()?; + let seed = attestation::seed(&context.sql(&statement( + "SELECT push_cert_seed FROM repository_identity WHERE singleton=1", + vec![], + ))?)?; + if !proof.0.authenticated(&seed) + || data.token != check.token + || data.actor != check.actor + || data.format != format + || crate::repository_target( + cellule_runtime::TenantId::from_bytes(data.tenant), + cellule_runtime::ApplicationId::from_bytes(data.application), + data.token.repository, + )? + .cell_id() + != context.cell_id() + { + return Ok(None); + } + Ok(Some(proof)) + } +} + +/// Command-local custody barrier for newly published catalogs using borrowed +/// native inputs. Completed outcome replay precedes this check upstream. +pub(super) fn retention_matches( + context: &CommandContext<'_, '_>, + data: &certificate::CertificateData, +) -> cellule_runtime::Result { + let Some(expected) = data.input_checkpoint_digest else { + return Ok(true); + }; + let retained = context.sql(&checkpoint(data.token)?)?; + let Some([SqlValue::Blob(bytes), digest, SqlValue::Integer(expires)]) = + rows(&retained)?.first().map(Vec::as_slice) + else { + return Ok(false); + }; + if *expires <= now(context.now_ms())? + || fixed::<32>(digest)? != expected + || *blake3::hash(bytes).as_bytes() != expected + { + return Ok(false); + } + let proof = NativeInputCertificate::from_bytes(bytes)?; + let inputs: Inputs = proof.0.data()?; + let seed = attestation::seed(&context.sql(&statement( + "SELECT push_cert_seed FROM repository_identity WHERE singleton=1", + vec![], + ))?)?; + Ok(proof.0.authenticated(&seed) + && inputs.token == data.token + && inputs.actor == data.actor + && inputs.format == data.catalog.format + && inputs.tenant == data.tenant + && inputs.application == data.application) +} + +/// A durable receipt is not usable custody. Check the exact independent input +/// pin, then observe the same live bound attempt/floor with a fresh clock. +pub(super) async fn observe_bound_registration( + session: &PreparationSession, + digest: [u8; 32], + minimum: cellule_runtime::Receipt, +) -> Result<(), InputCheckpointError> { + let current = session + .client + .query::(&session.target, Some(minimum), session.check.clone()) + .await + .map_err(|e| InputCheckpointError::Retained(Box::new(e)))? + .output + .ok_or(PreparationBaseError::Inactive)?; + if *blake3::hash(¤t.bytes()?).as_bytes() != digest { + return Err(PreparationBaseError::Context.into()); + } + session.refresh(minimum).await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/inputs/custody.rs b/crates/canopy-server/src/packs/publication/inputs/custody.rs new file mode 100644 index 0000000..a4951e3 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/inputs/custody.rs @@ -0,0 +1,93 @@ +//! Private bridge from live SQL checkpoint custody to an exact native witness. +use super::*; +use crate::packs::closure::{ClosureContext, ClosureError}; + +pub(in crate::packs) struct RetainedNativeInput { + native: NativePackDescriptor, + context: ClosureContext, + session: PreparationSession, + digest: [u8; 32], +} +impl RetainedNativeInput { + pub(in crate::packs::publication) async fn open( + base: &PreparationBaseResolver, + native: NativePackDescriptor, + ) -> Result { + let (lease, _) = base.live_lease()?; + if native.repository != lease.token.repository || native.format != lease.format { + return Err(PreparationBaseError::Context.into()); + } + let proof = read_checkpoint(base).await?; + let inputs: Inputs = proof.0.data()?; + let indexes = base.indexes(); + let key = crate::packs::directory::SegmentKey { + operation: native.operation, + digest: native.pack.digest, + }; + if indexes.inputs().find(inputs.root, key).await? != Some(native) { + return Err(PreparationBaseError::Context.into()); + } + let digest = *blake3::hash(&proof.bytes()?).as_bytes(); + verify_digest(base, digest).await?; + Ok(Self { + native, + context: base.context(), + session: base.session.clone(), + digest, + }) + } + pub(in crate::packs::publication) fn digest(&self) -> [u8; 32] { + self.digest + } + pub(in crate::packs) fn authorize( + &self, + context: ClosureContext, + native: NativePackDescriptor, + ) -> Result<(), ClosureError> { + if self.context != context || self.native != native || self.session.live_lease().is_err() { + return Err(ClosureError::Integrity); + } + Ok(()) + } +} +async fn read_checkpoint( + base: &PreparationBaseResolver, +) -> Result { + let (lease, _) = base.live_lease()?; + let (client, target, check) = base.capability(); + let proof = client + .query::(target, None, check.clone()) + .await + .map_err(|e| InputCheckpointError::Retained(Box::new(e)))? + .output + .ok_or(PreparationBaseError::Inactive)?; + let inputs: Inputs = proof.0.data()?; + if proof.scoped_check(target)? != *check || inputs.format != lease.format { + return Err(PreparationBaseError::Context.into()); + } + Ok(proof) +} +pub(in crate::packs::publication) async fn verify_digest( + base: &PreparationBaseResolver, + digest: [u8; 32], +) -> Result<(), InputCheckpointError> { + let proof = read_checkpoint(base).await?; + if *blake3::hash(&proof.bytes()?).as_bytes() != digest { + return Err(PreparationBaseError::Context.into()); + } + let (client, target, check) = base.capability(); + let live = client + .query::(target, None, check.clone()) + .await + .map_err(|e| InputCheckpointError::BoundCustody(Box::new(e)))? + .output + .ok_or(PreparationBaseError::Inactive)?; + if live.token != base.context_token() + || live.base != base.retention_floor() + || live.format != base.context().format + { + return Err(PreparationBaseError::Context.into()); + } + base.live_lease()?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/inputs/limits_tests.rs b/crates/canopy-server/src/packs/publication/inputs/limits_tests.rs new file mode 100644 index 0000000..ef38405 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/inputs/limits_tests.rs @@ -0,0 +1,77 @@ +use super::*; +use crate::packs::{ + directory::{SegmentKey, index::NodeRef}, + input_artifact::StoredInputRoot, +}; +use canopy_object_storage::artifact::ArtifactDescriptor; + +#[test] +fn adopted_result_append_with_all_roots_and_maximum_actor_fits_checkpoint_envelope() +-> Result<(), CodecError> { + let mut operation = *b"CANOPY01\0\0\0\0\0\0\0\x01"; + let artifact = ArtifactDescriptor { + size: 1, + digest: [1; 32], + manifest_digest: [2; 32], + }; + let stored = StoredInputRoot { + operation, + artifact, + }; + let mut e = BoundedEncoder::new(256)?; + stored.encode(&mut e)?; + let bytes = e.finish(); + let wire_request = WireRequestRoot::decode(&mut BoundedDecoder::new(&bytes, 256)?)?; + let native_result = NativeResultRoot::decode(&mut BoundedDecoder::new(&bytes, 256)?)?; + let root = NativeInputRoot { + operation, + artifact, + height: 0, + first_key: SegmentKey { + operation, + digest: [3; 32], + }, + last_key: SegmentKey { + operation, + digest: [4; 32], + }, + record_count: i64::MAX as u64, + object_count: i64::MAX as u64, + }; + let mut repository = [5; 16]; + repository[6] = 0x45; + repository[8] = 0x85; + assert!(crate::validate_repository_id(repository).is_ok()); + let token = PreparationToken { + repository, + operation: [6; 16], + artifact_operation: operation, + request_digest: [7; 32], + owner: OwnerFence { + incarnation: IncarnationId::from_bytes([8; 16]), + epoch: u64::MAX, + }, + attempt: i64::MAX as u64, + }; + operation[15] = 2; + let mut current = token; + current.artifact_operation = operation; + current.owner.epoch -= 1; + let inputs = Inputs { + tenant: [9; 16], + application: [10; 16], + token: current, + actor: "a".repeat(64), + format: ObjectFormat::Sha256, + root: Some(NodeRef { operation, ..root }), + wire_request: Some(wire_request), + native_result: Some(native_result), + source: Some((token, [11; 32])), + previous: Some([12; 32]), + }; + let proof = NativeInputCertificate(CertificateEnvelope::seal(&inputs, &[13; 32])?); + let bytes = proof.bytes()?; + assert!(bytes.len() <= CERTIFICATE_BYTES as usize); + assert_eq!(NativeInputCertificate::from_bytes(&bytes)?, proof); + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/mod.rs b/crates/canopy-server/src/packs/publication/mod.rs new file mode 100644 index 0000000..a1eb7fe --- /dev/null +++ b/crates/canopy-server/src/packs/publication/mod.rs @@ -0,0 +1,237 @@ +//! Fenced preparation and retained generation facts in the Repository Cell. +//! The fresh schema is selected with the final producer/reader hard cutover; +//! these commands are not registered on the legacy repository serving path. +use super::{catalog::StoredCatalog, ref_state::RefStateSnapshotRoot}; +use crate::{ + ObjectFormat, RepositoryModule, + access::{access_statement, decode_access}, + directory::{TokenScope, validate_component}, +}; +use cellule_runtime::{ + CellModule, Command, Error, Query, RegistryBuilder, + codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue}, + identity::IncarnationId, + primitives::sql::{SqlBatch, SqlResultSet, SqlStatement, SqlValue}, + registry::{CommandContext, CommandResult, OwnerFence, QueryContext}, +}; +mod session; +pub use session::PreparationSession; +mod base; +pub use base::{PreparationBaseError, PreparationBaseResolver}; +mod certificate; +pub(in crate::packs) mod codec; +pub use certificate::{ + AttestationOutcome, CERTIFICATE_BYTES, CatalogCertificate, RegisteredCatalog, +}; +mod attestation; +pub use attestation::{CatalogAttestationError, RegisterCatalogAttestation}; +mod prepare; +pub use prepare::{CatalogPreparation, CatalogPreparationError, PreparedCatalog}; +pub(in crate::packs) mod ref_proof; +pub use ref_proof::{RefProofError, RefPublicationProof}; +mod initialization; +pub use initialization::{ + CheckInitializedCatalog, INITIALIZATION_BYTES, InitialRefProof, InitializationPreparationError, + InitializationReply, InitializeCatalogRefs, +}; +mod ref_snapshot; +pub use ref_snapshot::{PreparedRefSnapshot, RefSnapshotPreparationError}; +mod ref_policy; +pub use ref_policy::{ + CheckRefPolicyGuard, MAX_REF_POLICY_GUARDS, MAX_REF_POLICY_WATCHES, PreparedRefPolicyGuard, + REF_POLICY_PAGE_BYTES, REF_POLICY_PAGE_UPDATES, ReapRefPolicyGuard, RefPolicyIntent, + RefPolicyLookup, RefPolicyPage, RefPolicyPreparation, RefPolicyPreparationError, + RefPolicyProgress, RefPolicyReap, RefPolicyReapReply, RefPolicyReply, RefRootPublicationProof, + RegisterRefPolicyPage, +}; +mod publish; +pub use publish::{PublicationReply, PublishCatalogRefs, PublishedRefs}; +mod outcome; +pub use outcome::OutcomeCertificate; +mod completion; +mod coordinator; +pub use completion::{ + CatalogCompletionReply, CatalogPushCompletion, CatalogPushResponseError, CheckCompletedPush, + CompleteCatalogPush, CompletedCatalogPush, CompletionCatalogProof, PushCompletionProofError, + PushCompletionRequest, SignedPushAnnotation, replay_push_response, +}; +pub use coordinator::{ + CompactionReadyError, NativeInputReadyError, PreparationCommandKind, PreparationCommandOutcome, + PreparationReadyError, PublicationAdmissionFailure, PublicationClass, PublicationCoordinator, + PublicationError, PublicationLimits, PublicationOutcome, PublicationScheduleError, + PublicationState, PublicationStats, PublicationTicket, ReadyCatalogCompaction, + ReadyCatalogPush, ReadyNativeInputs, ReadyPreparation, ReadyPublication, + RegisteredNativeInputs, +}; +mod commands; +mod compaction; +pub use compaction::{ + CheckCompletedCompaction, CompactionLimits, CompactionPlanner, CompactionPolicy, + CompactionPressure, CompactionReply, CompactionSource, PreparedCompaction, + PublishCatalogCompaction, PublishedCompaction, +}; +mod exact; +mod staging_service; +pub use staging_service::{ + ReadyStaging, StagedInputsTicket, StagedPublicationFailure, StagedPublicationTicket, + StagingBound, StagingContext, StagingCoordinator, StagingError, StagingLimits, StagingState, + StagingStats, StagingTask, StagingTicket, +}; +mod inputs; +mod native_result; +mod root_completion; +pub(in crate::packs) use inputs::RetainedNativeInput; +pub use native_result::{NativeResultError, NativeResultRoot, SavedNativeResult}; +pub use root_completion::{ + NativeOutcomeRoot, ROOT_COMPLETION_BYTES, RootCompletionPreparationError, RootPushCompletion, + RootPushOutcomes, RootSignedPushFact, +}; +mod staging; +pub use inputs::{ + CheckStagedInputs, InputCheckpointError, NativeInputCertificate, RegisterStagedInputs, +}; +pub use staging::{ + BeginStaging, BindStaging, CheckStaging, ClaimStaging, RenewStaging, StagingLease, StagingReply, +}; +mod sql; +pub use commands::{ + AbortPreparation, BeginPreparation, CheckPreparation, CheckPreparationFrontier, + ClaimPreparation, ReapPreparation, RenewPreparation, +}; + +pub const SCHEMA: &str = concat!( + include_str!("schema.sql"), + include_str!("ref_policy/schema.sql") +); +pub const MAX_OPERATIONS: u64 = 1024; +pub const MAX_GENERATION_LEASES: u64 = 4096; +/// Includes the reserved empty generation. Old eligible facts are reaped; +/// removing a local SQL fact never authorizes deleting remote artifacts. +pub const MAX_RETAINED_GENERATIONS: u64 = 8192; +pub const MAX_LEASE_MS: u64 = 300_000; +pub const DEFAULT_LEASE_MS: u64 = 60_000; +pub const REAP_ROWS: u64 = 512; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct PreparationToken { + pub repository: [u8; 16], + pub operation: [u8; 16], + /// Durable creating namespace, independent of the logical request ID. + /// Allocated once by Begin/Claim; never supplied by a product client. + pub artifact_operation: [u8; 16], + pub request_digest: [u8; 32], + pub owner: OwnerFence, + /// Admitted execution sequence, not a counter reset by record pruning. + pub attempt: u64, +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct GenerationFact { + pub generation: u64, + pub catalog: Option, + /// Ref metadata retained under this same immutable catalog generation. + pub refs: Option, + pub certificate: Option<[u8; 32]>, +} +impl GenerationFact { + fn validate(self) -> Result<(), CodecError> { + if self.generation > i64::MAX as u64 { + return Err(CodecError::Invalid("invalid catalog generation")); + } + match (self.generation, self.catalog, self.certificate) { + (0, None, None) if self.refs.is_none() => Ok(()), + (1.., Some(catalog), Some(_)) => catalog + .validate() + .map_err(|_| CodecError::Invalid("invalid generation catalog")), + _ => Err(CodecError::Invalid("incomplete generation fact")), + } + } +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct PreparationLease { + pub token: PreparationToken, + pub base: GenerationFact, + pub format: ObjectFormat, + pub observed_at_ms: i64, + pub expires_at_ms: i64, +} +/// One authorized snapshot of the original attempt and latest catalog. The +/// attempt's immutable base is a retention floor: every subsequent generation +/// stays retained until its independent pin is reaped. This query result grants +/// neither canonical reconciliation nor publication authority. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct PreparationFrontier { + pub lease: PreparationLease, + pub current: GenerationFact, +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum PreparationDenial { + Unauthorized, + Conflict, + Stale, + Expired, + Capacity, + Missing, +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum PreparationReply { + Granted(Box), + Denied(PreparationDenial), +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct BeginRequest { + pub repository: [u8; 16], + pub operation: [u8; 16], + pub request_digest: [u8; 32], + pub actor: String, + pub lease_ms: u64, +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct LeaseCheck { + pub token: PreparationToken, + pub actor: String, +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct LeaseRequest { + pub check: LeaseCheck, + pub lease_ms: u64, +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct MaintenanceRequest { + pub repository: [u8; 16], + pub actor: String, + pub owner: OwnerFence, +} + +/// Register on the fresh RepositoryModule only, with bounded descriptors for +/// command IDs 11..14/16..19/22/24..26/28..29/31/33/35 and query IDs +/// 15/20..21/23/27/30/32/34, plus the existing trusted SQL query. No separate +/// Cell or compatibility API. +pub fn register(registry: &mut RegistryBuilder) -> cellule_runtime::Result<()> { + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_query::()?; + registry.bind_query::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_query::()?; + registry.bind_command::()?; + registry.bind_query::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_command::()?; + registry.bind_query::()?; + registry.bind_query::()?; + registry.bind_query::()?; + registry.bind_query::() +} +#[cfg(test)] +mod tests; diff --git a/crates/canopy-server/src/packs/publication/native_result.rs b/crates/canopy-server/src/packs/publication/native_result.rs new file mode 100644 index 0000000..d381559 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/native_result.rs @@ -0,0 +1,371 @@ +//! Authenticated native completion custody, before canonical/final publication. +//! Artifact decoding cannot create a native witness without registered custody. +use super::*; +use crate::packs::{ + input_artifact::{INPUT_ROOT_BYTES, StoredInputRoot}, + wire_request::{WireRequestError, WireRequestRoot}, +}; +use crate::{ + directory::DirectoryCell, git_http::GitHttpResponse, git_input::InputError, + push::VerifiedPushCertificate, +}; +use canopy_object_storage::artifact::{ + ArtifactDescriptor, ArtifactKey, ArtifactKind, ArtifactStore, +}; +use cellule_ltx::DiskBudget; +use cellule_runtime::CellTarget; +use std::path::Path; + +mod codec; +mod plan; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct NativeResultRoot(StoredInputRoot); +impl NativeResultRoot { + pub fn operation(self) -> [u8; 16] { + self.0.operation + } + pub fn artifact(self) -> ArtifactDescriptor { + self.0.artifact + } + pub(super) fn validate(self) -> Result<(), CodecError> { + self.0.validate(INPUT_ROOT_BYTES) + } + pub(super) async fn read( + self, + store: &ArtifactStore, + ) -> Result { + let record: ResultRecord = self.0.read(store, INPUT_ROOT_BYTES).await?; + if record.operation != self.operation() { + return Err(NativeResultError::Context); + } + Ok(record) + } + pub(super) async fn check_request( + self, + store: &ArtifactStore, + request: WireRequestRoot, + target: &CellTarget, + check: &LeaseCheck, + format: ObjectFormat, + ) -> Result<(), NativeResultError> { + let record = self.read(store).await?; + if record.request != request || !request.read(store).await?.matches(target, check, format) { + return Err(NativeResultError::Context); + } + Ok(()) + } +} +/// Only the native-result retention factory can supply a checkpoint attachment. +pub struct SavedNativeResult { + root: NativeResultRoot, + target: CellTarget, + check: LeaseCheck, + format: ObjectFormat, + request: WireRequestRoot, + previous: [u8; 32], +} +impl SavedNativeResult { + pub(super) fn scoped_root( + self, + target: &CellTarget, + check: &LeaseCheck, + format: ObjectFormat, + request: WireRequestRoot, + previous: [u8; 32], + ) -> Result { + if self.target != *target + || self.check != *check + || self.format != format + || self.request != request + || self.previous != previous + || self.root.operation() != check.token.artifact_operation + { + return Err(CodecError::Invalid("native result custody context")); + } + Ok(self.root) + } +} +pub(super) struct ResultRecord { + pub(super) operation: [u8; 16], + request: WireRequestRoot, + pub(super) response: GitHttpResponse, + plan: Option, + options: Vec, + signed: Option>, +} +#[derive(Debug, thiserror::Error)] +pub enum NativeResultError { + #[error("native result metadata transport failed")] + InputRoot(#[from] crate::packs::InputRootError), + #[error("native result codec failed")] + Codec(#[from] CodecError), + #[error("native result artifact failed")] + Artifact(#[from] canopy_object_storage::artifact::ArtifactError), + #[error("native result root failed")] + Root(#[from] WireRequestError), + #[error("native result spool failed")] + Input(#[from] InputError), + #[error("native result custody failed")] + Checkpoint(#[from] InputCheckpointError), + #[error("native result plan failed")] + Refs(#[from] RefProofError), + #[error("native result report failed")] + Report(#[from] crate::push::PushError), + #[error("native result hashing task failed")] + Task(#[from] tokio::task::JoinError), + #[error("native signer lookup failed")] + Signers(#[source] Box>>), + #[error("native result context differs or signing key is no longer authorized")] + Context, +} +pub(super) fn body_key(operation: [u8; 16], body: ArtifactDescriptor) -> ArtifactKey { + ArtifactKey { + operation, + binding_digest: body.digest, + kind: ArtifactKind::InputBody, + } +} +pub(super) async fn retain_body( + store: &ArtifactStore, + operation: [u8; 16], + body: Vec, +) -> Result { + let (body, digest) = tokio::task::spawn_blocking(move || { + let digest = *blake3::hash(&body).as_bytes(); + (body, digest) + }) + .await?; + Ok(store + .put( + ArtifactKey { + operation, + binding_digest: digest, + kind: ArtifactKind::InputBody, + }, + body.len() as u64, + digest, + &mut body.as_slice(), + ) + .await?) +} +async fn reopen_body( + store: &ArtifactStore, + operation: [u8; 16], + body: ArtifactDescriptor, +) -> Result, NativeResultError> { + if body.size > crate::push::MAX_RESPONSE_BYTES as u64 { + return Err(CodecError::Limit.into()); + } + let mut reader = store.read(body_key(operation, body), body).await?; + let mut bytes = Vec::with_capacity(body.size as usize); + while let Some(part) = reader.next().await? { + bytes.extend_from_slice(&part); + } + Ok(bytes) +} +fn validate_plan( + plan: Option<&crate::PushPlan>, + check: &LeaseCheck, + format: ObjectFormat, +) -> Result<(), NativeResultError> { + if let Some(plan) = plan { + if plan.actor != check.actor { + return Err(NativeResultError::Context); + } + super::ref_proof::shape(plan, format)?; + } + Ok(()) +} +impl StagingContext { + /// Preserve exact native bytes and versioned intent. Canonical verification + /// and final authority/CAS still gate acknowledgement independently. + pub async fn retain_native_result( + &self, + store: &ArtifactStore, + prior: &NativeInputCertificate, + request: PushCompletionRequest, + directory: &Path, + budget: &DiskBudget, + ) -> Result { + let (current, target, check, format) = self.push_checkpoint().await?; + if current != *prior + || store.repository() != check.token.repository + || prior.native_result()?.is_some() + { + return Err(NativeResultError::Context); + } + let wire = prior.wire_request()?.ok_or(NativeResultError::Context)?; + if !wire.read(store).await?.matches(&target, &check, format) { + return Err(NativeResultError::Context); + } + let signed = + super::completion::scoped_signed_annotation(&target, &check, request.certificate)?; + super::completion::validate_payload(&request.response, &request.options, signed.as_ref())?; + validate_plan(request.plan.as_ref(), &check, format)?; + crate::push::report::publication_matches(&request.response, request.plan.as_ref())?; + let operation = check.token.artifact_operation; + let GitHttpResponse { + status, + headers, + body, + } = request.response; + let response = GitHttpResponse { + status, + headers, + body: retain_body(store, operation, body).await?, + }; + let signed = if let Some(signed) = signed { + Some(SignedPushAnnotation { + body: retain_body(store, operation, signed.body).await?, + signer: signed.signer, + key: signed.key, + }) + } else { + None + }; + let plan = if let Some(plan) = request.plan { + Some(plan::retain(plan, store, operation, directory, budget).await?) + } else { + None + }; + let record = ResultRecord { + operation, + request: wire, + response, + plan, + options: request.options, + signed, + }; + let root = NativeResultRoot( + StoredInputRoot::upload(store, operation, &record, INPUT_ROOT_BYTES).await?, + ); + let (current, _, _, _) = self.push_checkpoint().await?; + if current != *prior { + return Err(NativeResultError::Context); + } + Ok(SavedNativeResult { + root, + target, + check, + format, + request: wire, + previous: prior.checkpoint_lineage()?.0, + }) + } + pub async fn reopen_native_result( + &self, + store: &ArtifactStore, + directory: &Path, + budget: &DiskBudget, + signers: Option<&DirectoryCell>, + ) -> Result { + let (proof, target, check, format) = self.push_checkpoint().await?; + let result = reopen( + &proof, + (&target, &check, format), + store, + directory, + budget, + signers, + ) + .await?; + let (current, _, _, _) = self.push_checkpoint().await?; + if current != proof { + return Err(NativeResultError::Context); + } + Ok(result) + } +} +impl PreparationSession { + pub async fn reopen_native_result( + &self, + store: &ArtifactStore, + directory: &Path, + budget: &DiskBudget, + signers: Option<&DirectoryCell>, + ) -> Result { + let (proof, target, check, format) = self.push_checkpoint().await?; + let result = reopen( + &proof, + (&target, &check, format), + store, + directory, + budget, + signers, + ) + .await?; + let (current, _, _, _) = self.push_checkpoint().await?; + if current != proof { + return Err(NativeResultError::Context); + } + Ok(result) + } +} +async fn reopen( + proof: &NativeInputCertificate, + custody: (&CellTarget, &LeaseCheck, ObjectFormat), + store: &ArtifactStore, + directory: &Path, + budget: &DiskBudget, + signers: Option<&DirectoryCell>, +) -> Result { + let (target, check, format) = custody; + let root = proof.native_result()?.ok_or(NativeResultError::Context)?; + let wire = proof.wire_request()?.ok_or(NativeResultError::Context)?; + let record = root.read(store).await?; + if record.request != wire || !wire.read(store).await?.matches(target, check, format) { + return Err(NativeResultError::Context); + } + let plan = if let Some(plan) = record.plan { + Some(plan::reopen(store, record.operation, plan, directory, budget).await?) + } else { + None + }; + validate_plan(plan.as_ref(), check, format)?; + let response = GitHttpResponse { + status: record.response.status, + headers: record.response.headers, + body: reopen_body(store, record.operation, record.response.body).await?, + }; + let signed = if let Some(signed) = record.signed { + Some(SignedPushAnnotation { + body: reopen_body(store, record.operation, signed.body).await?, + signer: signed.signer, + key: signed.key, + }) + } else { + None + }; + super::completion::validate_payload(&response, &record.options, signed.as_ref())?; + crate::push::report::publication_matches(&response, plan.as_ref())?; + let certificate = if let Some(signed) = signed { + let directory = signers + .filter(|directory| directory.matches_repository_scope(target)) + .ok_or(NativeResultError::Context)?; + let keys = directory + .push_signers(&check.actor) + .await + .map_err(|error| NativeResultError::Signers(Box::new(error)))?; + if signed.signer != check.actor || !keys.iter().any(|key| key.fingerprint() == signed.key) { + return Err(NativeResultError::Context); + } + // This constructor is reached only from exact MAC-registered custody, + // complete authenticated bodies and fresh scoped key authorization. + Some(VerifiedPushCertificate { + target: target.clone(), + request_digest: check.token.request_digest, + body: signed.body, + signer: signed.signer, + key: signed.key, + }) + } else { + None + }; + Ok(PushCompletionRequest { + plan, + response, + options: record.options, + certificate, + }) +} diff --git a/crates/canopy-server/src/packs/publication/native_result/codec.rs b/crates/canopy-server/src/packs/publication/native_result/codec.rs new file mode 100644 index 0000000..a79691b --- /dev/null +++ b/crates/canopy-server/src/packs/publication/native_result/codec.rs @@ -0,0 +1,126 @@ +use super::*; +use crate::packs::directory::index::codec::{artifact, fixed, read_artifact}; +const DOMAIN: &[u8] = b"canopy.native-result.v1\0"; +impl WireValue for NativeResultRoot { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.validate()?; + self.0.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self(StoredInputRoot::decode(d)?); + value.validate()?; + Ok(value) + } +} +impl ResultRecord { + fn validate(&self) -> Result<(), CodecError> { + super::super::codec::artifact_valid(self.operation)?; + self.request.validate()?; + if !(100..=599).contains(&self.response.status) + || self.response.body.size > crate::push::MAX_RESPONSE_BYTES as u64 + || self.response.body.manifest_digest == [0; 32] + || !crate::push::valid_options(&self.options) + || self.plan.is_some_and(|plan| { + plan.size == 0 + || plan.size > super::plan::MAX_PLAN_BYTES + || plan.manifest_digest == [0; 32] + }) + || self.signed.as_ref().is_some_and(|signed| { + signed.body.size == 0 + || signed.body.size > crate::push::MAX_RESPONSE_BYTES as u64 + || signed.body.manifest_digest == [0; 32] + || validate_component(&signed.signer).is_err() + || signed.key.is_empty() + || signed.key.len() > 4096 + || signed.key.chars().any(char::is_control) + }) + { + return Err(CodecError::Invalid("native result metadata")); + } + Ok(()) + } +} +impl WireValue for ResultRecord { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.validate()?; + e.write_bytes(DOMAIN)?; + e.write_bytes(&self.operation)?; + self.request.encode(e)?; + e.write_u32(self.response.status.into())?; + e.write_count(self.response.headers.len())?; + for (name, value) in &self.response.headers { + e.write_text(name)?; + e.write_text(value)?; + } + artifact(e, self.response.body)?; + e.write_bool(self.plan.is_some())?; + if let Some(plan) = self.plan { + artifact(e, plan)?; + } + e.write_count(self.options.len())?; + for option in &self.options { + e.write_text(option)?; + } + e.write_bool(self.signed.is_some())?; + if let Some(signed) = &self.signed { + artifact(e, signed.body)?; + e.write_text(&signed.signer)?; + e.write_text(&signed.key)?; + } + Ok(()) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + if d.read_bytes()? != DOMAIN { + return Err(CodecError::Invalid("native result domain")); + } + let operation = fixed(d)?; + let request = WireRequestRoot::decode(d)?; + let status = u16::try_from(d.read_u32()?) + .map_err(|_| CodecError::Invalid("native result status"))?; + let count = d.read_count()?; + if count > INPUT_ROOT_BYTES as usize / 8 { + return Err(CodecError::Limit); + } + let mut headers = Vec::with_capacity(count); + for _ in 0..count { + headers.push((d.read_text()?.into(), d.read_text()?.into())); + } + let response = GitHttpResponse { + status, + headers, + body: read_artifact(d)?, + }; + let plan = if d.read_bool()? { + Some(read_artifact(d)?) + } else { + None + }; + let count = d.read_count()?; + if count > 16 { + return Err(CodecError::Limit); + } + let mut options = Vec::with_capacity(count); + for _ in 0..count { + options.push(d.read_text()?.into()); + } + let signed = if d.read_bool()? { + Some(SignedPushAnnotation { + body: read_artifact(d)?, + signer: d.read_text()?.into(), + key: d.read_text()?.into(), + }) + } else { + None + }; + let value = Self { + operation, + request, + response, + plan, + options, + signed, + }; + value.validate()?; + Ok(value) + } +} diff --git a/crates/canopy-server/src/packs/publication/native_result/plan.rs b/crates/canopy-server/src/packs/publication/native_result/plan.rs new file mode 100644 index 0000000..177c05d --- /dev/null +++ b/crates/canopy-server/src/packs/publication/native_result/plan.rs @@ -0,0 +1,134 @@ +use super::*; +use crate::git_input::GitInput; +use axum::body::Body; +use bytes::Bytes; +use futures_core::Stream; +use std::{ + fs::File, + io::{self, Read}, + pin::Pin, + task::{Context, Poll}, +}; + +pub(super) const MAX_PLAN_BYTES: u64 = 128 << 20; +const FRAME_BYTES: u32 = 64 << 10; +const FRAME_UPDATES: usize = 32; +const DOMAIN: &[u8] = b"canopy.retained-push-plan.v1\0"; +#[cfg(test)] +mod tests; + +struct Frames { + plan: crate::PushPlan, + offset: usize, + started: bool, + failed: bool, +} +impl Stream for Frames { + type Item = Result; + fn poll_next(mut self: Pin<&mut Self>, _: &mut Context<'_>) -> Poll> { + if self.failed || self.started && self.offset == self.plan.updates.len() { + return Poll::Ready(None); + } + let encoded = (|| { + let mut e = BoundedEncoder::new(FRAME_BYTES)?; + if !self.started { + e.write_bytes(DOMAIN)?; + e.write_text(&self.plan.actor)?; + e.write_count(self.plan.updates.len())?; + self.started = true; + } else { + let end = (self.offset + FRAME_UPDATES).min(self.plan.updates.len()); + self.plan.encode_range(self.offset..end, &mut e)?; + self.offset = end; + } + let bytes = e.finish(); + let mut frame = Vec::with_capacity(bytes.len() + 4); + frame.extend_from_slice(&(bytes.len() as u32).to_be_bytes()); + frame.extend_from_slice(&bytes); + Ok(Bytes::from(frame)) + })(); + self.failed = encoded.is_err(); + Poll::Ready(Some(encoded)) + } +} +pub(super) async fn retain( + plan: crate::PushPlan, + store: &ArtifactStore, + operation: [u8; 16], + directory: &Path, + budget: &DiskBudget, +) -> Result { + let body = GitInput::receive( + Body::from_stream(Frames { + plan, + offset: 0, + started: false, + failed: false, + }), + directory, + budget, + Some(MAX_PLAN_BYTES), + None, + ) + .await?; + let digest = body.content_digest().await?; + let (_, artifact) = body.retain(store, operation, digest).await?; + Ok(artifact) +} +fn frame(file: &mut File) -> io::Result> { + let mut length = [0; 4]; + file.read_exact(&mut length)?; + let length = u32::from_be_bytes(length); + if length == 0 || length > FRAME_BYTES { + return Err(io::ErrorKind::InvalidData.into()); + } + let mut bytes = vec![0; length as usize]; + file.read_exact(&mut bytes)?; + Ok(bytes) +} +fn invalid(error: CodecError) -> io::Error { + io::Error::new(io::ErrorKind::InvalidData, error) +} +fn read(file: &mut File) -> io::Result { + let header = frame(file)?; + let mut d = BoundedDecoder::new(&header, FRAME_BYTES).map_err(invalid)?; + if d.read_bytes().map_err(invalid)? != DOMAIN { + return Err(io::ErrorKind::InvalidData.into()); + } + let actor = d.read_text().map_err(invalid)?.to_owned(); + let count = d.read_count().map_err(invalid)?; + d.finish().map_err(invalid)?; + if crate::directory::validate_component(&actor).is_err() + || !(1..=crate::refs::MAX_UPDATES).contains(&count) + { + return Err(io::ErrorKind::InvalidData.into()); + } + let mut updates = Vec::with_capacity(count); + while updates.len() < count { + let bytes = frame(file)?; + let mut d = BoundedDecoder::new(&bytes, FRAME_BYTES).map_err(invalid)?; + let chunk = crate::PushPlan::decode(&mut d).map_err(invalid)?; + d.finish().map_err(invalid)?; + if chunk.actor != actor || chunk.updates.len() != (count - updates.len()).min(FRAME_UPDATES) + { + return Err(io::ErrorKind::InvalidData.into()); + } + updates.extend(chunk.updates); + } + let mut extra = [0]; + if file.read(&mut extra)? != 0 { + return Err(io::ErrorKind::InvalidData.into()); + } + Ok(crate::PushPlan { actor, updates }) +} +pub(super) async fn reopen( + store: &ArtifactStore, + operation: [u8; 16], + artifact: ArtifactDescriptor, + directory: &Path, + budget: &DiskBudget, +) -> Result { + let reader = store.read(body_key(operation, artifact), artifact).await?; + let body = GitInput::reopen(reader, directory, budget, Some(MAX_PLAN_BYTES), None).await?; + Ok(body.read_owned(read).await?) +} diff --git a/crates/canopy-server/src/packs/publication/native_result/plan/tests.rs b/crates/canopy-server/src/packs/publication/native_result/plan/tests.rs new file mode 100644 index 0000000..d3a1021 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/native_result/plan/tests.rs @@ -0,0 +1,96 @@ +use super::*; +use std::io::Write; + +fn framed(bytes: Vec, output: &mut Vec) { + output.extend_from_slice(&(bytes.len() as u32).to_be_bytes()); + output.extend(bytes); +} +fn header(count: usize) -> Vec { + let mut e = BoundedEncoder::new(FRAME_BYTES).unwrap(); + e.write_bytes(DOMAIN).unwrap(); + e.write_text("owner").unwrap(); + e.write_count(count).unwrap(); + e.finish() +} +fn chunk(actor: &str, count: usize) -> Vec { + let plan = crate::PushPlan { + actor: actor.into(), + updates: (0..count) + .map(|n| crate::RefUpdate { + name: format!("refs/heads/{n}"), + expected: None, + new_oid: Some(crate::ObjectId::try_from(&[1; 32][..]).unwrap()), + }) + .collect(), + }; + let mut e = BoundedEncoder::new(FRAME_BYTES).unwrap(); + plan.encode(&mut e).unwrap(); + e.finish() +} +#[tokio::test] +async fn authenticated_plan_recovery_rejects_malformed_frames_and_releases_disk() +-> Result<(), Box> { + let store = ArtifactStore::new( + std::sync::Arc::new(object_store::memory::InMemory::new()), + [1; 16], + ); + let operation = *b"CANOPY01\0\0\0\0\0\0\0\x01"; + let directory = tempfile::TempDir::new()?; + let budget = DiskBudget::new(1 << 20); + let mut valid = Vec::new(); + framed(header(33), &mut valid); + framed(chunk("owner", 32), &mut valid); + framed(chunk("owner", 1), &mut valid); + let mut cases = vec![vec![0; 4], ((FRAME_BYTES + 1).to_be_bytes()).to_vec()]; + let mut trailing = valid.clone(); + trailing.push(1); + cases.push(trailing); + cases.push(valid[..valid.len() - 1].to_vec()); + for (count, actor, chunk_count) in [ + (0, "owner", 1), + (crate::refs::MAX_UPDATES + 1, "owner", 1), + (33, "owner", 1), + (1, "other", 1), + (1, "owner", 2), + ] { + let mut bytes = Vec::new(); + framed(header(count), &mut bytes); + framed(chunk(actor, chunk_count), &mut bytes); + cases.push(bytes); + } + for bytes in cases { + let digest = *blake3::hash(&bytes).as_bytes(); + let descriptor = retain_body(&store, operation, bytes).await?; + assert_eq!(descriptor.digest, digest); + assert!( + reopen(&store, operation, descriptor, directory.path(), &budget) + .await + .is_err() + ); + assert_eq!(budget.used(), 0); + } + let descriptor = retain_body(&store, operation, valid).await?; + let recovered = reopen(&store, operation, descriptor, directory.path(), &budget).await?; + assert_eq!(recovered.actor, "owner"); + assert_eq!(recovered.updates.len(), 33); + assert_eq!(budget.used(), 0); + // Disk rejection happens before a completion can escape and leaves no spool. + assert!( + reopen( + &store, + operation, + descriptor, + directory.path(), + &DiskBudget::new(1) + ) + .await + .is_err() + ); + assert_eq!(std::fs::read_dir(directory.path())?.count(), 0); + // Also exercise a short length prefix independently of artifact integrity. + let mut file = tempfile::tempfile()?; + file.write_all(&[0, 0])?; + std::io::Seek::rewind(&mut file)?; + assert!(read(&mut file).is_err()); + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/outcome.rs b/crates/canopy-server/src/packs/publication/outcome.rs new file mode 100644 index 0000000..2f8bbc3 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/outcome.rs @@ -0,0 +1,183 @@ +//! Ref-free native outcomes reuse the admitted session, bounded certificate +//! carrier and exact completion command; they cannot publish catalog or refs. +use super::*; +use super::{ + certificate::CertificateEnvelope, + commands::{authorized, fact}, + completion::{load_response, payload_binding}, + sql::*, +}; +use crate::packs::directory::index::codec::fixed as wire_fixed; +use cellule_runtime::{Committed, MutationIdentity, primitives::sql::SqlCell}; +use tokio::time::timeout_at; + +const DOMAIN: &[u8] = b"canopy.push-outcome.v1\0"; +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct OutcomeCertificate(pub(super) CertificateEnvelope); +#[derive(Clone, Debug, PartialEq, Eq)] +pub(super) struct OutcomeData { + tenant: [u8; 16], + application: [u8; 16], + pub(super) check: LeaseCheck, + pub(super) format: ObjectFormat, + pub(super) floor: GenerationFact, + digest: [u8; 32], +} +impl OutcomeData { + fn validate(&self) -> Result<(), CodecError> { + self.floor.validate()?; + if self.floor.catalog.is_some_and(|catalog| { + catalog.repository != self.check.token.repository || catalog.format != self.format + }) { + return Err(CodecError::Invalid("outcome retention context")); + } + Ok(()) + } +} +impl WireValue for OutcomeData { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.validate()?; + e.write_bytes(DOMAIN)?; + e.write_bytes(&self.tenant)?; + e.write_bytes(&self.application)?; + self.check.encode(e)?; + e.write_u8(self.format.bytes() as u8)?; + self.floor.encode(e)?; + e.write_bytes(&self.digest) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + if d.read_bytes()? != DOMAIN { + return Err(CodecError::Invalid("outcome certificate domain")); + } + let value = Self { + tenant: wire_fixed(d)?, + application: wire_fixed(d)?, + check: LeaseCheck::decode(d)?, + format: match d.read_u8()? { + 20 => ObjectFormat::Sha1, + 32 => ObjectFormat::Sha256, + _ => return Err(CodecError::Invalid("outcome format")), + }, + floor: GenerationFact::decode(d)?, + digest: wire_fixed(d)?, + }; + value.validate()?; + Ok(value) + } +} +impl WireValue for OutcomeCertificate { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.0.data::()?; + self.0.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self(CertificateEnvelope::decode(d)?); + value.0.data::()?; + Ok(value) + } +} +impl PreparationSession { + /// The session is authoritative but grants no catalog/ref proof. This path + /// takes no artifact loader, disk budget, native worker or scratch root. + pub async fn push_outcome( + &self, + request: PushCompletionRequest, + ) -> Result { + let (_, deadline) = self.live_lease()?; + timeout_at(deadline, async { + if request.plan.is_some() { return Err(CodecError::Invalid("outcome-only completion has a ref plan").into()); } + let signed = super::completion::signed_annotation(self, request.certificate)?; + crate::push::report::publication_matches(&request.response, None)?; + let response_id = uuid::Uuid::new_v4().into_bytes(); + let digest = payload_binding(None, &response_id, &request.response, &request.options, signed.as_ref())?; + let observed = self.client.query::(&self.target, None, self.check.clone()).await.map_err(|error| PreparationBaseError::Query(Box::new(error)))?; + let live = observed.output.ok_or(PreparationBaseError::Inactive)?; + if live.token != self.lease.token || live.base != self.lease.base || live.format != self.lease.format { return Err(PreparationBaseError::Context.into()); } + let sql = SqlCell::::new(self.client.clone(), self.target.clone()).map_err(CatalogAttestationError::from)?; + let seed = super::attestation::seed(&sql.query(Some(observed.receipt), statement("SELECT push_cert_seed FROM repository_identity WHERE singleton=1 AND repository_id=?1 AND object_format=?2", vec![blob(live.token.repository), SqlValue::Text(live.format.as_str().into())])).await.map_err(|error| CatalogAttestationError::Query(Box::new(error)))?.output).map_err(CatalogAttestationError::from)?; + let data = OutcomeData { tenant: *self.target.tenant().as_bytes(), application: *self.target.application().as_bytes(), check: self.check.clone(), format: live.format, floor: live.base, digest }; + let proof = OutcomeCertificate(CertificateEnvelope::seal(&data, &seed)?); + self.live_lease()?; + Ok(CatalogPushCompletion { proof: CompletionCatalogProof::OutcomeOnly(proof), response_id, response: request.response, options: request.options, signed }) + }).await.map_err(|_| PreparationBaseError::Inactive)? + } + /// Admitted final commands must settle or retain exact uncertainty. A local + /// lease timeout must never turn a possibly durable response into refusal. + pub async fn complete_outcome( + &self, + identity: MutationIdentity, + request: PushCompletionRequest, + ) -> Result, PushCompletionProofError> { + let input = self.push_outcome(request).await?; + self.live_lease()?; + self.client + .command::(&self.target, identity, input) + .await + .map_err(|error| PushCompletionProofError::Command(Box::new(error))) + } + pub async fn completed_push_response( + &self, + completed: &Committed, + ) -> Result { + let CatalogCompletionReply::Completed(output) = completed.output else { + return Err(CatalogPushResponseError::Invalid); + }; + let request = BeginRequest { + repository: self.check.token.repository, + operation: self.check.token.operation, + request_digest: self.check.token.request_digest, + actor: self.check.actor.clone(), + lease_ms: DEFAULT_LEASE_MS, + }; + load_response( + &self.client, + &self.target, + &request, + completed.receipt, + output, + ) + .await + } +} +pub(super) fn authenticate( + context: &CommandContext<'_, '_>, + certificate: &OutcomeCertificate, + digest: [u8; 32], +) -> cellule_runtime::Result> { + let data = certificate.0.data::()?; + if data.tenant != *context.target().tenant().as_bytes() + || data.application != *context.target().application().as_bytes() + || context.target() + != &crate::repository_target( + context.target().tenant(), + context.target().application(), + data.check.token.repository, + )? + { + return Ok(None); + } + let seed = super::attestation::seed(&context.sql(&statement("SELECT push_cert_seed FROM repository_identity WHERE singleton=1 AND repository_id=?1 AND object_format=?2", vec![blob(data.check.token.repository), SqlValue::Text(data.format.as_str().into())]))?)?; + if !certificate.0.authenticated(&seed) || data.digest != digest { + return Ok(None); + } + Ok(Some((data, seed))) +} +pub(super) fn current_authority( + context: &CommandContext<'_, '_>, + data: &OutcomeData, + generation: Option, +) -> cellule_runtime::Result { + Ok(authorized( + context, + data.check.token.repository, + &data.check.actor, + TokenScope::Write, + )? == Some(data.format) + && generation == Some(data.floor.generation) + && fact( + context, + data.check.token.repository, + data.format, + generation, + )? == data.floor) +} diff --git a/crates/canopy-server/src/packs/publication/prepare.rs b/crates/canopy-server/src/packs/publication/prepare.rs new file mode 100644 index 0000000..fc04f7b --- /dev/null +++ b/crates/canopy-server/src/packs/publication/prepare.rs @@ -0,0 +1,493 @@ +//! Construct catalog changes from complete physical witnesses and the queried +//! certified base. Callers cannot supply directory/source roots or closure bits. +use super::*; +use crate::packs::{ + catalog::CatalogSnapshot, + closure::{ClosureError, ClosureVerifier, RetainedClosure}, + directory::{ + DirectoryBuilder, DirectoryPartitioner, RUN_TARGET_BYTES, + index::{IndexError, NodeRef}, + snapshot::DirectorySnapshot, + }, + metadata::{MetadataError, MetadataLimits, MetadataSegment}, + sources::{NativePackDescriptor, SourceIndex, SourceRecord, SourceRoot}, + verification::{PhysicalError, PhysicalPackWitness}, +}; +use canopy_object_storage::artifact::ArtifactStore; +use cellule_ltx::DiskBudget; +use std::{path::Path, sync::Arc}; +use tokio::time::timeout_at; + +#[derive(Debug, thiserror::Error)] +pub enum CatalogPreparationError { + #[error("preparation encoding failed")] + Codec(#[from] CodecError), + #[error("preparation base is inactive or inconsistent")] + Base(#[from] PreparationBaseError), + #[error("preparation closure failed")] + Closure(#[from] ClosureError), + #[error("preparation input custody failed")] + Inputs(#[from] InputCheckpointError), + #[error("preparation metadata failed")] + Metadata(#[from] MetadataError), + #[error("preparation physical input failed")] + Physical(#[from] PhysicalError), + #[error("preparation catalog failed")] + Catalog(#[from] IndexError), + #[error("preparation worker failed")] + Task(#[from] tokio::task::JoinError), + #[error("catalog preparation is incomplete, failed or inconsistent")] + Integrity, +} + +/// Private construction is conditional verification, not the commit point. +/// The final publisher must check the actual fence, attempt and base generation, +/// current policy/ref expectations and the complete durability gate. +pub struct PreparedCatalog { + pub(super) base: Arc, + catalog: StoredCatalog, + object_count: u64, + edge_count: u64, + input_count: u64, + inputs_digest: [u8; 32], + inventory_digest: [u8; 32], + incoming_root: Option, + incoming_sources: Option, + closure: Arc, + pub(super) input_checkpoint_digest: Option<[u8; 32]>, +} +impl PreparedCatalog { + pub fn token(&self) -> PreparationToken { + self.base.context_token() + } + pub fn base(&self) -> GenerationFact { + self.base.generation_fact() + } + pub fn catalog(&self) -> StoredCatalog { + self.catalog + } + pub fn object_count(&self) -> u64 { + self.object_count + } + pub fn edge_count(&self) -> u64 { + self.edge_count + } + pub fn input_count(&self) -> u64 { + self.input_count + } + pub fn inputs_digest(&self) -> [u8; 32] { + self.inputs_digest + } + pub fn inventory_digest(&self) -> [u8; 32] { + self.inventory_digest + } + pub fn ensure_live(&self) -> Result<(), PreparationBaseError> { + self.base.live_lease().map(|_| ()) + } + /// Reuse exact physical inputs and the verified incoming DAG. Only incoming + /// overlaps/external anchors are read from the newly queried certified base; + /// no full pack decode or historical graph scan runs again. + pub async fn reconcile(&self) -> Result { + let (_, deadline) = self.base.live_lease()?; + timeout_at(deadline, async { + let base = Arc::new(self.base.select_current().await?); + let catalog = if base.generation_fact() == self.base() { + self.catalog + } else { + self.closure.reconcile(base.context(), &*base).await?; + let (mut directory, sources) = base.catalog_parts(); + let indexes = base.indexes(); + if let Some(root) = self.incoming_root { + directory.append(indexes.ranges(), root).await?; + } + let store = indexes.store(); + let sources = merge_sources( + &indexes.sources(), + sources, + self.incoming_sources, + base.context_token().artifact_operation, + ) + .await?; + CatalogSnapshot { + directory: directory + .upload(&store, base.context_token().artifact_operation) + .await?, + sources, + } + .upload(&store, base.context_token().artifact_operation) + .await? + }; + base.live_lease()?; + Ok(Self { + base, + catalog, + object_count: self.object_count, + edge_count: self.edge_count, + input_count: self.input_count, + inputs_digest: self.inputs_digest, + inventory_digest: self.inventory_digest, + incoming_root: self.incoming_root, + incoming_sources: self.incoming_sources, + closure: Arc::clone(&self.closure), + input_checkpoint_digest: self.input_checkpoint_digest, + }) + }) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } +} + +async fn merge_sources( + sources: &SourceIndex, + mut base: Option, + incoming: Option, + operation: [u8; 16], +) -> Result, IndexError> { + // Both roots are private assembler outputs. Initial preparation already + // validated every incoming source; avoid copying its tree into itself. + if base.is_none() { + return Ok(incoming); + } + if incoming.is_none() || base == incoming { + return Ok(base); + } + let mut cursor = sources.cursor(incoming, None)?; + while let Some(record) = cursor.next().await? { + base = Some(sources.insert(base, operation, record).await?); + } + Ok(base) +} + +/// One operation's disk-backed incoming inventory. Existing certified source +/// subtrees are reused; only exact verified incoming shards insert new leaves. +/// Failure/cancellation poisons the operation and never yields PreparedCatalog. +pub struct CatalogPreparation { + base: Arc, + store: Arc, + sources: Arc, + source_root: Option, + incoming_sources: Option, + snapshot: DirectorySnapshot, + directory: Option, + budget: DiskBudget, + output_limits: MetadataLimits, + closure: Option, + active: Option, + input_checkpoint_digest: Option<[u8; 32]>, + failed: bool, +} +impl CatalogPreparation { + pub async fn new( + root: &Path, + budget: DiskBudget, + base: Arc, + limits: MetadataLimits, + ) -> Result { + Self::new_with_run_limits( + root, + budget, + base, + limits, + MetadataLimits { + max_file_bytes: limits.max_file_bytes.min(RUN_TARGET_BYTES), + ..limits + }, + ) + .await + } + /// Separate the incoming verification spool class from bounded immutable + /// output runs. Larger spool admission never raises the point-lookup bound. + pub async fn new_with_run_limits( + root: &Path, + budget: DiskBudget, + base: Arc, + limits: MetadataLimits, + output_limits: MetadataLimits, + ) -> Result { + DirectoryPartitioner::validate_limits(output_limits)?; + let (lease, deadline) = base.live_lease()?; + let indexes = base.indexes(); + let (snapshot, source_root) = base.catalog_parts(); + let root = root.to_owned(); + let work = async { + let workspace = tokio::task::spawn_blocking(move || { + tempfile::Builder::new() + .prefix("canopy-catalog-preparation-") + .tempdir_in(root) + .map(Arc::new) + .map_err(MetadataError::from) + }) + .await??; + let context = base.context(); + let closure = ClosureVerifier::new_in_workspace( + Arc::clone(&workspace), + budget.clone(), + context, + limits, + ) + .await?; + let directory_budget = budget.clone(); + let directory = tokio::task::spawn_blocking(move || { + let mut builder = DirectoryBuilder::new( + workspace.path(), + directory_budget, + lease.token.repository, + lease.token.artifact_operation, + lease.format, + limits, + )?; + builder.retain_workspace(workspace); + Ok::<_, MetadataError>(builder) + }) + .await??; + base.live_lease()?; + Ok(Self { + base, + store: indexes.store(), + sources: indexes.sources(), + source_root, + incoming_sources: None, + snapshot, + directory: Some(directory), + budget, + output_limits, + closure: Some(closure), + active: None, + input_checkpoint_digest: None, + failed: false, + }) + }; + timeout_at(deadline, work) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } + fn start(&mut self) -> Result { + if self.failed { + return Err(CatalogPreparationError::Integrity); + } + self.failed = true; + Ok(self.base.live_lease()?.1) + } + pub fn begin_pack( + &mut self, + witness: PhysicalPackWitness, + ) -> Result<(), CatalogPreparationError> { + self.start()?; + if self.active.is_some() { + return Err(CatalogPreparationError::Integrity); + } + let native = witness.native(); + witness.verify_store(&self.store)?; + self.closure + .as_mut() + .ok_or(CatalogPreparationError::Integrity)? + .begin_pack(witness)?; + self.active = Some(native); + self.failed = false; + Ok(()) + } + /// Accept a complete physical witness from the exact authenticated input + /// checkpoint retained by this attempt. Raw descriptors cannot bypass the + /// public begin_pack namespace guard or construct the private custody proof. + pub async fn begin_retained_pack( + &mut self, + witness: PhysicalPackWitness, + ) -> Result<(), CatalogPreparationError> { + let deadline = self.start()?; + if self.active.is_some() { + return Err(CatalogPreparationError::Integrity); + } + let native = witness.native(); + witness.verify_store(&self.store)?; + let custody = timeout_at( + deadline, + inputs::RetainedNativeInput::open(&self.base, native), + ) + .await + .map_err(|_| PreparationBaseError::Inactive)??; + let digest = custody.digest(); + if self + .input_checkpoint_digest + .is_some_and(|old| old != digest) + { + return Err(CatalogPreparationError::Integrity); + } + self.closure + .as_mut() + .ok_or(CatalogPreparationError::Integrity)? + .begin_retained_pack(witness, custody)?; + self.input_checkpoint_digest = Some(digest); + self.active = Some(native); + self.failed = false; + Ok(()) + } + pub async fn add_segment( + &mut self, + segment: Arc, + ) -> Result<(), CatalogPreparationError> { + let deadline = self.start()?; + timeout_at(deadline, self.add_inner(segment)) + .await + .map_err(|_| PreparationBaseError::Inactive)??; + self.base.live_lease()?; + self.failed = false; + Ok(()) + } + async fn add_inner( + &mut self, + segment: Arc, + ) -> Result<(), CatalogPreparationError> { + let native = self.active.ok_or(CatalogPreparationError::Integrity)?; + self.closure + .as_mut() + .ok_or(CatalogPreparationError::Integrity)? + .add_segment(Arc::clone(&segment)) + .await?; + let directory = self + .directory + .take() + .ok_or(CatalogPreparationError::Integrity)?; + let pinned = Arc::clone(&segment); + self.directory = Some( + tokio::task::spawn_blocking(move || { + let mut directory = directory; + directory.add_segment(&pinned)?; + Ok::<_, MetadataError>(directory) + }) + .await??, + ); + let metadata = segment.upload(&self.store).await?; + let record = SourceRecord { + metadata, + pack: native.pack, + index: native.index, + pack_object_count: native.object_count, + }; + record.validate(native.repository, native.format)?; + if record.native() != native { + return Err(CatalogPreparationError::Integrity); + } + self.incoming_sources = Some( + self.sources + .insert( + self.incoming_sources, + self.base.context_token().artifact_operation, + record, + ) + .await?, + ); + Ok(()) + } + pub async fn finish_pack(&mut self) -> Result<(), CatalogPreparationError> { + let deadline = self.start()?; + if self.active.is_none() { + return Err(CatalogPreparationError::Integrity); + } + timeout_at( + deadline, + self.closure + .as_mut() + .ok_or(CatalogPreparationError::Integrity)? + .finish_pack(), + ) + .await + .map_err(|_| PreparationBaseError::Inactive)??; + self.base.live_lease()?; + self.active = None; + self.failed = false; + Ok(()) + } + pub async fn finish(mut self) -> Result { + let deadline = self.start()?; + if self.active.is_some() { + return Err(CatalogPreparationError::Integrity); + } + timeout_at(deadline, self.finish_inner()) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } + async fn finish_inner(mut self) -> Result { + let context = self.base.context(); + let (witness, closure) = self + .closure + .take() + .ok_or(CatalogPreparationError::Integrity)? + .finish_retained(context.base.map(|_| &*self.base)) + .await?; + let directory = self + .directory + .take() + .ok_or(CatalogPreparationError::Integrity)?; + let budget = self.budget; + let limits = self.output_limits; + let (partitioner, witness) = tokio::task::spawn_blocking(move || { + if witness.object_count() == 0 { + drop(directory); + Ok((None, witness)) + } else { + let run = directory.seal()?; + witness.verify_run(run.descriptor())?; + let partitioner = DirectoryPartitioner::new(Arc::new(run), budget, limits)?; + Ok::<_, CatalogPreparationError>((Some(partitioner), witness)) + } + }) + .await??; + let indexes = self.base.indexes(); + let index = indexes.ranges(); + let mut incoming_root = None; + if let Some(mut partitioner) = partitioner { + loop { + let (next, retained) = tokio::task::spawn_blocking(move || { + let next = partitioner.next_run()?; + Ok::<_, MetadataError>((next, partitioner)) + }) + .await??; + partitioner = retained; + let Some(run) = next else { + break; + }; + let stored = run.upload(&self.store).await?; + incoming_root = Some( + index + .insert(incoming_root, context.operation, stored) + .await?, + ); + } + // Exhaustion checked the exact canonical inventory before any root + // can escape this private preparation. Earlier uploads grant no authority. + } + match (incoming_root, witness.object_count()) { + (Some(root), count) if count > 0 && root.object_count == count => { + self.snapshot.append(index, root).await?; + } + (None, 0) => {} + _ => return Err(CatalogPreparationError::Integrity), + } + self.source_root = merge_sources( + &self.sources, + self.source_root, + self.incoming_sources, + context.operation, + ) + .await?; + let snapshot = CatalogSnapshot { + directory: self.snapshot.upload(&self.store, context.operation).await?, + sources: self.source_root, + }; + let catalog = snapshot.upload(&self.store, context.operation).await?; + self.base.live_lease()?; + Ok(PreparedCatalog { + base: self.base, + catalog, + object_count: witness.object_count(), + edge_count: witness.edge_count(), + input_count: witness.input_count(), + inputs_digest: witness.inputs_digest(), + inventory_digest: witness.inventory_digest(), + incoming_root, + incoming_sources: self.incoming_sources, + closure: Arc::new(closure), + input_checkpoint_digest: self.input_checkpoint_digest, + }) + } +} diff --git a/crates/canopy-server/src/packs/publication/publish.rs b/crates/canopy-server/src/packs/publication/publish.rs new file mode 100644 index 0000000..4da6004 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/publish.rs @@ -0,0 +1,383 @@ +//! Atomic typed catalog/ref publication. The HTTP completion command must call +//! the same core inside its response transaction; this command alone does not +//! integrate network reports, push options/certificates or reviewed merges. +use super::*; +use super::{ + commands::{authorized, check_pin, fact, load, matched}, + sql::*, +}; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct PublishedRefs { + pub generation: u64, + pub ref_generation: u64, + pub certificate_digest: [u8; 32], +} +impl WireValue for PublishedRefs { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + if self.generation == 0 + || self.generation > i64::MAX as u64 + || self.ref_generation == 0 + || self.ref_generation > i64::MAX as u64 + { + return Err(CodecError::Invalid("invalid publication receipt")); + } + e.write_u64(self.generation)?; + e.write_u64(self.ref_generation)?; + e.write_bytes(&self.certificate_digest) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + generation: d.read_u64()?, + ref_generation: d.read_u64()?, + certificate_digest: crate::packs::directory::index::codec::fixed(d)?, + }; + let mut e = BoundedEncoder::new(128)?; + value.encode(&mut e)?; + Ok(value) + } +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum PublicationReply { + Published(PublishedRefs), + Denied(PreparationDenial), +} +impl WireValue for PublicationReply { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + match self { + Self::Published(value) => { + e.write_u8(0)?; + value.encode(e) + } + Self::Denied(reason) => PreparationReply::Denied(*reason).encode(e), + } + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + match d.read_u8()? { + 0 => Ok(Self::Published(PublishedRefs::decode(d)?)), + 1 => Ok(Self::Denied(PreparationDenial::Unauthorized)), + 2 => Ok(Self::Denied(PreparationDenial::Conflict)), + 3 => Ok(Self::Denied(PreparationDenial::Stale)), + 4 => Ok(Self::Denied(PreparationDenial::Expired)), + 5 => Ok(Self::Denied(PreparationDenial::Capacity)), + 6 => Ok(Self::Denied(PreparationDenial::Missing)), + _ => Err(CodecError::Invalid("invalid publication reply")), + } + } +} +pub struct PublishCatalogRefs; +impl Command for PublishCatalogRefs { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 18; + const CODEC_VERSION: u32 = 1; + type Input = RefPublicationProof; + type Output = PublicationReply; + fn execute( + context: &mut CommandContext<'_, '_>, + input: RefPublicationProof, + ) -> cellule_runtime::Result> { + publish(context, &input) + } +} +fn denied(reason: PreparationDenial) -> CommandResult { + CommandResult::Rejected(PublicationReply::Denied(reason)) +} +/// Every negative decision precedes writes. Errors after writes abort the whole +/// Cell transaction. Network completion must persist its exact response here, +/// in the same command, rather than call this typed command then stage a reply. +pub(super) fn publish( + context: &mut CommandContext<'_, '_>, + input: &RefPublicationProof, +) -> cellule_runtime::Result> { + let Some((data, key)) = authenticate( + context, + &input.certificate, + Some(super::ref_proof::binding(&input.plan, &input.ancestry)?), + None, + )? + else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + publish_authenticated(context, input, data, key) +} +/// Private command-local boundary, shared by typed and network completion. +pub(super) fn authenticate( + context: &CommandContext<'_, '_>, + certificate: &CatalogCertificate, + refs_digest: Option<[u8; 32]>, + completion_digest: Option<[u8; 32]>, +) -> cellule_runtime::Result> { + let data = certificate.data()?; + if data.tenant != *context.target().tenant().as_bytes() + || data.application != *context.target().application().as_bytes() + || context.target() + != &crate::repository_target( + context.target().tenant(), + context.target().application(), + data.token.repository, + )? + { + return Ok(None); + } + let secret = context.sql(&statement("SELECT push_cert_seed FROM repository_identity WHERE singleton=1 AND repository_id=?1 AND object_format=?2", vec![blob(data.token.repository), SqlValue::Text(data.catalog.format.as_str().into())]))?; + let key = super::attestation::seed(&secret)?; + if !certificate.authenticated(&key) + || data.refs_digest != refs_digest + || data.completion_digest != completion_digest + { + return Ok(None); + } + Ok(Some((data, key))) +} +pub(super) fn retention_matches( + context: &CommandContext<'_, '_>, + data: &super::certificate::CertificateData, + row_generation: Option, + format: ObjectFormat, +) -> cellule_runtime::Result { + let Some(row_generation) = row_generation else { + return Ok(false); + }; + Ok(inputs::retention_matches(context, data)? + && row_generation == data.retention_floor + && fact(context, data.token.repository, format, Some(row_generation))?.certificate + == data.retention_certificate) +} +pub(super) fn publish_authenticated( + context: &mut CommandContext<'_, '_>, + input: &RefPublicationProof, + data: super::certificate::CertificateData, + key: [u8; 32], +) -> cellule_runtime::Result> { + if data.compaction { + return Ok(denied(PreparationDenial::Unauthorized)); + } + if data.actor != input.plan.actor { + return Ok(denied(PreparationDenial::Unauthorized)); + } + let saved = context.sql(&statement( + "SELECT actor,request_digest,publication,response_id,publication_plan_digest FROM pushes WHERE id=?1", + vec![blob(data.token.operation)], + ))?; + if let Some(row) = rows(&saved)?.first() { + let [ + SqlValue::Text(actor), + digest, + outcome, + response, + original_plan, + ] = row.as_slice() + else { + return Err(Error::Command("invalid publication push identity")); + }; + if *actor != data.actor || fixed::<32>(digest)? != data.token.request_digest { + return Ok(denied(PreparationDenial::Conflict)); + } + match outcome { + SqlValue::Blob(bytes) => { + if fixed::<32>(original_plan)? != super::ref_proof::plan_digest(&input.plan)? { + return Ok(denied(PreparationDenial::Conflict)); + } + let mut d = BoundedDecoder::new(bytes, 128)?; + let result = PublishedRefs::decode(&mut d)?; + d.finish()?; + return Ok(CommandResult::Success(PublicationReply::Published(result))); + } + SqlValue::Null if *response == SqlValue::Null => {} + SqlValue::Null => return Ok(denied(PreparationDenial::Conflict)), + _ => return Err(Error::Command("invalid saved publication outcome")), + } + } + // Logical outcome replay above cannot grant a new write. All new writes + // require the actual admitted fence and current authority/lease/policy. + if data.token.owner != context.owner_fence() { + return Ok(denied(PreparationDenial::Stale)); + } + let Some(format) = authorized( + context, + data.token.repository, + &data.actor, + TokenScope::Write, + )? + else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + if format != data.catalog.format { + return Ok(denied(PreparationDenial::Conflict)); + } + let Some(row) = load(context, data.token)? else { + return Ok(denied(PreparationDenial::Missing)); + }; + if !matched( + &row, + &LeaseCheck { + token: data.token, + actor: data.actor.clone(), + }, + ) { + return Ok(denied(PreparationDenial::Stale)); + } + if row.expires <= now(context.now_ms())? { + return Ok(denied(PreparationDenial::Expired)); + } + check_pin(context, &row)?; + if !retention_matches(context, &data, row.generation, format)? + || fact(context, data.token.repository, format, None)? != data.base + { + return Ok(denied(PreparationDenial::Conflict)); + } + // A selected immutable ref snapshot cannot be mutated by the inline SQL + // publisher. The root publisher and all readers must cut over together. + if data.base.refs.is_some() { + return Ok(denied(PreparationDenial::Conflict)); + } + let Some(validated) = crate::refs::validate_refs(context, &input.plan)? else { + return Ok(denied(PreparationDenial::Conflict)); + }; + for (page, updates) in input.plan.updates.chunks(128).enumerate() { + let policies = context.sql(&SqlBatch { + statements: updates + .iter() + .enumerate() + .map(|(at, update)| { + crate::branch_rules::policy_statement_with_ancestry( + update, + super::ref_proof::proven(&input.ancestry, page * 128 + at), + ) + }) + .collect(), + })?; + if policies.len() != updates.len() { + return Err(Error::Command("missing final publication policies")); + } + for (update, policy) in updates.iter().zip(policies) { + if crate::branch_rules::decode_policy(&[policy])? + .is_some_and(|rule| !rule.allows(update, true)) + { + return Ok(denied(PreparationDenial::Conflict)); + } + } + } + let counts = context.sql(&statement( + "SELECT count(*) FROM (SELECT generation FROM catalog_generations LIMIT ?1)", + vec![number(MAX_RETAINED_GENERATIONS)?], + ))?; + let Some([count]) = rows(&counts)?.first().map(Vec::as_slice) else { + return Err(Error::Command("missing publication generation count")); + }; + if unsigned(count)? >= MAX_RETAINED_GENERATIONS || data.base.generation >= i64::MAX as u64 { + return Ok(denied(PreparationDenial::Capacity)); + } + let certificate = input.certificate.bytes()?; + let certificate_digest = *blake3::hash(&certificate).as_bytes(); + let Some(checkpoint_missing) = checkpoint(context, &data, &key)? else { + return Ok(denied(PreparationDenial::Conflict)); + }; + if row.expires <= now(context.now_ms())? { + return Ok(denied(PreparationDenial::Expired)); + } + // No rejection path below this point. All remaining failures must roll back. + if checkpoint_missing { + changed(context.sql(&statement("UPDATE catalog_operations SET attestation=?1,attestation_digest=?2 WHERE id=?3 AND attestation IS NULL", vec![blob(&certificate), blob(certificate_digest), blob(data.token.operation)]))?)?; + changed(context.sql(&statement("UPDATE catalog_leases SET attestation=?1,attestation_digest=?2 WHERE incarnation=?3 AND admission_sequence=?4 AND attestation IS NULL", vec![blob(&certificate), blob(certificate_digest), blob(data.token.owner.incarnation.as_bytes()), number(data.token.attempt)?]))?)?; + } + let generation = data.base.generation + 1; + let mut encoded_catalog = BoundedEncoder::new(256)?; + data.catalog.encode(&mut encoded_catalog)?; + changed(context.sql(&statement( + "INSERT INTO catalog_generations(generation,catalog,certificate) VALUES(?1,?2,?3)", + vec![ + number(generation)?, + blob(encoded_catalog.finish()), + blob(certificate_digest), + ], + ))?)?; + changed(context.sql(&statement( + "UPDATE catalog_state SET generation=?1 WHERE singleton=1 AND generation=?2", + vec![number(generation)?, number(data.base.generation)?], + ))?)?; + validated.apply(context)?; + let refs = context.sql(&statement( + "SELECT generation FROM ref_generation WHERE singleton=1", + vec![], + ))?; + let Some([ref_generation]) = rows(&refs)?.first().map(Vec::as_slice) else { + return Err(Error::Command("missing published ref generation")); + }; + let result = PublishedRefs { + generation, + ref_generation: unsigned(ref_generation)?, + certificate_digest, + }; + let mut encoded_result = BoundedEncoder::new(128)?; + result.encode(&mut encoded_result)?; + if rows(&saved)?.is_empty() { + changed(context.sql(&statement( + "INSERT INTO pushes(id,actor,request_digest,publication,publication_plan_digest) VALUES(?1,?2,?3,?4,?5)", + vec![ + blob(data.token.operation), + SqlValue::Text(data.actor), + blob(data.token.request_digest), + blob(encoded_result.finish()), + blob(super::ref_proof::plan_digest(&input.plan)?), + ], + ))?)?; + } else { + changed(context.sql(&statement( + "UPDATE pushes SET publication=?1,publication_plan_digest=?3 WHERE id=?2 AND publication IS NULL", + vec![blob(encoded_result.finish()), blob(data.token.operation), blob(super::ref_proof::plan_digest(&input.plan)?)], + ))?)?; + } + changed(context.sql(&statement( + "DELETE FROM catalog_operations WHERE id=?1", + vec![blob(data.token.operation)], + ))?)?; + Ok(CommandResult::Success(PublicationReply::Published(result))) +} +pub(super) fn checkpoint( + context: &CommandContext<'_, '_>, + data: &super::certificate::CertificateData, + key: &[u8; 32], +) -> cellule_runtime::Result> { + let previous = context.sql(&statement( + "SELECT attestation,attestation_digest FROM catalog_operations WHERE id=?1", + vec![blob(data.token.operation)], + ))?; + let pinned = context.sql(&statement("SELECT attestation,attestation_digest FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2", vec![blob(data.token.owner.incarnation.as_bytes()), number(data.token.attempt)?]))?; + if rows(&previous)? != rows(&pinned)? { + return Err(Error::Command("publication checkpoint differs from pin")); + } + let missing = match rows(&previous)?.first().map(Vec::as_slice) { + Some([SqlValue::Null, SqlValue::Null]) => true, + Some([SqlValue::Blob(bytes), digest]) => { + if fixed::<32>(digest)? != *blake3::hash(bytes).as_bytes() { + return Err(Error::Command("publication checkpoint digest differs")); + } + let mut d = BoundedDecoder::new(bytes, CERTIFICATE_BYTES)?; + let old = CatalogCertificate::decode(&mut d)?; + d.finish()?; + let mut old_data = old.data()?; + if old_data.base.generation > data.base.generation { + return Ok(None); + } + // Reconciliation changes only selected roots. Exact verified input + // facts, attempt and retention floor must still match the checkpoint. + old_data.base = data.base; + old_data.catalog = data.catalog; + old_data.refs_digest = data.refs_digest; + old_data.completion_digest = data.completion_digest; + if !old.authenticated(key) || old_data != *data { + return Ok(None); + } + false + } + _ => return Err(Error::Command("invalid publication checkpoint")), + }; + Ok(Some(missing)) +} +pub(super) fn changed(sets: Vec) -> cellule_runtime::Result<()> { + if sets.first().is_none_or(|set| set.rows_affected != 1) { + return Err(Error::Command("publication changed unexpected rows")); + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/ref_policy/codec.rs b/crates/canopy-server/src/packs/publication/ref_policy/codec.rs new file mode 100644 index 0000000..1ca99f4 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/ref_policy/codec.rs @@ -0,0 +1,204 @@ +use super::*; + +impl RefPolicyIntent { + pub(super) fn shape(self) -> Result<(), CodecError> { + crate::validate_repository_id(self.id) + .map_err(|_| CodecError::Invalid("invalid ref policy guard ID"))?; + if self.epoch > i64::MAX as u64 + || !(1..=crate::refs::MAX_UPDATES as u64).contains(&self.updates) + { + return Err(CodecError::Invalid("invalid ref policy intent")); + } + Ok(()) + } +} +impl RefPolicyPage { + pub(super) fn shape(&self) -> Result<(), CodecError> { + self.intent.shape()?; + let n = self.proof.plan.updates.len() as u64; + if n == 0 + || n > REF_POLICY_PAGE_UPDATES as u64 + || self + .offset + .checked_add(n) + .is_none_or(|end| end > self.intent.updates) + { + return Err(CodecError::Invalid("invalid ref policy page")); + } + super::super::ref_proof::binding(&self.proof.plan, &self.proof.ancestry)?; + Ok(()) + } +} +impl WireValue for RefPolicyIntent { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.shape()?; + e.write_bytes(&self.id)?; + e.write_u64(self.epoch)?; + e.write_u64(self.updates)?; + e.write_bytes(&self.plan_digest)?; + e.write_bytes(&self.evidence_digest) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + id: crate::packs::directory::index::codec::fixed(d)?, + epoch: d.read_u64()?, + updates: d.read_u64()?, + plan_digest: crate::packs::directory::index::codec::fixed(d)?, + evidence_digest: crate::packs::directory::index::codec::fixed(d)?, + }; + value.shape()?; + Ok(value) + } +} +impl WireValue for RefPolicyPage { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.shape()?; + self.intent.encode(e)?; + e.write_u64(self.offset)?; + self.proof.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + intent: RefPolicyIntent::decode(d)?, + offset: d.read_u64()?, + proof: RefPublicationProof::decode(d)?, + }; + value.shape()?; + Ok(value) + } +} +impl WireValue for RefPolicyProgress { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + if self.next > self.total || !(1..=crate::refs::MAX_UPDATES as u64).contains(&self.total) { + return Err(CodecError::Invalid("invalid ref policy progress")); + } + e.write_u64(self.next)?; + e.write_u64(self.total)?; + e.write_bool(self.valid) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + next: d.read_u64()?, + total: d.read_u64()?, + valid: d.read_bool()?, + }; + value.encode(&mut BoundedEncoder::new(32)?)?; + Ok(value) + } +} +fn denial(tag: u8) -> Result { + match tag { + 1 => Ok(PreparationDenial::Unauthorized), + 2 => Ok(PreparationDenial::Conflict), + 3 => Ok(PreparationDenial::Stale), + 4 => Ok(PreparationDenial::Expired), + 5 => Ok(PreparationDenial::Capacity), + 6 => Ok(PreparationDenial::Missing), + _ => Err(CodecError::Invalid("invalid ref policy denial")), + } +} +impl WireValue for RefPolicyReply { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + match self { + Self::Registered(value) => { + e.write_u8(0)?; + value.encode(e) + } + Self::Denied(reason) => PreparationReply::Denied(*reason).encode(e), + } + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let tag = d.read_u8()?; + if tag == 0 { + Ok(Self::Registered(RefPolicyProgress::decode(d)?)) + } else { + Ok(Self::Denied(denial(tag)?)) + } + } +} +impl WireValue for RefPolicyLookup { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.check.encode(e)?; + self.intent.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + Ok(Self { + check: LeaseCheck::decode(d)?, + intent: RefPolicyIntent::decode(d)?, + }) + } +} +impl WireValue for RefPolicyReap { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.maintenance.encode(e)?; + crate::validate_repository_id(self.id) + .map_err(|_| CodecError::Invalid("invalid ref policy guard ID"))?; + e.write_bytes(&self.id) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + maintenance: MaintenanceRequest::decode(d)?, + id: crate::packs::directory::index::codec::fixed(d)?, + }; + value.encode(&mut BoundedEncoder::new(512)?)?; + Ok(value) + } +} +impl WireValue for RefPolicyReapReply { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + match self { + Self::Reaped { watches, removed } => { + if *watches > WATCH_REAP_ROWS { + return Err(CodecError::Invalid("invalid watch reap count")); + } + e.write_u8(0)?; + e.write_u64(*watches)?; + e.write_bool(*removed) + } + Self::Denied(reason) => PreparationReply::Denied(*reason).encode(e), + } + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let tag = d.read_u8()?; + if tag == 0 { + let value = Self::Reaped { + watches: d.read_u64()?, + removed: d.read_bool()?, + }; + value.encode(&mut BoundedEncoder::new(32)?)?; + Ok(value) + } else { + Ok(Self::Denied(denial(tag)?)) + } + } +} +impl WireValue for RefRootPublicationProof { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.shape()?; + self.certificate.encode(e)?; + self.guard.encode(e)?; + self.snapshot.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self { + certificate: CatalogCertificate::decode(d)?, + guard: RefPolicyIntent::decode(d)?, + snapshot: RefStateSnapshotRoot::decode(d)?, + }; + value.shape()?; + Ok(value) + } +} +impl RefRootPublicationProof { + pub(in crate::packs::publication) fn shape(&self) -> Result<(), CodecError> { + let data = self.certificate.data()?; + if data.compaction + || data.base.refs.is_none() + || data.token.artifact_operation != self.snapshot.operation() + || data.refs_digest != Some(root_binding(self.guard, self.snapshot)?) + { + return Err(CodecError::Invalid("invalid guarded root proof")); + } + Ok(()) + } +} diff --git a/crates/canopy-server/src/packs/publication/ref_policy/commands.rs b/crates/canopy-server/src/packs/publication/ref_policy/commands.rs new file mode 100644 index 0000000..1ac5047 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/ref_policy/commands.rs @@ -0,0 +1,413 @@ +use super::super::{ + commands::{authorized, check_pin, fact, load, matched, pin, pin_query}, + publish::{authenticate, changed, retention_matches}, + sql::*, +}; +use super::*; + +const GUARD: &str = + "SELECT scope,token,policy_epoch,total,next,valid FROM ref_policy_guards WHERE id=?1"; +pub(super) const EPOCH: &str = "SELECT version FROM ref_policy_epoch WHERE singleton=1"; +pub(super) fn epoch(sets: &[SqlResultSet]) -> cellule_runtime::Result { + let Some([value]) = rows(sets)?.first().map(Vec::as_slice) else { + return Err(Error::Command("missing ref policy epoch")); + }; + unsigned(value) +} +struct Guard { + scope: [u8; 32], + token: PreparationToken, + epoch: u64, + progress: RefPolicyProgress, +} +fn guard(sets: &[SqlResultSet]) -> cellule_runtime::Result> { + let Some(row) = rows(sets)?.first() else { + return Ok(None); + }; + let [ + hash, + SqlValue::Blob(bytes), + version, + total, + next, + SqlValue::Integer(valid), + ] = row.as_slice() + else { + return Err(Error::Command("invalid ref policy guard")); + }; + let mut d = BoundedDecoder::new(bytes, TOKEN_BYTES)?; + let token = PreparationToken::decode(&mut d)?; + d.finish()?; + if ![0, 1].contains(valid) { + return Err(Error::Command("invalid ref policy validity")); + } + let progress = RefPolicyProgress { + next: unsigned(next)?, + total: unsigned(total)?, + valid: *valid == 1, + }; + progress.encode(&mut BoundedEncoder::new(32)?)?; + Ok(Some(Guard { + scope: fixed(hash)?, + token, + epoch: unsigned(version)?, + progress, + })) +} +fn rejected(reason: PreparationDenial) -> CommandResult { + CommandResult::Rejected(RefPolicyReply::Denied(reason)) +} + +pub struct RegisterRefPolicyPage; +impl Command for RegisterRefPolicyPage { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 33; + const CODEC_VERSION: u32 = 1; + type Input = RefPolicyPage; + type Output = RefPolicyReply; + fn execute( + context: &mut CommandContext<'_, '_>, + input: Self::Input, + ) -> cellule_runtime::Result> { + input.shape()?; + let Some((data, _)) = authenticate( + context, + &input.proof.certificate, + Some(page_binding(&input)?), + None, + )? + else { + return Ok(rejected(PreparationDenial::Unauthorized)); + }; + if data.actor != input.proof.plan.actor || data.base.refs.is_none() { + return Ok(rejected(PreparationDenial::Conflict)); + } + if super::super::ref_proof::shape(&input.proof.plan, data.catalog.format).is_err() { + return Ok(rejected(PreparationDenial::Conflict)); + } + if data.token.owner != context.owner_fence() { + return Ok(rejected(PreparationDenial::Stale)); + } + let Some(format) = authorized( + context, + data.token.repository, + &data.actor, + TokenScope::Write, + )? + else { + return Ok(rejected(PreparationDenial::Unauthorized)); + }; + if format != data.catalog.format { + return Ok(rejected(PreparationDenial::Conflict)); + } + let Some(row) = load(context, data.token)? else { + return Ok(rejected(PreparationDenial::Missing)); + }; + if !matched( + &row, + &LeaseCheck { + token: data.token, + actor: data.actor.clone(), + }, + ) { + return Ok(rejected(PreparationDenial::Stale)); + } + if row.expires <= now(context.now_ms())? { + return Ok(rejected(PreparationDenial::Expired)); + } + check_pin(context, &row)?; + // Only a retained immutable base is required here. Unrelated root + // advancement must not invalidate generation-independent predicates. + if !retention_matches(context, &data, row.generation, format)? + || fact( + context, + data.token.repository, + format, + Some(data.base.generation), + )? != data.base + || epoch(&context.sql(&statement(EPOCH, vec![]))?)? != input.intent.epoch + { + return Ok(rejected(PreparationDenial::Conflict)); + } + let scoped = scope(data.token, &data.actor, format, input.intent)?; + let old = guard(&context.sql(&statement(GUARD, vec![blob(input.intent.id)]))?)?; + let end = input.offset + input.proof.plan.updates.len() as u64; + if let Some(old) = &old { + if old.scope != scoped + || old.token != data.token + || old.epoch != input.intent.epoch + || old.progress.total != input.intent.updates + || !old.progress.valid + { + return Ok(rejected(PreparationDenial::Conflict)); + } + if end <= old.progress.next { + return Ok(CommandResult::Success(RefPolicyReply::Registered( + old.progress, + ))); + } + if input.offset != old.progress.next { + return Ok(rejected(PreparationDenial::Conflict)); + } + } else { + if input.offset != 0 { + return Ok(rejected(PreparationDenial::Missing)); + } + let count = context.sql(&statement( + "SELECT count(*) FROM (SELECT id FROM ref_policy_guards LIMIT ?1)", + vec![number(MAX_REF_POLICY_GUARDS)?], + ))?; + let Some([count]) = rows(&count)?.first().map(Vec::as_slice) else { + return Err(Error::Command("missing guard capacity")); + }; + if unsigned(count)? >= MAX_REF_POLICY_GUARDS { + return Ok(rejected(PreparationDenial::Capacity)); + } + } + let policies = context.sql(&SqlBatch { + statements: input + .proof + .plan + .updates + .iter() + .enumerate() + .map(|(i, update)| { + crate::branch_rules::policy_statement_with_ancestry( + update, + super::super::ref_proof::proven(&input.proof.ancestry, i), + ) + }) + .collect(), + })?; + if policies.len() != input.proof.plan.updates.len() { + return Err(Error::Command("missing guarded ref policies")); + } + for (update, policy) in input.proof.plan.updates.iter().zip(policies) { + if crate::branch_rules::decode_policy(&[policy])? + .is_some_and(|rule| !rule.allows(update, true)) + { + return Ok(rejected(PreparationDenial::Conflict)); + } + } + let mut dependencies = std::collections::BTreeSet::new(); + for update in &input.proof.plan.updates { + let Some(oid) = update.new_oid else { + continue; + }; + let sets=context.sql(&statement("SELECT q.context,c.version,r.number FROM branch_required_checks q JOIN branch_rules b ON b.reference=q.reference AND b.enabled=1 JOIN check_contexts c ON c.name=q.context JOIN check_runs r ON r.number=(SELECT number FROM check_runs WHERE oid=?2 AND context=q.context AND context_version=c.version ORDER BY number DESC LIMIT 1) WHERE q.reference=?1 LIMIT 17",vec![SqlValue::Text(update.name.clone()),blob(oid)]))?; + if rows(&sets)?.len() > 16 { + return Ok(rejected(PreparationDenial::Capacity)); + } + for row in rows(&sets)? { + let [SqlValue::Text(name), version, run] = row.as_slice() else { + return Err(Error::Command("invalid guarded check dependency")); + }; + validate_component(name)?; + let version = unsigned(version)?; + let run = unsigned(run)?; + if version == 0 || run == 0 { + return Err(Error::Command("invalid guarded check version")); + } + dependencies.insert((oid, name.clone(), version, run)); + } + } + let mut missing = Vec::new(); + for (oid, name, version, run) in dependencies { + if rows(&context.sql(&statement("SELECT 1 FROM ref_policy_watches WHERE guard=?1 AND oid=?2 AND context=?3 AND context_version=?4 AND run_number=?5",vec![blob(input.intent.id),blob(oid),SqlValue::Text(name.clone()),number(version)?,number(run)?]))?)?.is_empty() { + missing.push((oid,name,version,run)); + } + } + let budget = context.sql(&statement( + "SELECT watches FROM ref_policy_budget WHERE singleton=1", + vec![], + ))?; + let Some([budget]) = rows(&budget)?.first().map(Vec::as_slice) else { + return Err(Error::Command("missing ref policy watch budget")); + }; + let added = missing.len() as u64; + if unsigned(budget)? + .checked_add(added) + .is_none_or(|n| n > MAX_REF_POLICY_WATCHES) + { + return Ok(rejected(PreparationDenial::Capacity)); + } + let mut token = BoundedEncoder::new(TOKEN_BYTES)?; + data.token.encode(&mut token)?; + if row.expires <= now(context.now_ms())? { + return Ok(rejected(PreparationDenial::Expired)); + } + // No rejection after the first write. Policy validation and installing + // every watch are one Cell transaction, without an observation gap. + if old.is_none() { + changed(context.sql(&statement("INSERT INTO ref_policy_guards(id,scope,token,policy_epoch,total,next,valid) VALUES(?1,?2,?3,?4,?5,0,1)",vec![blob(input.intent.id),blob(scoped),blob(token.finish()),number(input.intent.epoch)?,number(input.intent.updates)?]))?)?; + } + for (oid, name, version, run) in missing { + changed(context.sql(&statement("INSERT INTO ref_policy_watches(guard,oid,context,context_version,run_number) VALUES(?1,?2,?3,?4,?5)",vec![blob(input.intent.id),blob(oid),SqlValue::Text(name),number(version)?,number(run)?]))?)?; + } + changed(context.sql(&statement( + "UPDATE ref_policy_budget SET watches=watches+?1 WHERE singleton=1 AND watches<=?2", + vec![number(added)?, number(MAX_REF_POLICY_WATCHES - added)?], + ))?)?; + changed(context.sql(&statement( + "UPDATE ref_policy_guards SET next=?1 WHERE id=?2 AND next=?3 AND valid=1", + vec![number(end)?, blob(input.intent.id), number(input.offset)?], + ))?)?; + Ok(CommandResult::Success(RefPolicyReply::Registered( + RefPolicyProgress { + next: end, + total: input.intent.updates, + valid: true, + }, + ))) + } +} +pub struct CheckRefPolicyGuard; +impl Query for CheckRefPolicyGuard { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 34; + const CODEC_VERSION: u32 = 1; + type Input = RefPolicyLookup; + type Output = Option; + fn execute( + context: &mut QueryContext<'_>, + input: Self::Input, + ) -> cellule_runtime::Result { + validate_component(&input.check.actor)?; + if !decode_access(&context.sql(&SqlBatch { + statements: vec![access_statement(&input.check.actor)], + })?)? + .is_some_and(|role| role >= TokenScope::Write) + { + return Ok(None); + } + let Some(format) = identity( + &context.sql(&statement(IDENTITY, vec![]))?, + input.check.token.repository, + )? + else { + return Ok(None); + }; + let Some(op) = operation( + &context.sql(&statement( + OPERATION, + vec![blob(input.check.token.operation)], + ))?, + input.check.token.repository, + input.check.token.operation, + )? + else { + return Ok(None); + }; + if !matched(&op, &input.check) || op.expires <= now(context.now_ms())? { + return Ok(None); + } + pin(&context.sql(&pin_query(op.token)?)?, &op)?; + let Some(row) = guard(&context.sql(&statement(GUARD, vec![blob(input.intent.id)]))?)? + else { + return Ok(None); + }; + if row.scope != scope(input.check.token, &input.check.actor, format, input.intent)? + || row.token != input.check.token + || row.epoch != input.intent.epoch + || row.progress.total != input.intent.updates + { + return Ok(None); + } + Ok(Some(RefPolicyProgress { + valid: row.progress.valid + && epoch(&context.sql(&statement(EPOCH, vec![]))?)? == input.intent.epoch, + ..row.progress + })) + } +} +pub struct ReapRefPolicyGuard; +impl Command for ReapRefPolicyGuard { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 35; + const CODEC_VERSION: u32 = 1; + type Input = RefPolicyReap; + type Output = RefPolicyReapReply; + fn execute( + context: &mut CommandContext<'_, '_>, + input: Self::Input, + ) -> cellule_runtime::Result> { + let request = input.maintenance; + if request.owner != context.owner_fence() { + return Ok(CommandResult::Rejected(RefPolicyReapReply::Denied( + PreparationDenial::Stale, + ))); + } + if authorized( + context, + request.repository, + &request.actor, + TokenScope::Admin, + )? + .is_none() + { + return Ok(CommandResult::Rejected(RefPolicyReapReply::Denied( + PreparationDenial::Unauthorized, + ))); + } + let Some(row) = guard(&context.sql(&statement(GUARD, vec![blob(input.id)]))?)? else { + return Ok(CommandResult::Success(RefPolicyReapReply::Reaped { + watches: 0, + removed: false, + })); + }; + if row.token.repository != request.repository { + return Err(Error::Command("foreign ref policy guard token")); + } + let at = now(context.now_ms())?; + let live = if row.token.owner == context.owner_fence() { + load(context, row.token)?.is_some_and(|op| op.token == row.token && op.expires > at) + } else { + false + }; + if live + && row.progress.valid + && row.epoch == epoch(&context.sql(&statement(EPOCH, vec![]))?)? + { + return Ok(CommandResult::Rejected(RefPolicyReapReply::Denied( + PreparationDenial::Conflict, + ))); + } + if row.progress.valid { + changed(context.sql(&statement( + "UPDATE ref_policy_guards SET valid=0 WHERE id=?1 AND valid=1", + vec![blob(input.id)], + ))?)?; + } + let deleted=context.sql(&statement("DELETE FROM ref_policy_watches WHERE guard=?1 AND (oid,context,context_version,run_number) IN (SELECT oid,context,context_version,run_number FROM ref_policy_watches WHERE guard=?1 ORDER BY oid,context,context_version,run_number LIMIT ?2)",vec![blob(input.id),number(WATCH_REAP_ROWS)?]))?; + let watches = deleted + .first() + .ok_or(Error::Command("missing watch reap result"))? + .rows_affected; + if watches > WATCH_REAP_ROWS { + return Err(Error::Command("watch reaper exceeded its budget")); + } + changed(context.sql(&statement( + "UPDATE ref_policy_budget SET watches=watches-?1 WHERE singleton=1 AND watches>=?1", + vec![number(watches)?], + ))?)?; + let remaining = context.sql(&statement( + "SELECT EXISTS(SELECT 1 FROM ref_policy_watches WHERE guard=?1)", + vec![blob(input.id)], + ))?; + let Some([SqlValue::Integer(remaining)]) = rows(&remaining)?.first().map(Vec::as_slice) + else { + return Err(Error::Command("missing remaining watches")); + }; + let removed = !live && *remaining == 0; + if removed { + changed(context.sql(&statement( + "DELETE FROM ref_policy_guards WHERE id=?1", + vec![blob(input.id)], + ))?)?; + } + Ok(CommandResult::Success(RefPolicyReapReply::Reaped { + watches, + removed, + })) + } +} diff --git a/crates/canopy-server/src/packs/publication/ref_policy/mod.rs b/crates/canopy-server/src/packs/publication/ref_policy/mod.rs new file mode 100644 index 0000000..ad44f46 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/ref_policy/mod.rs @@ -0,0 +1,167 @@ +//! Paged direct-ref policy predicates with exact check dependency invalidation. +//! Raw transport DTOs grant no authority. Final root publication must check the +//! live guard and current epoch in its owner-fenced root/outcome transaction. +use super::*; +use crate::{PushPlan, packs::metadata::MetadataLimits}; +use cellule_ltx::DiskBudget; +use cellule_runtime::{InvocationError, primitives::sql::SqlCell}; +use std::path::Path; +use tokio::time::timeout_at; + +mod codec; +mod commands; +mod prepare; +pub use commands::{CheckRefPolicyGuard, ReapRefPolicyGuard, RegisterRefPolicyPage}; +pub(super) use prepare::ensure_ready; + +pub const REF_POLICY_PAGE_BYTES: u32 = 256 << 10; +pub const REF_POLICY_PAGE_UPDATES: usize = 128; +pub const MAX_REF_POLICY_GUARDS: u64 = 4096; +pub const MAX_REF_POLICY_WATCHES: u64 = 2_097_152; +const WATCH_REAP_ROWS: u64 = 512; +const TOKEN_BYTES: u32 = 256; + +/// Generation-independent intent. The existing catalog certificate binds its +/// exact proposal separately, while this guard can survive unrelated rebases. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct RefPolicyIntent { + pub id: [u8; 16], + pub epoch: u64, + pub updates: u64, + pub plan_digest: [u8; 32], + pub evidence_digest: [u8; 32], +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct RefPolicyPage { + pub intent: RefPolicyIntent, + pub offset: u64, + pub proof: RefPublicationProof, +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct RefPolicyProgress { + pub next: u64, + pub total: u64, + pub valid: bool, +} +impl RefPolicyProgress { + pub fn ready(self) -> bool { + self.valid && self.next == self.total + } +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum RefPolicyReply { + Registered(RefPolicyProgress), + Denied(PreparationDenial), +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct RefPolicyLookup { + pub check: LeaseCheck, + pub intent: RefPolicyIntent, +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct RefPolicyReap { + pub maintenance: MaintenanceRequest, + pub id: [u8; 16], +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum RefPolicyReapReply { + Reaped { watches: u64, removed: bool }, + Denied(PreparationDenial), +} +/// Owned immutable original intent and evidence. Observers retain this value +/// and the exact issued page/MutationIdentity through uncertain page outcomes. +pub struct RefPolicyPreparation { + intent: RefPolicyIntent, + token: PreparationToken, + format: ObjectFormat, + plan: PushPlan, + ancestry: Vec, +} +/// Private readiness result. Final signing rechecks catalog evidence and the +/// conditional ref transition; final execution must recheck guard freshness. +pub struct PreparedRefPolicyGuard { + intent: RefPolicyIntent, + token: PreparationToken, + actor: String, + format: ObjectFormat, +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct RefRootPublicationProof { + pub certificate: CatalogCertificate, + pub guard: RefPolicyIntent, + pub snapshot: RefStateSnapshotRoot, +} +#[derive(Debug, thiserror::Error)] +pub enum RefPolicyPreparationError { + #[error("ref policy preparation is inactive")] + Base(#[from] PreparationBaseError), + #[error("ref policy catalog evidence failed")] + Evidence(#[from] RefProofError), + #[error("ref policy conditional snapshot failed")] + Snapshot(#[from] RefSnapshotPreparationError), + #[error("ref policy certificate failed")] + Certificate(#[from] CatalogAttestationError), + #[error("ref policy encoding failed")] + Codec(#[from] CodecError), + #[error("ref policy SQL capability failed")] + Capability(#[from] Error), + #[error("ref policy epoch query failed")] + Query(#[source] Box>>), + #[error("ref policy guard query failed")] + Guard(#[source] Box>>), + #[error("ref policy intent, readiness or custody differs")] + Context, +} + +fn scope( + token: PreparationToken, + actor: &str, + format: ObjectFormat, + intent: RefPolicyIntent, +) -> Result<[u8; 32], CodecError> { + let mut e = BoundedEncoder::new(512)?; + token.encode(&mut e)?; + e.write_text(actor)?; + e.write_u8(format.bytes() as u8)?; + intent.encode(&mut e)?; + let mut h = blake3::Hasher::new(); + h.update(b"canopy.ref-policy-scope.v1\0"); + h.update(&e.finish()); + Ok(*h.finalize().as_bytes()) +} +fn page_binding(page: &RefPolicyPage) -> Result<[u8; 32], CodecError> { + page.shape()?; + page_payload_binding( + page.intent, + page.offset, + &page.proof.plan, + &page.proof.ancestry, + ) +} +fn page_payload_binding( + intent: RefPolicyIntent, + offset: u64, + plan: &PushPlan, + ancestry: &[u8], +) -> Result<[u8; 32], CodecError> { + let mut e = BoundedEncoder::new(256)?; + intent.encode(&mut e)?; + e.write_u64(offset)?; + e.write_bytes(&super::ref_proof::binding(plan, ancestry)?)?; + let mut h = blake3::Hasher::new(); + h.update(b"canopy.ref-policy-page.v1\0"); + h.update(&e.finish()); + Ok(*h.finalize().as_bytes()) +} +pub(super) fn root_binding( + intent: RefPolicyIntent, + snapshot: RefStateSnapshotRoot, +) -> Result<[u8; 32], CodecError> { + let mut e = BoundedEncoder::new(512)?; + intent.encode(&mut e)?; + snapshot.encode(&mut e)?; + let mut h = blake3::Hasher::new(); + h.update(b"canopy.ref-root-publication.v1\0"); + h.update(&e.finish()); + Ok(*h.finalize().as_bytes()) +} diff --git a/crates/canopy-server/src/packs/publication/ref_policy/prepare.rs b/crates/canopy-server/src/packs/publication/ref_policy/prepare.rs new file mode 100644 index 0000000..fd08b73 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/ref_policy/prepare.rs @@ -0,0 +1,225 @@ +use super::*; + +impl PreparedCatalog { + /// Freeze original intent and catalog-verified evidence before minting + /// pages. Caller roots, page lists and decoded certificates cannot enter. + pub async fn ref_policy_preparation( + &self, + plan: PushPlan, + root: &Path, + budget: DiskBudget, + limits: MetadataLimits, + ) -> Result { + let (_, deadline) = self.base.live_lease()?; + timeout_at(deadline, async { + if self.base().refs.is_none() { + return Err(RefPolicyPreparationError::Context); + } + let (client, target, _) = self.base.capability(); + let sql = SqlCell::::new(client.clone(), target.clone())?; + let epoch = commands::epoch( + &sql.query( + None, + SqlBatch { + statements: vec![SqlStatement { + sql: commands::EPOCH.into(), + parameters: vec![], + }], + }, + ) + .await + .map_err(|error| RefPolicyPreparationError::Query(Box::new(error)))? + .output, + )?; + let (plan, ancestry) = self.ref_evidence(plan, root, budget, limits).await?; + let intent = RefPolicyIntent { + id: *uuid::Uuid::new_v4().as_bytes(), + epoch, + updates: plan.updates.len() as u64, + plan_digest: super::super::ref_proof::plan_digest(&plan)?, + evidence_digest: super::super::ref_proof::binding(&plan, &ancestry)?, + }; + self.ensure_live()?; + Ok(RefPolicyPreparation { + intent, + token: self.token(), + format: self.catalog().format, + plan, + ancestry, + }) + }) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } + /// Conditional root certification only. The admitted final root/outcome + /// command must still check this live guard, actual fence, ACL/pin and CAS. + pub async fn guarded_ref_snapshot( + &self, + guard: &PreparedRefPolicyGuard, + plan: PushPlan, + root: &Path, + budget: DiskBudget, + limits: MetadataLimits, + ) -> Result { + let (_, deadline) = self.base.live_lease()?; + timeout_at(deadline, async { + if guard.token != self.token() + || guard.actor != self.base.capability().2.actor + || guard.format != self.catalog().format + || guard.intent.plan_digest != super::super::ref_proof::plan_digest(&plan)? + { + return Err(RefPolicyPreparationError::Context); + } + let (plan, ancestry) = self.ref_evidence(plan, root, budget, limits).await?; + if super::super::ref_proof::binding(&plan, &ancestry)? != guard.intent.evidence_digest { + return Err(RefPolicyPreparationError::Context); + } + let snapshot = self.prepare_ref_snapshot(&plan).await?; + if snapshot.base() != self.base() || snapshot.plan_digest() != guard.intent.plan_digest + { + return Err(RefPolicyPreparationError::Context); + } + ensure_ready(self, guard.intent).await?; + let snapshot = snapshot.snapshot(); + let certificate = self + .issue_certificate(Some(root_binding(guard.intent, snapshot)?), None) + .await?; + let value = RefRootPublicationProof { + certificate, + guard: guard.intent, + snapshot, + }; + value.shape()?; + self.ensure_live()?; + Ok(value) + }) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } +} +pub(in crate::packs::publication) async fn ensure_ready( + prepared: &PreparedCatalog, + intent: RefPolicyIntent, +) -> Result<(), RefPolicyPreparationError> { + let (client, target, check) = prepared.base.capability(); + let progress = client + .query::( + target, + None, + RefPolicyLookup { + check: check.clone(), + intent, + }, + ) + .await + .map_err(|error| RefPolicyPreparationError::Guard(Box::new(error)))? + .output; + if !progress.is_some_and(|progress| progress.total == intent.updates && progress.ready()) { + return Err(RefPolicyPreparationError::Context); + } + prepared.ensure_live()?; + Ok(()) +} +impl RefPolicyPreparation { + pub fn intent(&self) -> RefPolicyIntent { + self.intent + } + pub fn plan(&self) -> &PushPlan { + &self.plan + } + pub fn into_plan(self) -> PushPlan { + self.plan + } + fn matches(&self, prepared: &PreparedCatalog) -> bool { + self.token == prepared.token() + && self.format == prepared.catalog().format + && self.plan.actor == prepared.base.capability().2.actor + } + /// At most 128 updates and 256 KiB, including certificate/framing. Byte + /// boundaries may split ancestry bytes; copy only this bounded page. + pub async fn page( + &self, + prepared: &PreparedCatalog, + start: usize, + ) -> Result { + let (_, deadline) = prepared.base.live_lease()?; + timeout_at(deadline, async { + if !self.matches(prepared) || start >= self.plan.updates.len() { + return Err(RefPolicyPreparationError::Context); + } + let mut used = CERTIFICATE_BYTES as usize + 512; + let mut end = start; + while end < self.plan.updates.len() && end - start < REF_POLICY_PAGE_UPDATES { + let mut e = BoundedEncoder::new(REF_POLICY_PAGE_BYTES)?; + crate::refs::encode_update(&self.plan.updates[end], &mut e)?; + let n = e.finish().len(); + if used + .checked_add(n) + .is_none_or(|bytes| bytes > REF_POLICY_PAGE_BYTES as usize) + { + break; + } + used += n; + end += 1; + } + if end == start { + return Err(CodecError::Limit.into()); + } + let plan = PushPlan { + actor: self.plan.actor.clone(), + updates: self.plan.updates[start..end].to_vec(), + }; + let mut ancestry = vec![0; plan.updates.len().div_ceil(8)]; + for i in 0..plan.updates.len() { + if super::super::ref_proof::proven(&self.ancestry, start + i) { + ancestry[i / 8] |= 1 << (i % 8); + } + } + let certificate = prepared + .issue_certificate( + Some(page_payload_binding( + self.intent, + start as u64, + &plan, + &ancestry, + )?), + None, + ) + .await?; + let page = RefPolicyPage { + intent: self.intent, + offset: start as u64, + proof: RefPublicationProof { + plan, + certificate, + ancestry, + }, + }; + page.encode(&mut BoundedEncoder::new(REF_POLICY_PAGE_BYTES)?)?; + prepared.ensure_live()?; + Ok(page) + }) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } + pub async fn ready( + &self, + prepared: &PreparedCatalog, + ) -> Result { + let (_, deadline) = prepared.base.live_lease()?; + timeout_at(deadline, async { + if !self.matches(prepared) { + return Err(RefPolicyPreparationError::Context); + } + ensure_ready(prepared, self.intent).await?; + Ok(PreparedRefPolicyGuard { + intent: self.intent, + token: self.token, + actor: self.plan.actor.clone(), + format: self.format, + }) + }) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } +} diff --git a/crates/canopy-server/src/packs/publication/ref_policy/schema.sql b/crates/canopy-server/src/packs/publication/ref_policy/schema.sql new file mode 100644 index 0000000..c14c601 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/ref_policy/schema.sql @@ -0,0 +1,136 @@ +-- Ephemeral direct-ref publication predicates. They are independent of the +-- catalog/ref generation, allowing unrelated root advancement during paging. +CREATE TABLE ref_policy_epoch ( + singleton INTEGER PRIMARY KEY CHECK(singleton=1), + version INTEGER NOT NULL CHECK(typeof(version)='integer' AND version>=0) +) WITHOUT ROWID; +INSERT INTO ref_policy_epoch VALUES(1,0); +CREATE TRIGGER ref_policy_epoch_monotonic BEFORE UPDATE ON ref_policy_epoch +WHEN NEW.singleton!=OLD.singleton OR NEW.version!=OLD.version+1 +BEGIN SELECT RAISE(ABORT,'ref policy epoch must advance once'); END; +CREATE TRIGGER ref_policy_epoch_retained BEFORE DELETE ON ref_policy_epoch +BEGIN SELECT RAISE(ABORT,'ref policy epoch must be retained'); END; +CREATE TRIGGER ref_policy_epoch_not_replaced BEFORE INSERT ON ref_policy_epoch +WHEN EXISTS(SELECT 1 FROM ref_policy_epoch WHERE singleton=NEW.singleton) +BEGIN SELECT RAISE(ABORT,'ref policy epoch cannot be replaced'); END; + +CREATE TABLE ref_policy_guards ( + id BLOB PRIMARY KEY CHECK(length(id)=16), + scope BLOB NOT NULL CHECK(length(scope)=32), + token BLOB NOT NULL CHECK(length(token) BETWEEN 1 AND 256), + policy_epoch INTEGER NOT NULL CHECK(typeof(policy_epoch)='integer' AND policy_epoch>=0), + total INTEGER NOT NULL CHECK(typeof(total)='integer' AND total BETWEEN 1 AND 100000), + next INTEGER NOT NULL CHECK(typeof(next)='integer' AND next BETWEEN 0 AND total), + valid INTEGER NOT NULL CHECK(valid IN (0,1)) +) WITHOUT ROWID; +CREATE TRIGGER ref_policy_guards_immutable BEFORE UPDATE ON ref_policy_guards +WHEN NEW.id!=OLD.id OR NEW.scope!=OLD.scope OR NEW.token!=OLD.token + OR NEW.policy_epoch!=OLD.policy_epoch OR NEW.total!=OLD.total + OR NEW.nextOLD.valid +BEGIN SELECT RAISE(ABORT,'ref policy guard cannot reset'); END; +CREATE TRIGGER ref_policy_guards_not_replaced BEFORE INSERT ON ref_policy_guards +WHEN EXISTS(SELECT 1 FROM ref_policy_guards WHERE id=NEW.id) +BEGIN SELECT RAISE(ABORT,'ref policy guard cannot be replaced'); END; + +CREATE TABLE ref_policy_watches ( + guard BLOB NOT NULL REFERENCES ref_policy_guards(id), + oid BLOB NOT NULL CHECK(length(oid) IN (20,32)), + context TEXT NOT NULL, + context_version INTEGER NOT NULL CHECK(typeof(context_version)='integer' AND context_version>0), + run_number INTEGER NOT NULL CHECK(typeof(run_number)='integer' AND run_number>0), + PRIMARY KEY(guard,oid,context,context_version,run_number) +) WITHOUT ROWID; +CREATE INDEX ref_policy_watches_by_check +ON ref_policy_watches(oid,context,context_version,run_number,guard); +CREATE TRIGGER ref_policy_watches_immutable BEFORE UPDATE ON ref_policy_watches +BEGIN SELECT RAISE(ABORT,'ref policy watch is immutable'); END; +CREATE TRIGGER ref_policy_watches_not_replaced BEFORE INSERT ON ref_policy_watches +WHEN EXISTS(SELECT 1 FROM ref_policy_watches WHERE guard=NEW.guard AND oid=NEW.oid + AND context=NEW.context AND context_version=NEW.context_version AND run_number=NEW.run_number) +BEGIN SELECT RAISE(ABORT,'ref policy watch cannot be replaced'); END; +CREATE TRIGGER ref_policy_watches_live BEFORE DELETE ON ref_policy_watches +WHEN EXISTS(SELECT 1 FROM ref_policy_guards WHERE id=OLD.guard AND valid=1) +BEGIN SELECT RAISE(ABORT,'live ref policy watch must be retained'); END; +CREATE TABLE ref_policy_budget ( + singleton INTEGER PRIMARY KEY CHECK(singleton=1), + watches INTEGER NOT NULL CHECK(typeof(watches)='integer' AND watches BETWEEN 0 AND 2097152) +) WITHOUT ROWID; +INSERT INTO ref_policy_budget VALUES(1,0); +CREATE TRIGGER ref_policy_budget_retained BEFORE DELETE ON ref_policy_budget +BEGIN SELECT RAISE(ABORT,'ref policy budget must be retained'); END; +CREATE TRIGGER ref_policy_budget_not_replaced BEFORE INSERT ON ref_policy_budget +WHEN EXISTS(SELECT 1 FROM ref_policy_budget WHERE singleton=NEW.singleton) +BEGIN SELECT RAISE(ABORT,'ref policy budget cannot be replaced'); END; +CREATE TRIGGER ref_policy_required_capacity_insert BEFORE INSERT ON branch_required_checks +WHEN NOT EXISTS(SELECT 1 FROM branch_required_checks WHERE reference=NEW.reference AND context=NEW.context) + AND (SELECT count(*) FROM branch_required_checks WHERE reference=NEW.reference)>=16 +BEGIN SELECT RAISE(ABORT,'too many required branch checks'); END; +CREATE TRIGGER ref_policy_required_capacity_update BEFORE UPDATE ON branch_required_checks +WHEN NEW.reference!=OLD.reference + AND NOT EXISTS(SELECT 1 FROM branch_required_checks WHERE reference=NEW.reference AND context=NEW.context) + AND (SELECT count(*) FROM branch_required_checks WHERE reference=NEW.reference)>=16 +BEGIN SELECT RAISE(ABORT,'too many required branch checks'); END; + +-- Rule/context configuration changes are rare. Actual run reports do not +-- advance this epoch; their exact dependencies are invalidated below. +CREATE TRIGGER ref_policy_rule_insert AFTER INSERT ON branch_rules +BEGIN UPDATE ref_policy_epoch SET version=version+1 WHERE singleton=1; END; +CREATE TRIGGER ref_policy_rule_update AFTER UPDATE ON branch_rules +BEGIN UPDATE ref_policy_epoch SET version=version+1 WHERE singleton=1; END; +CREATE TRIGGER ref_policy_rule_delete AFTER DELETE ON branch_rules +BEGIN UPDATE ref_policy_epoch SET version=version+1 WHERE singleton=1; END; +CREATE TRIGGER ref_policy_required_insert AFTER INSERT ON branch_required_checks +BEGIN UPDATE ref_policy_epoch SET version=version+1 WHERE singleton=1; END; +CREATE TRIGGER ref_policy_required_update AFTER UPDATE ON branch_required_checks +BEGIN UPDATE ref_policy_epoch SET version=version+1 WHERE singleton=1; END; +CREATE TRIGGER ref_policy_required_delete AFTER DELETE ON branch_required_checks +BEGIN UPDATE ref_policy_epoch SET version=version+1 WHERE singleton=1; END; +CREATE TRIGGER ref_policy_context_insert AFTER INSERT ON check_contexts +BEGIN UPDATE ref_policy_epoch SET version=version+1 WHERE singleton=1; END; +CREATE TRIGGER ref_policy_context_update AFTER UPDATE ON check_contexts +BEGIN UPDATE ref_policy_epoch SET version=version+1 WHERE singleton=1; END; +CREATE TRIGGER ref_policy_context_delete AFTER DELETE ON check_contexts +BEGIN UPDATE ref_policy_epoch SET version=version+1 WHERE singleton=1; END; + +-- The watched run is exactly the newest attempt of the required version. +-- Newer attempts invalidate even when queued. Older/different-version reports +-- and unrelated commits/contexts leave the guard unchanged. +-- REPLACE may suppress DELETE triggers. Observe the row it would replace by +-- either unique key before insertion, even when the replacement changes pair. +CREATE TRIGGER ref_policy_check_replacement BEFORE INSERT ON check_runs +BEGIN + UPDATE ref_policy_guards SET valid=0 WHERE valid=1 AND id IN ( + SELECT w.guard FROM check_runs r JOIN ref_policy_watches w + ON w.oid=r.oid AND w.context=r.context + AND w.context_version=r.context_version AND w.run_number=r.number + WHERE r.number=NEW.number OR r.id=NEW.id + ); +END; +CREATE TRIGGER ref_policy_check_insert AFTER INSERT ON check_runs +BEGIN + UPDATE ref_policy_guards SET valid=0 WHERE valid=1 AND id IN ( + SELECT guard FROM ref_policy_watches + WHERE oid=NEW.oid AND context=NEW.context + AND context_version=NEW.context_version AND run_number<=NEW.number + ); +END; +CREATE TRIGGER ref_policy_check_update AFTER UPDATE ON check_runs +BEGIN + UPDATE ref_policy_guards SET valid=0 WHERE valid=1 AND id IN ( + SELECT guard FROM ref_policy_watches + WHERE oid=OLD.oid AND context=OLD.context + AND context_version=OLD.context_version AND run_number=OLD.number + UNION + SELECT guard FROM ref_policy_watches + WHERE oid=NEW.oid AND context=NEW.context + AND context_version=NEW.context_version AND run_number<=NEW.number + ); +END; +CREATE TRIGGER ref_policy_check_delete AFTER DELETE ON check_runs +BEGIN + UPDATE ref_policy_guards SET valid=0 WHERE valid=1 AND id IN ( + SELECT guard FROM ref_policy_watches + WHERE oid=OLD.oid AND context=OLD.context + AND context_version=OLD.context_version AND run_number=OLD.number + ); +END; diff --git a/crates/canopy-server/src/packs/publication/ref_proof.rs b/crates/canopy-server/src/packs/publication/ref_proof.rs new file mode 100644 index 0000000..8f69525 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/ref_proof.rs @@ -0,0 +1,308 @@ +//! Privately prepared catalog -> exact ref targets and conditional ancestry. +//! Decoded proof DTOs are untrusted. Only MAC verification in an admitted final +//! command can turn them into publication authority. +use super::*; +use crate::packs::{ + catalog::CatalogReader, + directory::index::IndexError, + metadata::{MetadataError, MetadataLimits, PAGE_OBJECTS}, +}; +use crate::{ObjectId, ObjectKind, PushPlan}; +use cellule_ltx::DiskBudget; +use cellule_runtime::{InvocationError, primitives::sql::SqlCell}; +use std::{collections::BTreeSet, path::Path}; +use tokio::time::timeout_at; + +pub(super) mod ancestry; + +#[derive(Debug, thiserror::Error)] +pub enum RefProofError { + #[error("ref proof preparation lease is inactive")] + Base(#[from] PreparationBaseError), + #[error("ref proof catalog lookup failed")] + Catalog(#[from] IndexError), + #[error("ref proof scratch failed")] + Metadata(#[from] MetadataError), + #[error("ref proof issuance failed")] + Attestation(#[from] CatalogAttestationError), + #[error("ref proof SQL capability failed")] + Capability(#[from] Error), + #[error("ref proof policy query failed")] + Query(#[source] Box>>), + #[error("ref proof encoding failed")] + Codec(#[from] CodecError), + #[error("ref plan has invalid or inconsistent targets")] + Invalid, + #[error("ref proof worker failed")] + Task(#[from] tokio::task::JoinError), + #[error("ref proof preparation was canceled")] + Canceled, +} + +/// Bounded evidence accompanying the existing PushPlan. Public fields permit +/// transport/inspection only; edits invalidate the signed binding. The ancestry +/// bit proves the fast-forward predicate (including vacuous creation/deletion +/// and identical tips); deletion permissions remain a separate current policy. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct RefPublicationProof { + pub plan: PushPlan, + pub certificate: CatalogCertificate, + pub ancestry: Vec, +} + +impl WireValue for RefPublicationProof { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + evidence_shape(&self.plan, &self.ancestry)?; + self.plan.encode(e)?; + self.certificate.encode(e)?; + e.write_bytes(&self.ancestry) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let plan = PushPlan::decode(d)?; + let certificate = CatalogCertificate::decode(d)?; + let ancestry = d.read_bytes()?; + evidence_shape(&plan, ancestry)?; + Ok(Self { + plan, + certificate, + ancestry: ancestry.to_vec(), + }) + } +} +pub(super) fn proven(bits: &[u8], index: usize) -> bool { + bits.get(index / 8) + .is_some_and(|byte| byte & (1 << (index % 8)) != 0) +} +fn mark(bits: &mut [u8], index: usize) { + bits[index / 8] |= 1 << (index % 8); +} +fn evidence_shape(plan: &PushPlan, bits: &[u8]) -> Result<(), CodecError> { + let n = plan.updates.len(); + if n == 0 + || n > crate::refs::MAX_UPDATES + || bits.len() != n.div_ceil(8) + || (!n.is_multiple_of(8) && bits.last().is_some_and(|byte| *byte >> (n % 8) != 0)) + { + return Err(CodecError::Invalid("invalid ref ancestry evidence")); + } + Ok(()) +} +/// Hash the existing wire bytes in bounded chunks. Even a long valid ref name +/// is fed directly, without another full plan/name allocation. +pub(super) fn binding(plan: &PushPlan, bits: &[u8]) -> Result<[u8; 32], CodecError> { + evidence_shape(plan, bits)?; + let mut hash = blake3::Hasher::new(); + hash.update(b"canopy.ref-publication.v1\0"); + hash.update(&plan_digest(plan)?); + let mut count = BoundedEncoder::new(4)?; + count.write_count(bits.len())?; + hash.update(&count.finish()); + hash.update(bits); + Ok(*hash.finalize().as_bytes()) +} +pub(in crate::packs) fn plan_digest(plan: &PushPlan) -> Result<[u8; 32], CodecError> { + let mut hash = blake3::Hasher::new(); + hash.update(b"canopy.ref-plan.v1\0"); + let mut prefix = BoundedEncoder::new(128)?; + plan.encode_prefix(&mut prefix)?; + hash.update(&prefix.finish()); + for update in &plan.updates { + let mut name = BoundedEncoder::new(4)?; + name.write_count(update.name.len())?; + hash.update(&name.finish()); + hash.update(update.name.as_bytes()); + let mut suffix = BoundedEncoder::new(128)?; + crate::refs::encode_update_suffix(update, &mut suffix)?; + hash.update(&suffix.finish()); + } + Ok(*hash.finalize().as_bytes()) +} +pub(in crate::packs) fn shape(plan: &PushPlan, format: ObjectFormat) -> Result<(), RefProofError> { + let mut prefix = BoundedEncoder::new(128)?; + plan.encode_prefix(&mut prefix)?; + let mut names = BTreeSet::new(); + for update in &plan.updates { + if !crate::refs::valid_ref_name(&update.name) + || crate::refs::server_owned_ref(&update.name) + || !names.insert(update.name.as_str()) + || update + .expected + .as_ref() + .is_some_and(|old| old.version <= 0 || old.version == i64::MAX) + || [ + update.new_oid, + update.expected.as_ref().and_then(|old| old.oid), + ] + .into_iter() + .flatten() + .any(|oid| oid.format() != format || oid.is_zero()) + || (update.new_oid.is_none() + && update.expected.as_ref().and_then(|old| old.oid).is_none()) + { + return Err(RefProofError::Invalid); + } + } + Ok(()) +} +const POLICY_QUERY_BYTES: u32 = 256 << 10; +fn ancestry_policy_statement(update: &crate::RefUpdate) -> SqlStatement { + SqlStatement { + sql: "SELECT EXISTS(SELECT 1 FROM branch_rules WHERE reference=?1 AND enabled=1 AND fast_forward=1)".into(), + parameters: vec![SqlValue::Text(update.name.clone())], + } +} +/// Count the exact existing SQL wire encoding before accumulating statements. +/// Valid long ref names must not turn a 128-row query into an oversized request. +fn ancestry_policy_page( + updates: &[crate::RefUpdate], + start: usize, +) -> Result<(usize, SqlBatch), CodecError> { + let mut prefix = BoundedEncoder::new(4)?; + prefix.write_count(0)?; + let mut used = prefix.finish().len(); + let mut statements = Vec::new(); + let mut end = start; + while end < updates.len() && statements.len() < 128 { + let statement = ancestry_policy_statement(&updates[end]); + let mut e = BoundedEncoder::new(POLICY_QUERY_BYTES)?; + statement.encode(&mut e)?; + let n = e.finish().len(); + if used + .checked_add(n) + .is_none_or(|bytes| bytes > POLICY_QUERY_BYTES as usize) + { + break; + } + used += n; + statements.push(statement); + end += 1; + } + if statements.is_empty() { + return Err(CodecError::Limit); + } + Ok((end, SqlBatch { statements })) +} +impl PreparedCatalog { + /// Validate targets through this prepared catalog. Ancestry is computed + /// only for currently enabled fast-forward rules; final publication checks + /// the current rule again and denies a newly required missing proof. + pub async fn ref_proof( + &self, + plan: PushPlan, + root: &Path, + budget: DiskBudget, + limits: MetadataLimits, + ) -> Result { + let (_, deadline) = self.base.live_lease()?; + timeout_at(deadline, self.ref_proof_inner(plan, root, budget, limits)) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } + async fn ref_proof_inner( + &self, + plan: PushPlan, + root: &Path, + budget: DiskBudget, + limits: MetadataLimits, + ) -> Result { + let (plan, bits) = self.ref_evidence(plan, root, budget, limits).await?; + let certificate = self + .issue_certificate(Some(binding(&plan, &bits)?), None) + .await?; + self.ensure_live()?; + Ok(RefPublicationProof { + plan, + certificate, + ancestry: bits, + }) + } + /// Checks the private catalog once. Completion signs these facts together + /// with the native outcome, avoiding an intermediate certificate/query. + pub(super) async fn ref_evidence( + &self, + plan: PushPlan, + root: &Path, + budget: DiskBudget, + limits: MetadataLimits, + ) -> Result<(PushPlan, Vec), RefProofError> { + shape(&plan, self.catalog().format)?; + if plan.actor != self.base.capability().2.actor { + return Err(RefProofError::Invalid); + } + let reader = CatalogReader::open(self.base.indexes(), self.catalog()).await?; + let files = self.base.files(); + for updates in plan.updates.chunks(PAGE_OBJECTS) { + self.ensure_live()?; + let ids: Vec<_> = updates.iter().filter_map(|update| update.new_oid).collect(); + let headers = reader.headers(&ids, &*files, &*files).await?; + for (update, header) in updates + .iter() + .filter(|update| update.new_oid.is_some()) + .zip(headers) + { + let header = header.ok_or(RefProofError::Invalid)?; + if Some(header.object.oid) != update.new_oid + || (update.name.starts_with("refs/heads/") + && header.object.kind != ObjectKind::Commit) + { + return Err(RefProofError::Invalid); + } + } + } + let (client, target, _) = self.base.capability(); + let sql = SqlCell::::new(client.clone(), target.clone())?; + let mut bits = vec![0; plan.updates.len().div_ceil(8)]; + let mut walk: Option = None; + let mut start = 0; + while start < plan.updates.len() { + self.ensure_live()?; + let (end, batch) = ancestry_policy_page(&plan.updates, start)?; + let policies = sql + .query(None, batch) + .await + .map_err(|error| RefProofError::Query(Box::new(error)))?; + let updates = &plan.updates[start..end]; + if policies.output.len() != updates.len() { + return Err(RefProofError::Invalid); + } + for (at, (update, policy)) in updates.iter().zip(policies.output).enumerate() { + let index = start + at; + let required = match policy.rows.first().map(Vec::as_slice) { + Some([SqlValue::Integer(0)]) => false, + Some([SqlValue::Integer(1)]) => true, + _ => return Err(RefProofError::Invalid), + }; + let old = update.expected.as_ref().and_then(|old| old.oid); + if old.is_none() || old == update.new_oid || update.new_oid.is_none() { + mark(&mut bits, index); + continue; + } + if required { + if walk.is_none() { + walk = Some(ancestry::Walker::new(root, budget.clone(), limits).await?); + } + if walk + .as_mut() + .ok_or(RefProofError::Invalid)? + .is_ancestor( + &reader, + &files, + old.ok_or(RefProofError::Invalid)?, + update.new_oid.ok_or(RefProofError::Invalid)?, + &self.base, + ) + .await? + { + mark(&mut bits, index); + } + } + } + start = end; + } + self.ensure_live()?; + Ok((plan, bits)) + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/canopy-server/src/packs/publication/ref_proof/ancestry.rs b/crates/canopy-server/src/packs/publication/ref_proof/ancestry.rs new file mode 100644 index 0000000..8c5adba --- /dev/null +++ b/crates/canopy-server/src/packs/publication/ref_proof/ancestry.rs @@ -0,0 +1,348 @@ +//! Bounded heap, admitted disk traversal of certified commit-parent metadata. +//! The indexed queue and memoized pair results are disposable private scratch. +use super::*; +use crate::packs::{ + catalog::CatalogFiles, + metadata::{AdmittedFile, TypedEdge, growth}, +}; +use rusqlite::{Connection, OptionalExtension, params}; +use std::sync::{ + Arc, Mutex, + atomic::{AtomicBool, Ordering}, +}; + +const CLEAR_PAGE: &str = + "DELETE FROM visits WHERE oid IN (SELECT oid FROM visits ORDER BY oid LIMIT ?1)"; + +struct Cancellation(Arc); +impl Drop for Cancellation { + fn drop(&mut self) { + self.0.store(true, Ordering::Release); + } +} + +struct Scratch { + connection: Connection, + _file: AdmittedFile, + canceled: Arc, + limits: MetadataLimits, +} +impl Scratch { + fn clear_page(&mut self) -> Result { + self.write(|tx| { + tx.execute(CLEAR_PAGE, [PAGE_OBJECTS as i64]) + .map_err(MetadataError::from) + .map_err(Into::into) + }) + } + fn write( + &mut self, + mut body: impl FnMut(&rusqlite::Transaction<'_>) -> Result, + ) -> Result { + let canceled = Arc::clone(&self.canceled); + growth::transaction( + &mut self.connection, + &mut self._file, + self.limits.max_file_bytes, + |tx| { + if canceled.load(Ordering::Acquire) { + return Err(RefProofError::Canceled); + } + let result = body(tx)?; + if canceled.load(Ordering::Acquire) { + return Err(RefProofError::Canceled); + } + Ok(result) + }, + ) + } +} +pub(in crate::packs::publication) struct Walker { + scratch: Arc>, + _cancel: Cancellation, + catalog: Option, + failed: bool, +} +impl Walker { + pub(in crate::packs::publication) async fn new( + root: &Path, + budget: DiskBudget, + limits: MetadataLimits, + ) -> Result { + let root = root.to_owned(); + let canceled = Arc::new(AtomicBool::new(false)); + let cancel = Cancellation(Arc::clone(&canceled)); + let scratch = tokio::task::spawn_blocking(move || { + if canceled.load(Ordering::Acquire) { return Err(RefProofError::Canceled); } + if limits.cache_kib == 0 || limits.cache_kib > i32::MAX as u32 || limits.max_file_bytes < 16 << 10 || limits.max_file_bytes > canopy_object_storage::external::MAX_ARTIFACT_BYTES || !limits.max_file_bytes.is_multiple_of(4096) { return Err(MetadataError::Limit.into()); } + let reservation = growth::reserve(&budget, limits.max_file_bytes)?; + let workspace = Arc::new(tempfile::Builder::new().prefix("canopy-ref-ancestry-").tempdir_in(root).map_err(MetadataError::from)?); + let file = tempfile::Builder::new().prefix("walk-").tempfile_in(workspace.path()).map_err(MetadataError::from)?; + let mut admitted = AdmittedFile::new(file, reservation); + admitted.retain_workspace(workspace); + let connection = Connection::open(admitted.file().path()).map_err(MetadataError::from)?; + connection.execute_batch("PRAGMA page_size=4096; PRAGMA journal_mode=DELETE; PRAGMA synchronous=OFF; PRAGMA trusted_schema=OFF; PRAGMA mmap_size=0;").map_err(MetadataError::from)?; + connection.pragma_update(None, "cache_size", -(limits.cache_kib as i64)).map_err(MetadataError::from)?; + growth::configure(&connection, &mut admitted)?; + let interrupted = Arc::clone(&canceled); + connection.progress_handler(10_000, Some(move || interrupted.load(Ordering::Acquire))); + let mut scratch = Scratch { connection, _file: admitted, canceled, limits }; + scratch.write(|tx| { + tx.execute_batch("CREATE TABLE visits(oid BLOB PRIMARY KEY,expanded INTEGER NOT NULL DEFAULT 0 CHECK(expanded IN (0,1))) WITHOUT ROWID; CREATE INDEX visits_ready ON visits(expanded,oid); CREATE TABLE answers(ancestor BLOB,descendant BLOB,result INTEGER NOT NULL CHECK(result IN(0,1)),PRIMARY KEY(ancestor,descendant)) WITHOUT ROWID;").map_err(MetadataError::from)?; + Ok(()) + })?; + Ok::<_, RefProofError>(Arc::new(Mutex::new(scratch))) + }).await??; + Ok(Self { + scratch, + _cancel: cancel, + catalog: None, + failed: false, + }) + } + async fn call( + &self, + f: impl FnOnce(&mut Scratch) -> Result + Send + 'static, + ) -> Result { + let scratch = Arc::clone(&self.scratch); + tokio::task::spawn_blocking(move || { + let mut scratch = scratch.lock().map_err(|_| RefProofError::Invalid)?; + if scratch.canceled.load(Ordering::Acquire) { + return Err(RefProofError::Canceled); + } + f(&mut scratch) + }) + .await? + } + pub(in crate::packs::publication) async fn is_ancestor( + &mut self, + reader: &CatalogReader, + files: &Arc, + old: ObjectId, + new: ObjectId, + base: &PreparationBaseResolver, + ) -> Result { + // Exclusive borrowing spans the entire traversal, rather than just one + // SQL callback. Failure/cancellation cannot leave a reusable queue or + // let a queued old write contaminate a later pair's proof. + if self.failed || self._cancel.0.load(Ordering::Acquire) { + return Err(RefProofError::Canceled); + } + self.failed = true; + let mut guard = WalkCancellation::new(Arc::clone(&self._cancel.0)); + let catalog = reader.stored(); + if self.catalog.is_some_and(|bound| bound != catalog) + || old.format() != catalog.format + || new.format() != catalog.format + { + return Err(RefProofError::Invalid); + } + self.catalog = Some(catalog); + let result = self.walk(reader, files, old, new, base).await?; + // This also covers identical tips and memo hits after asynchronous + // membership/SQL work; neither may return an answer from an expired + // preparation session. + base.live_lease()?; + guard.complete = true; + self.failed = false; + Ok(result) + } + + async fn walk( + &self, + reader: &CatalogReader, + files: &Arc, + old: ObjectId, + new: ObjectId, + base: &PreparationBaseResolver, + ) -> Result { + base.live_lease()?; + let endpoints = reader.headers(&[old, new], &**files, &**files).await?; + if endpoints.len() != 2 + || endpoints + .iter() + .any(|header| header.is_none_or(|h| h.object.kind != ObjectKind::Commit)) + { + return Err(RefProofError::Invalid); + } + if old == new { + return Ok(true); + } + if let Some(answer) = self + .call(move |scratch| { + scratch + .connection + .query_row( + "SELECT result FROM answers WHERE ancestor=?1 AND descendant=?2", + params![old.as_ref(), new.as_ref()], + |row| row.get::<_, bool>(0), + ) + .optional() + .map_err(MetadataError::from) + .map_err(Into::into) + }) + .await? + { + return Ok(answer); + } + // Reuse the same queue and indexes. Clear in bounded transactions so a + // second pair does not build one history-sized rollback journal. + loop { + base.live_lease()?; + let removed = self.call(Scratch::clear_page).await?; + if removed == 0 { + break; + } + } + self.call(move |scratch| { + scratch.write(|tx| { + tx.execute("INSERT INTO visits(oid) VALUES(?1)", [new.as_ref()]) + .map_err(MetadataError::from)?; + Ok(()) + }) + }) + .await?; + let result = 'walk: loop { + base.live_lease()?; + let page = self + .call(|scratch| { + scratch + .connection + .prepare_cached( + "SELECT oid FROM visits WHERE expanded=0 ORDER BY oid LIMIT ?1", + ) + .map_err(MetadataError::from)? + .query_map([PAGE_OBJECTS as i64], |row| { + crate::packs::metadata::oid(row.get(0)?) + }) + .map_err(MetadataError::from)? + .collect::>>() + .map_err(MetadataError::from) + .map_err(Into::into) + }) + .await?; + if page.is_empty() { + break false; + } + for oid in page { + base.live_lease()?; + let object = reader + .lookup(oid, &**files, &**files) + .await? + .ok_or(RefProofError::Invalid)?; + if object.entry.header.object.kind != ObjectKind::Commit { + return Err(RefProofError::Invalid); + } + let mut after = None; + loop { + base.live_lease()?; + let metadata = Arc::clone(&object.source.metadata); + let edges = + tokio::task::spawn_blocking(move || metadata.edges_after(oid, after)) + .await??; + if edges.is_empty() { + break; + } + after = edges.last().map(|edge| edge.child); + if edges + .iter() + .any(|edge| edge.expected_kind == ObjectKind::Commit && edge.child == old) + { + break 'walk true; + } + self.parents(edges).await?; + } + self.call(move |scratch| { + scratch.write(|tx| { + if tx + .execute( + "UPDATE visits SET expanded=1 WHERE oid=?1 AND expanded=0", + [oid.as_ref()], + ) + .map_err(MetadataError::from)? + != 1 + { + return Err(RefProofError::Invalid); + } + Ok(()) + }) + }) + .await?; + } + }; + base.live_lease()?; + self.call(move |scratch| { + scratch.write(|tx| { + tx.execute( + "INSERT INTO answers VALUES(?1,?2,?3)", + params![old.as_ref(), new.as_ref(), result], + ) + .map_err(MetadataError::from)?; + Ok(()) + }) + }) + .await?; + Ok(result) + } + async fn parents(&self, edges: Vec) -> Result<(), RefProofError> { + if edges.len() > PAGE_OBJECTS { + return Err(RefProofError::Invalid); + } + self.call(move |scratch| { + scratch.write(|tx| { + { + let mut insert = tx + .prepare_cached("INSERT OR IGNORE INTO visits(oid) VALUES(?1)") + .map_err(MetadataError::from)?; + for edge in &edges { + if edge.expected_kind == ObjectKind::Commit { + insert + .execute([edge.child.as_ref()]) + .map_err(MetadataError::from)?; + } + } + } + Ok(()) + }) + }) + .await + } + + #[cfg(test)] + pub(in crate::packs::publication) fn test_blocker( + &self, + entered: tokio::sync::oneshot::Sender<()>, + release: std::sync::mpsc::Receiver<()>, + ) -> tokio::task::JoinHandle> { + let scratch = Arc::clone(&self.scratch); + tokio::task::spawn_blocking(move || { + let _owned = scratch.lock().map_err(|_| RefProofError::Invalid)?; + let _ = entered.send(()); + release.recv().map_err(|_| RefProofError::Invalid)?; + Ok(()) + }) + } +} + +struct WalkCancellation { + canceled: Arc, + complete: bool, +} +impl WalkCancellation { + fn new(canceled: Arc) -> Self { + Self { + canceled, + complete: false, + } + } +} +impl Drop for WalkCancellation { + fn drop(&mut self) { + if !self.complete { + self.canceled.store(true, Ordering::Release); + } + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/canopy-server/src/packs/publication/ref_proof/ancestry/tests.rs b/crates/canopy-server/src/packs/publication/ref_proof/ancestry/tests.rs new file mode 100644 index 0000000..ec60c7e --- /dev/null +++ b/crates/canopy-server/src/packs/publication/ref_proof/ancestry/tests.rs @@ -0,0 +1,158 @@ +use super::*; +type Result = std::result::Result>; +#[tokio::test] +async fn large_ancestry_queue_uses_an_index_and_releases_admitted_scratch() -> Result { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let walker = Walker::new( + root.path(), + budget.clone(), + MetadataLimits { + max_file_bytes: 32 << 20, + cache_kib: 64, + }, + ) + .await?; + assert_eq!(budget.used(), growth::INITIAL_BYTES * 3); + for start in (0..100_001u64).step_by(PAGE_OBJECTS) { + walker + .call(move |scratch| { + scratch.write(|tx| { + { + let mut insert = tx + .prepare_cached("INSERT INTO visits(oid,expanded) VALUES(?1,?2)") + .map_err(MetadataError::from)?; + for n in start..(start + PAGE_OBJECTS as u64).min(100_001) { + let mut oid = [1; 20]; + oid[..8].copy_from_slice(&n.to_be_bytes()); + insert + .execute(params![oid.as_slice(), n < 100_000]) + .map_err(MetadataError::from)?; + } + } + Ok(()) + }) + }) + .await?; + } + walker.call(|scratch| { + let plans=scratch.connection.prepare("EXPLAIN QUERY PLAN SELECT oid FROM visits WHERE expanded=0 ORDER BY oid LIMIT 512").map_err(MetadataError::from)?.query_map([],|row|row.get::<_,String>(3)).map_err(MetadataError::from)?.collect::>>().map_err(MetadataError::from)?; + assert!(plans.iter().any(|plan|plan.contains("SEARCH")&&plan.contains("visits_ready")),"{plans:?}"); + let n:i64=scratch.connection.query_row("SELECT count(*) FROM (SELECT oid FROM visits WHERE expanded=0 ORDER BY oid LIMIT 512)",[],|row|row.get(0)).map_err(MetadataError::from)?; + assert_eq!(n,1); + Ok(()) + }).await?; + walker + .call(|scratch| { + let plans = scratch + .connection + .prepare(&format!("EXPLAIN QUERY PLAN {CLEAR_PAGE}")) + .map_err(MetadataError::from)? + .query_map([PAGE_OBJECTS as i64], |r| r.get::<_, String>(3)) + .map_err(MetadataError::from)? + .collect::>>() + .map_err(MetadataError::from)?; + assert!( + plans + .iter() + .any(|p| p.contains("SEARCH visits USING PRIMARY KEY")), + "{plans:?}" + ); + assert!( + !plans.iter().any(|p| p.contains("TEMP B-TREE")), + "{plans:?}" + ); + scratch.write(|tx| { + tx.execute( + "INSERT INTO answers VALUES(?1,?2,1)", + params![[1u8; 20].as_slice(), [2u8; 20].as_slice()], + ) + .map_err(MetadataError::from)?; + Ok(()) + }) + }) + .await?; + let mut removed = 0; + loop { + let n = walker.call(Scratch::clear_page).await?; + assert!(n <= PAGE_OBJECTS); + removed += n; + if n == 0 { + break; + } + } + assert_eq!(removed, 100_001); + walker + .parents(vec![TypedEdge { + child: ObjectId::Sha1([3; 20]), + expected_kind: ObjectKind::Commit, + }]) + .await?; + walker + .call(|scratch| { + assert_eq!( + scratch + .connection + .query_row("SELECT count(*) FROM visits", [], |r| r.get::<_, usize>(0)) + .map_err(MetadataError::from)?, + 1 + ); + assert_eq!( + scratch + .connection + .query_row("SELECT count(*) FROM answers", [], |r| r.get::<_, usize>(0)) + .map_err(MetadataError::from)?, + 1 + ); + Ok(()) + }) + .await?; + assert!(budget.used() > growth::INITIAL_BYTES * 3); + assert!(budget.used() <= 64 << 20); + drop(walker); + assert_eq!(budget.used(), 0); + assert_eq!(std::fs::read_dir(root)?.count(), 0); + Ok(()) +} +#[tokio::test] +async fn cancellation_retains_a_running_scratch_worker_until_it_drains() -> Result { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(32 << 20); + let walker = Arc::new( + Walker::new( + root.path(), + budget.clone(), + MetadataLimits { + max_file_bytes: 4 << 20, + cache_kib: 64, + }, + ) + .await?, + ); + let (entered_tx, entered_rx) = tokio::sync::oneshot::channel(); + let (release_tx, release_rx) = std::sync::mpsc::channel(); + let work = Arc::clone(&walker); + let task = tokio::spawn(async move { + work.call(move |_| { + let _ = entered_tx.send(()); + release_rx.recv().map_err(|_| RefProofError::Invalid)?; + Ok(()) + }) + .await + }); + entered_rx.await?; + task.abort(); + assert!(task.await.is_err()); + drop(walker); + assert_eq!(budget.used(), growth::INITIAL_BYTES * 3); + assert_eq!(std::fs::read_dir(root.path())?.count(), 1); + release_tx.send(())?; + tokio::time::timeout(std::time::Duration::from_secs(5), async { + while budget.used() != 0 { + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + }) + .await?; + assert_eq!(std::fs::read_dir(root.path())?.count(), 0); + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/ref_proof/tests.rs b/crates/canopy-server/src/packs/publication/ref_proof/tests.rs new file mode 100644 index 0000000..3405022 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/ref_proof/tests.rs @@ -0,0 +1,100 @@ +use super::*; +use crate::RefExpectation; + +#[test] +fn ancestry_queries_bound_exact_wire_bytes_and_row_count() -> Result<(), CodecError> { + for long in [false, true] { + let updates: Vec<_> = (0..256) + .map(|i| crate::RefUpdate { + name: if long { + format!("refs/tags/{}-{i:03}", "x".repeat(65_520)) + } else { + format!("refs/tags/{i:03}") + }, + expected: None, + new_oid: Some(ObjectId::Sha256([7; 32])), + }) + .collect(); + if long { + let original = SqlBatch { + statements: updates[..128] + .iter() + .map(ancestry_policy_statement) + .collect(), + }; + assert!( + original + .encode(&mut BoundedEncoder::new(crate::operation(2).input_limit)?) + .is_err() + ); + } + let mut start = 0; + while start < updates.len() { + let (end, page) = ancestry_policy_page(&updates, start)?; + assert_eq!( + end - start, + if long { + 3.min(updates.len() - start) + } else { + 128 + } + ); + let mut e = BoundedEncoder::new(POLICY_QUERY_BYTES)?; + page.encode(&mut e)?; + let bytes = e.finish(); + assert!(bytes.len() <= POLICY_QUERY_BYTES as usize); + let mut d = BoundedDecoder::new(&bytes, POLICY_QUERY_BYTES)?; + assert_eq!(SqlBatch::decode(&mut d)?, page); + d.finish()?; + start = end; + } + } + Ok(()) +} +#[test] +fn ref_plan_digest_reuses_exact_wire_bytes_including_long_names_and_tombstones() +-> Result<(), CodecError> { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let oid = if format == ObjectFormat::Sha1 { + ObjectId::Sha1([7; 20]) + } else { + ObjectId::Sha256([7; 32]) + }; + let plan = PushPlan { + actor: "a".repeat(64), + updates: vec![ + crate::RefUpdate { + name: format!("refs/heads/{}/topic", "nested/".repeat(100_000)), + expected: None, + new_oid: Some(oid), + }, + crate::RefUpdate { + name: "refs/tags/é".into(), + expected: Some(RefExpectation { + oid: Some(oid), + version: 17, + }), + new_oid: None, + }, + crate::RefUpdate { + name: "refs/heads/recreated".into(), + expected: Some(RefExpectation { + oid: None, + version: 91, + }), + new_oid: Some(oid), + }, + ], + }; + let mut e = BoundedEncoder::new(4 << 20)?; + plan.encode(&mut e)?; + let mut expected = blake3::Hasher::new(); + expected.update(b"canopy.ref-plan.v1\0"); + expected.update(&e.finish()); + assert_eq!(plan_digest(&plan)?, *expected.finalize().as_bytes()); + assert_ne!(binding(&plan, &[0])?, binding(&plan, &[7])?); + assert!(binding(&plan, &[0x80]).is_err()); + assert!(binding(&plan, &[]).is_err()); + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/ref_snapshot.rs b/crates/canopy-server/src/packs/publication/ref_snapshot.rs new file mode 100644 index 0000000..7dd3bf3 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/ref_snapshot.rs @@ -0,0 +1,95 @@ +//! Query-derived joint catalog/ref preparation. Public root descriptors and +//! conditional tree transitions cannot construct this private prepared value. +use super::*; +use crate::{ + PushPlan, + packs::ref_state::{RefSnapshotError, RefStateError, RefStateIndex, RefStateSnapshot}, +}; +use std::sync::Arc; +use tokio::time::timeout_at; + +#[derive(Debug, thiserror::Error)] +pub enum RefSnapshotPreparationError { + #[error("ref snapshot preparation lease is inactive")] + Base(#[from] PreparationBaseError), + #[error("ref snapshot metadata failed")] + Snapshot(#[from] RefSnapshotError), + #[error("ref snapshot transition failed")] + State(#[from] RefStateError), + #[error("selected generation has no immutable ref snapshot")] + Unavailable, + #[error("selected ref snapshot context differs")] + Context, +} + +/// Conditional output bound to the catalog's query-derived retained generation. +/// Membership, policy/check facts and final authority still need to be bound by +/// the final certificate factory; this value itself grants no write authority. +pub struct PreparedRefSnapshot { + base: GenerationFact, + snapshot: RefStateSnapshotRoot, + plan_digest: [u8; 32], +} +impl PreparedRefSnapshot { + pub fn base(&self) -> GenerationFact { + self.base + } + pub fn snapshot(&self) -> RefStateSnapshotRoot { + self.snapshot + } + pub fn plan_digest(&self) -> [u8; 32] { + self.plan_digest + } +} +impl PreparedCatalog { + /// Read exactly this prepared catalog's selected ref metadata from its own + /// trusted store capability. No caller-selected root/store/transition is + /// accepted and no empty fallback can hide an unconverted SQL ref state. + pub async fn prepare_ref_snapshot( + &self, + plan: &PushPlan, + ) -> Result { + let (_, deadline) = self.base.live_lease()?; + timeout_at(deadline, async { + if plan.actor != self.base.capability().2.actor { + return Err(RefSnapshotPreparationError::Context); + } + let base = self.base(); + let selected = base.refs.ok_or(RefSnapshotPreparationError::Unavailable)?; + let store = self.base.indexes().store(); + let old = selected.read(&store).await?; + if old.repository != self.token().repository + || old.format != self.catalog().format + || old.generation > base.generation + || old.generation >= i64::MAX as u64 + { + return Err(RefSnapshotPreparationError::Context); + } + let index = RefStateIndex::new(Arc::clone(&store), old.format); + let transition = index + .prepare(old.root, self.token().artifact_operation, plan) + .await?; + self.ensure_live()?; + let snapshot = RefStateSnapshotRoot::upload( + &store, + self.token().artifact_operation, + RefStateSnapshot { + repository: old.repository, + format: old.format, + generation: old.generation + 1, + default_branch: old.default_branch, + root: Some(transition.root()), + }, + ) + .await?; + self.ensure_live()?; + Ok(PreparedRefSnapshot { + base, + snapshot, + plan_digest: transition.plan_digest(), + }) + }) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } +} diff --git a/crates/canopy-server/src/packs/publication/root_completion/codec.rs b/crates/canopy-server/src/packs/publication/root_completion/codec.rs new file mode 100644 index 0000000..57e4e5b --- /dev/null +++ b/crates/canopy-server/src/packs/publication/root_completion/codec.rs @@ -0,0 +1,89 @@ +use super::*; +use crate::packs::directory::index::codec::fixed; +const DOMAIN: &[u8] = b"canopy.root-push-completion.v1\0"; + +impl WireValue for NativeOutcomeRoot { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.0.validate(INPUT_ROOT_BYTES)?; + self.0.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let root = StoredInputRoot::decode(d)?; + root.validate(INPUT_ROOT_BYTES)?; + Ok(Self(root)) + } +} +impl WireValue for RootPushOutcomes { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + crate::validate_repository_id(self.response_id) + .map_err(|_| CodecError::Invalid("invalid root response ID"))?; + if self.ref_generation == 0 || self.ref_generation > i64::MAX as u64 { + return Err(CodecError::Invalid("invalid completion ref generation")); + } + e.write_bytes(&self.response_id)?; + e.write_u64(self.ref_generation)?; + self.native.encode(e)?; + self.rejected.encode(e)?; + self.replayed.encode(e)?; + e.write_bool(self.signed.is_some())?; + if let Some(signed) = &self.signed { + if signed.size == 0 + || signed.size > crate::push::MAX_RESPONSE_BYTES as u64 + || signed.key.is_empty() + || signed.key.len() > 4096 + || signed.key.chars().any(char::is_control) + { + return Err(CodecError::Invalid("invalid root signed facts")); + } + e.write_bytes(&signed.digest)?; + e.write_text(&signed.key)?; + e.write_u64(signed.size)?; + } + Ok(()) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let response_id = fixed(d)?; + let ref_generation = d.read_u64()?; + let native = NativeOutcomeRoot::decode(d)?; + let rejected = NativeOutcomeRoot::decode(d)?; + let replayed = NativeOutcomeRoot::decode(d)?; + let signed = if d.read_bool()? { + Some(RootSignedPushFact { + digest: fixed(d)?, + key: d.read_text()?.into(), + size: d.read_u64()?, + }) + } else { + None + }; + let value = Self { + response_id, + ref_generation, + native, + rejected, + replayed, + signed, + }; + value.encode(&mut BoundedEncoder::new(ROOT_COMPLETION_BYTES)?)?; + Ok(value) + } +} +impl WireValue for RootPushCompletion { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.shape()?; + e.write_bytes(DOMAIN)?; + self.proof.encode(e)?; + self.outcomes.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + if d.read_bytes()? != DOMAIN { + return Err(CodecError::Invalid("root completion domain")); + } + let value = Self { + proof: RefRootPublicationProof::decode(d)?, + outcomes: RootPushOutcomes::decode(d)?, + }; + value.shape()?; + Ok(value) + } +} diff --git a/crates/canopy-server/src/packs/publication/root_completion/mod.rs b/crates/canopy-server/src/packs/publication/root_completion/mod.rs new file mode 100644 index 0000000..13dd4ec --- /dev/null +++ b/crates/canopy-server/src/packs/publication/root_completion/mod.rs @@ -0,0 +1,130 @@ +//! Immutable native outcomes and bounded joint-root completion preparation. +//! This is conditional input for the final atomic publisher, not an ACK. Raw +//! roots cannot create native custody or authorize reading a completed response. +use super::*; +use crate::packs::{ + input_artifact::{INPUT_ROOT_BYTES, StoredInputRoot}, + metadata::MetadataLimits, +}; +use crate::{directory::DirectoryCell, git_http::GitHttpResponse}; +use canopy_object_storage::artifact::{ArtifactDescriptor, ArtifactStore}; +use cellule_ltx::DiskBudget; +use sha2::{Digest as _, Sha256}; +use std::path::Path; +use tokio::time::timeout_at; + +mod codec; +mod outcome; +mod prepare; +#[cfg(test)] +pub(super) mod tests; + +pub const ROOT_COMPLETION_BYTES: u32 = 8 << 10; +const REPLAYED: &str = "Canopy signed push certificate was already used"; + +/// Reuses the shared metadata-root representation. The body may borrow the +/// native creator namespace; new outcome metadata/failure bodies always belong +/// to the admitted completing attempt. Decoding grants no read authority. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct NativeOutcomeRoot(StoredInputRoot); +impl NativeOutcomeRoot { + pub fn operation(self) -> [u8; 16] { + self.0.operation + } + pub fn artifact(self) -> ArtifactDescriptor { + self.0.artifact + } +} + +/// Indexed signed-certificate ownership facts. The private factory derives +/// these from the registered native result and freshly authorized signing key. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct RootSignedPushFact { + pub digest: [u8; 32], + pub key: String, + pub size: u64, +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct RootPushOutcomes { + pub response_id: [u8; 16], + pub ref_generation: u64, + pub native: NativeOutcomeRoot, + pub rejected: NativeOutcomeRoot, + pub replayed: NativeOutcomeRoot, + pub signed: Option, +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct RootPushCompletion { + pub proof: RefRootPublicationProof, + pub outcomes: RootPushOutcomes, +} + +/// The completed traversal retains the native metadata's plan/options/signed +/// annotation, without making the original wire request a permanent audit root. +/// Active and uncertain preparation pins retain that request independently. +struct OutcomeRecord { + native: NativeResultRoot, + body_operation: [u8; 16], + response: GitHttpResponse, +} + +#[derive(Debug, thiserror::Error)] +pub enum RootCompletionPreparationError { + #[error("root completion preparation is inactive")] + Base(#[from] PreparationBaseError), + #[error("root completion native custody failed")] + Native(#[from] NativeResultError), + #[error("root completion registered input custody failed")] + Inputs(#[from] InputCheckpointError), + #[error("root completion conditional refs failed")] + Policy(#[source] Box), + #[error("root completion immutable metadata failed")] + Root(#[from] crate::packs::InputRootError), + #[error("root completion ref metadata failed")] + Snapshot(#[from] crate::packs::ref_state::RefSnapshotError), + #[error("root completion certificate failed")] + Certificate(#[from] CatalogAttestationError), + #[error("root completion encoding failed")] + Codec(#[from] CodecError), + #[error("root completion native report failed")] + Report(#[from] crate::push::PushError), + #[error("root completion custody, plan or metadata differs")] + Context, +} +impl From for RootCompletionPreparationError { + fn from(error: RefPolicyPreparationError) -> Self { + Self::Policy(Box::new(error)) + } +} + +impl RootPushOutcomes { + fn binding(&self) -> Result<[u8; 32], CodecError> { + let mut e = BoundedEncoder::new(ROOT_COMPLETION_BYTES)?; + self.encode(&mut e)?; + let mut h = blake3::Hasher::new(); + h.update(b"canopy.root-push-completion.v1\0"); + h.update(&e.finish()); + Ok(*h.finalize().as_bytes()) + } +} +impl RootPushCompletion { + fn shape(&self) -> Result<(), CodecError> { + self.proof.shape()?; + let data = self.proof.certificate.data()?; + if data.completion_digest != Some(self.outcomes.binding()?) + || data.input_checkpoint_digest.is_none() + || [ + self.outcomes.native, + self.outcomes.rejected, + self.outcomes.replayed, + ] + .iter() + .any(|root| root.operation() != data.token.artifact_operation) + || data.base.generation == i64::MAX as u64 + || self.outcomes.ref_generation > data.base.generation + 1 + { + return Err(CodecError::Invalid("invalid root completion proof")); + } + Ok(()) + } +} diff --git a/crates/canopy-server/src/packs/publication/root_completion/outcome.rs b/crates/canopy-server/src/packs/publication/root_completion/outcome.rs new file mode 100644 index 0000000..385f3e4 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/root_completion/outcome.rs @@ -0,0 +1,105 @@ +use super::*; +use crate::packs::directory::index::codec::{artifact, fixed, read_artifact}; +const DOMAIN: &[u8] = b"canopy.native-outcome.v1\0"; + +impl NativeOutcomeRoot { + pub(super) async fn upload( + store: &ArtifactStore, + operation: [u8; 16], + record: OutcomeRecord, + ) -> Result { + Ok(Self( + StoredInputRoot::upload(store, operation, &record, INPUT_ROOT_BYTES).await?, + )) + } +} +pub(super) async fn retain_rejection( + store: &ArtifactStore, + operation: [u8; 16], + native: NativeResultRoot, + response: &GitHttpResponse, + reason: &str, +) -> Result { + let GitHttpResponse { + status, + headers, + body, + } = crate::push::report::rejected_report(response, reason)?; + let body = native_result::retain_body(store, operation, body).await?; + NativeOutcomeRoot::upload( + store, + operation, + OutcomeRecord { + native, + body_operation: operation, + response: GitHttpResponse { + status, + headers, + body, + }, + }, + ) + .await +} +impl OutcomeRecord { + fn validate(&self) -> Result<(), CodecError> { + self.native.validate()?; + super::super::codec::artifact_valid(self.body_operation)?; + if !(100..=599).contains(&self.response.status) + || self.response.body.size > crate::push::MAX_RESPONSE_BYTES as u64 + || self.response.body.manifest_digest == [0; 32] + || self.response.headers.iter().any(|(name, value)| { + axum::http::HeaderName::from_bytes(name.as_bytes()).is_err() + || axum::http::HeaderValue::from_str(value).is_err() + || name.eq_ignore_ascii_case("Content-Length") + && value.parse::().ok() != Some(self.response.body.size) + }) + { + return Err(CodecError::Invalid("native outcome metadata")); + } + Ok(()) + } +} +impl WireValue for OutcomeRecord { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.validate()?; + e.write_bytes(DOMAIN)?; + self.native.encode(e)?; + e.write_bytes(&self.body_operation)?; + e.write_u32(self.response.status.into())?; + e.write_count(self.response.headers.len())?; + for (name, value) in &self.response.headers { + e.write_text(name)?; + e.write_text(value)?; + } + artifact(e, self.response.body) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + if d.read_bytes()? != DOMAIN { + return Err(CodecError::Invalid("native outcome domain")); + } + let native = NativeResultRoot::decode(d)?; + let body_operation = fixed(d)?; + let status = u16::try_from(d.read_u32()?) + .map_err(|_| CodecError::Invalid("native outcome status"))?; + let count = d.read_count()?; + if count > INPUT_ROOT_BYTES as usize / 8 { + return Err(CodecError::Limit); + } + let mut headers = Vec::with_capacity(count); + for _ in 0..count { + headers.push((d.read_text()?.into(), d.read_text()?.into())); + } + let value = Self { + native, + body_operation, + response: GitHttpResponse { + status, + headers, + body: read_artifact(d)?, + }, + }; + value.validate()?; + Ok(value) + } +} diff --git a/crates/canopy-server/src/packs/publication/root_completion/prepare.rs b/crates/canopy-server/src/packs/publication/root_completion/prepare.rs new file mode 100644 index 0000000..597fa78 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/root_completion/prepare.rs @@ -0,0 +1,101 @@ +use super::*; + +impl PreparedCatalog { + /// Reopen only registered native custody; certify its exact successful plan + /// and precompute both terminal refusals before entering the final Cell + /// command. This does not publish, select an outcome or acknowledge a push. + pub async fn root_push_completion( + &self, + guard: &PreparedRefPolicyGuard, + directory: &Path, + budget: DiskBudget, + limits: MetadataLimits, + signers: Option<&DirectoryCell>, + ) -> Result { + let (_, deadline) = self.base.live_lease()?; + timeout_at(deadline, async { + let (checkpoint, _, _, _) = self.base.session.push_checkpoint().await?; + let mut checkpoint_bytes = BoundedEncoder::new(CERTIFICATE_BYTES)?; + checkpoint.encode(&mut checkpoint_bytes)?; + let digest = *blake3::hash(&checkpoint_bytes.finish()).as_bytes(); + if self + .input_checkpoint_digest + .is_some_and(|expected| expected != digest) + || self.input_checkpoint_digest.is_none() + && (self.input_count() != 0 || checkpoint.root()?.is_some()) + { + return Err(RootCompletionPreparationError::Context); + } + inputs::verify_digest(&self.base, digest).await?; + let native = checkpoint + .native_result()? + .ok_or(RootCompletionPreparationError::Context)?; + let store = self.base.indexes().store(); + let request = self + .base + .session + .reopen_native_result(&store, directory, &budget, signers) + .await?; + let plan = request + .plan + .ok_or(RootCompletionPreparationError::Context)?; + let mut proof = self + .guarded_ref_snapshot(guard, plan, directory, budget, limits) + .await?; + let ref_generation = proof.snapshot.read(&store).await?.generation; + let record = native.read(&store).await?; + // No public annotation can supply ownership facts: the witness was + // recovered from registered custody and current scoped key lookup. + let signed = request.certificate.map(|signed| RootSignedPushFact { + digest: Sha256::digest(&signed.body).into(), + key: signed.key, + size: signed.body.len() as u64, + }); + let operation = self.token().artifact_operation; + let original = NativeOutcomeRoot::upload( + &store, + operation, + OutcomeRecord { + native, + body_operation: record.operation, + response: record.response, + }, + ) + .await?; + let rejected = outcome::retain_rejection( + &store, + operation, + native, + &request.response, + crate::push::report::REJECTED, + ) + .await?; + let replayed = + outcome::retain_rejection(&store, operation, native, &request.response, REPLAYED) + .await?; + let outcomes = RootPushOutcomes { + response_id: uuid::Uuid::new_v4().into_bytes(), + ref_generation, + native: original, + rejected, + replayed, + signed, + }; + // Freeze may outlast an ACL/check/config change or checkpoint + // update. Recheck both authorities after every artifact is durable. + inputs::verify_digest(&self.base, digest).await?; + ref_policy::ensure_ready(self, proof.guard).await?; + let mut data = certificate::CertificateData::from_prepared(self); + data.input_checkpoint_digest = Some(digest); + data.refs_digest = Some(ref_policy::root_binding(proof.guard, proof.snapshot)?); + data.completion_digest = Some(outcomes.binding()?); + proof.certificate = attestation::issue_data_certificate(&self.base, data).await?; + let value = RootPushCompletion { proof, outcomes }; + value.encode(&mut BoundedEncoder::new(ROOT_COMPLETION_BYTES)?)?; + self.ensure_live()?; + Ok(value) + }) + .await + .map_err(|_| PreparationBaseError::Inactive)? + } +} diff --git a/crates/canopy-server/src/packs/publication/root_completion/tests.rs b/crates/canopy-server/src/packs/publication/root_completion/tests.rs new file mode 100644 index 0000000..3058a15 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/root_completion/tests.rs @@ -0,0 +1,204 @@ +use super::*; +type Result = std::result::Result>; + +pub(in crate::packs::publication) async fn response( + root: NativeOutcomeRoot, + store: &ArtifactStore, +) -> Result { + let record: OutcomeRecord = root.0.read(store, INPUT_ROOT_BYTES).await?; + let mut reader = store + .read( + native_result::body_key(record.body_operation, record.response.body), + record.response.body, + ) + .await?; + let mut body = Vec::new(); + while let Some(part) = reader.next().await? { + body.extend_from_slice(&part); + } + Ok(GitHttpResponse { + status: record.response.status, + headers: record.response.headers, + body, + }) +} + +pub(in crate::packs::publication) fn change_namespace( + root: NativeOutcomeRoot, +) -> NativeOutcomeRoot { + NativeOutcomeRoot(StoredInputRoot { + operation: operation(1), + artifact: root.artifact(), + }) +} +pub(in crate::packs::publication) async fn audit_native( + root: NativeOutcomeRoot, + store: &ArtifactStore, +) -> Result { + let record: OutcomeRecord = root.0.read(store, INPUT_ROOT_BYTES).await?; + let mut e = BoundedEncoder::new(128)?; + root.encode(&mut e)?; + let bytes = e.finish(); + let mut d = BoundedDecoder::new(&bytes, 128)?; + assert!(NativeResultRoot::decode(&mut d)?.read(store).await.is_err()); + Ok(record.native) +} + +fn operation(sequence: u64) -> [u8; 16] { + let mut bytes = *b"CANOPY0100000000"; + bytes[8..].copy_from_slice(&sequence.to_be_bytes()); + bytes +} +fn root() -> NativeOutcomeRoot { + NativeOutcomeRoot(StoredInputRoot { + operation: operation(1), + artifact: ArtifactDescriptor { + size: 100, + digest: [1; 32], + manifest_digest: [2; 32], + }, + }) +} +#[test] +fn outcome_bundle_bounds_signed_key_and_all_three_roots_without_payload() -> Result { + let mut value = RootPushOutcomes { + response_id: *uuid::Uuid::new_v4().as_bytes(), + ref_generation: 1, + native: root(), + rejected: root(), + replayed: root(), + signed: Some(RootSignedPushFact { + digest: [3; 32], + key: "k".repeat(4096), + size: crate::push::MAX_RESPONSE_BYTES as u64, + }), + }; + let mut e = BoundedEncoder::new(ROOT_COMPLETION_BYTES)?; + value.encode(&mut e)?; + let bytes = e.finish(); + assert!(bytes.len() < 5120); + let mut d = BoundedDecoder::new(&bytes, ROOT_COMPLETION_BYTES)?; + assert_eq!(RootPushOutcomes::decode(&mut d)?, value); + d.finish()?; + value.signed.as_mut().unwrap().key.push('x'); + assert!( + value + .encode(&mut BoundedEncoder::new(ROOT_COMPLETION_BYTES)?) + .is_err() + ); + value.signed.as_mut().unwrap().key = "invalid\nkey".into(); + assert!( + value + .encode(&mut BoundedEncoder::new(ROOT_COMPLETION_BYTES)?) + .is_err() + ); + value.signed = None; + value.ref_generation = 0; + assert!( + value + .encode(&mut BoundedEncoder::new(ROOT_COMPLETION_BYTES)?) + .is_err() + ); + Ok(()) +} + +#[tokio::test] +async fn frozen_rejections_preserve_large_progress_and_native_failures_and_detect_late_corruption() +-> Result { + use object_store::ObjectStoreExt; + let provider = std::sync::Arc::new(object_store::memory::InMemory::new()); + let store = ArtifactStore::new(provider.clone(), *uuid::Uuid::new_v4().as_bytes()); + let operation = operation(2); + let descriptor = root(); + let mut e = BoundedEncoder::new(128)?; + descriptor.encode(&mut e)?; + let bytes = e.finish(); + let native = NativeResultRoot::decode(&mut BoundedDecoder::new(&bytes, 128)?)?; + assert_ne!(operation, native.operation()); + // This isolated report test intentionally has no request/native artifacts. + // Actual witness creation is tested through the native receive composition. + fn packet(body: &mut Vec, payload: &[u8]) { + body.extend(format!("{:04x}", payload.len() + 4).as_bytes()); + body.extend(payload); + } + let mut report = Vec::new(); + packet(&mut report, b"unpack ok\n"); + packet(&mut report, b"ok refs/heads/main\n"); + packet(&mut report, b"ng refs/heads/failed native failure\n"); + report.extend(b"0000"); + let mut body = Vec::new(); + while body.len() <= canopy_object_storage::external::PART_BYTES { + let mut progress = vec![b'p'; 60_000]; + progress[0] = 2; + packet(&mut body, &progress); + } + packet(&mut body, &[&[1][..], report.as_slice()].concat()); + body.extend(b"0000"); + let response = GitHttpResponse { + status: 200, + headers: vec![ + ("Content-Length".into(), body.len().to_string()), + ("X-Native-Test".into(), "preserved".into()), + ], + body, + }; + let root = outcome::retain_rejection(&store, operation, native, &response, REPLAYED).await?; + let record: OutcomeRecord = root.0.read(&store, INPUT_ROOT_BYTES).await?; + assert_eq!(record.body_operation, operation); + assert_eq!(record.native, native); + let expected = crate::push::report::rejected_report(&response, REPLAYED)?; + assert!( + expected + .body + .windows(b"ng refs/heads/failed native failure".len()) + .any(|part| part == b"ng refs/heads/failed native failure") + ); + assert_eq!(self::response(root, &store).await?, expected); + let path = store.path( + native_result::body_key(operation, record.response.body), + record.response.body.digest, + )?; + provider + .put( + &canopy_object_storage::external::part(&path, 1), + bytes::Bytes::from_static(b"corrupt late part").into(), + ) + .await?; + assert!(self::response(root, &store).await.is_err()); + Ok(()) +} + +#[tokio::test] +async fn frozen_no_report_rejection_is_explicit_http_failure() -> Result { + let store = ArtifactStore::new( + std::sync::Arc::new(object_store::memory::InMemory::new()), + *uuid::Uuid::new_v4().as_bytes(), + ); + let descriptor = root(); + let mut e = BoundedEncoder::new(128)?; + descriptor.encode(&mut e)?; + let bytes = e.finish(); + let native = NativeResultRoot::decode(&mut BoundedDecoder::new(&bytes, 128)?)?; + for body in [vec![], b"0000".to_vec()] { + let original = GitHttpResponse { + status: 200, + headers: vec![], + body, + }; + let root = outcome::retain_rejection( + &store, + operation(2), + native, + &original, + crate::push::report::REJECTED, + ) + .await?; + let frozen = response(root, &store).await?; + assert_eq!(frozen.status, 409); + assert_eq!( + frozen.body, + format!("{}\n", crate::push::report::REJECTED).into_bytes() + ); + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/schema.sql b/crates/canopy-server/src/packs/publication/schema.sql new file mode 100644 index 0000000..83f6e16 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/schema.sql @@ -0,0 +1,471 @@ +-- Fresh canopy-pack-v1 repository schema. No legacy Git bodies or per-object placement. +-- Product OID membership and ancestry decisions require certified catalog facts. +CREATE TABLE refs ( + name TEXT PRIMARY KEY, + oid BLOB CHECK(oid IS NULL OR length(oid) IN (20, 32)), + version INTEGER NOT NULL CHECK(version > 0) +) WITHOUT ROWID; + +CREATE INDEX refs_by_oid ON refs(oid) WHERE oid IS NOT NULL; + +CREATE TABLE ref_generation ( + singleton INTEGER PRIMARY KEY CHECK(singleton = 1), + generation INTEGER NOT NULL CHECK(typeof(generation) = 'integer' AND generation >= 0), + default_branch TEXT NOT NULL, + visibility TEXT NOT NULL DEFAULT 'private' CHECK(visibility IN ('private', 'public')) +) WITHOUT ROWID; +INSERT INTO ref_generation (singleton, generation, default_branch) VALUES (1, 0, 'refs/heads/main'); + +CREATE TABLE lfs_objects ( + sha256 BLOB PRIMARY KEY CHECK(length(sha256) = 32), + size INTEGER NOT NULL CHECK(size >= 0), + digest BLOB NOT NULL CHECK(length(digest) = 32) +) WITHOUT ROWID; + +CREATE TABLE lfs_locks ( + sequence INTEGER PRIMARY KEY AUTOINCREMENT, + id TEXT NOT NULL UNIQUE, + path TEXT NOT NULL UNIQUE, + locked_at TEXT NOT NULL, + owner TEXT NOT NULL +); + +CREATE TABLE repository_identity ( + object_format TEXT NOT NULL CHECK(object_format IN ('sha1', 'sha256')), + singleton INTEGER PRIMARY KEY CHECK(singleton = 1), + repository_id BLOB NOT NULL CHECK(length(repository_id) = 16), + owner TEXT NOT NULL, + push_cert_seed BLOB NOT NULL CHECK(length(push_cert_seed) = 32), + artifact_sequence INTEGER NOT NULL DEFAULT 0 CHECK(typeof(artifact_sequence) = 'integer' AND artifact_sequence >= 0) +) WITHOUT ROWID; +-- One bounded watermark, not an ever-growing namespace tombstone table. +CREATE TRIGGER repository_artifact_sequence_monotonic BEFORE UPDATE OF artifact_sequence ON repository_identity +WHEN NEW.artifact_sequence != OLD.artifact_sequence + 1 +BEGIN SELECT RAISE(ABORT, 'artifact allocation must advance once'); END; +CREATE TRIGGER repository_artifact_sequence_retained BEFORE DELETE ON repository_identity +WHEN OLD.artifact_sequence > 0 +BEGIN SELECT RAISE(ABORT, 'artifact allocation watermark must be retained'); END; +-- SQLite REPLACE can bypass DELETE triggers with recursive_triggers disabled. +CREATE TRIGGER repository_artifact_sequence_not_replaced BEFORE INSERT ON repository_identity +WHEN EXISTS(SELECT 1 FROM repository_identity WHERE singleton=NEW.singleton AND artifact_sequence>0) +BEGIN SELECT RAISE(ABORT, 'artifact allocation watermark cannot be replaced'); END; + +CREATE TABLE repository_members ( + account TEXT PRIMARY KEY, + role TEXT NOT NULL CHECK(role IN ('read', 'write')) +) WITHOUT ROWID; + +CREATE TABLE membership_versions ( + account TEXT PRIMARY KEY, + version INTEGER NOT NULL CHECK(typeof(version) = 'integer' AND version > 0) +) WITHOUT ROWID; + +CREATE TABLE pushes ( + id BLOB PRIMARY KEY CHECK(length(id) = 16), + actor TEXT NOT NULL, + request_digest BLOB NOT NULL CHECK(length(request_digest) = 32), + options TEXT NOT NULL DEFAULT '[]' CHECK(length(CAST(options AS BLOB)) <= 65536), + response_id BLOB CHECK(response_id IS NULL OR length(response_id) = 16), + completion_digest BLOB CHECK(completion_digest IS NULL OR length(completion_digest) = 32), + rejected INTEGER CHECK(rejected IN (0, 1)), + rejection_reason TEXT, + -- Bounded exact logical publication outcome. HTTP completion additionally + -- records its response in this same transaction; no per-object outcome rows. + publication BLOB CHECK(publication IS NULL OR length(publication) BETWEEN 1 AND 128), + publication_plan_digest BLOB CHECK(publication_plan_digest IS NULL OR length(publication_plan_digest) = 32), + CHECK((publication IS NULL) = (publication_plan_digest IS NULL)), + CHECK((response_id IS NULL) = (rejected IS NULL)), + CHECK((response_id IS NULL) = (completion_digest IS NULL)) +) WITHOUT ROWID; +CREATE TRIGGER push_publication_immutable BEFORE UPDATE OF publication,publication_plan_digest ON pushes +WHEN OLD.publication IS NOT NULL AND (NEW.publication IS NOT OLD.publication OR NEW.publication_plan_digest IS NOT OLD.publication_plan_digest) +BEGIN SELECT RAISE(ABORT, 'push publication outcome is immutable'); END; +CREATE TRIGGER push_completion_immutable BEFORE UPDATE OF response_id,completion_digest,rejected,rejection_reason,options ON pushes +WHEN OLD.response_id IS NOT NULL AND (NEW.response_id IS NOT OLD.response_id OR NEW.completion_digest IS NOT OLD.completion_digest OR NEW.rejected IS NOT OLD.rejected OR NEW.rejection_reason IS NOT OLD.rejection_reason OR NEW.options IS NOT OLD.options) +BEGIN SELECT RAISE(ABORT, 'push completion is immutable'); END; +CREATE TRIGGER push_identity_immutable BEFORE UPDATE OF id,actor,request_digest ON pushes +WHEN NEW.id IS NOT OLD.id OR NEW.actor IS NOT OLD.actor OR NEW.request_digest IS NOT OLD.request_digest +BEGIN SELECT RAISE(ABORT, 'push identity is immutable'); END; +CREATE TRIGGER push_identity_not_replaced BEFORE INSERT ON pushes +WHEN EXISTS(SELECT 1 FROM pushes WHERE id=NEW.id) +BEGIN SELECT RAISE(ABORT, 'push identity cannot be replaced'); END; + +CREATE TABLE push_certificates ( + digest BLOB PRIMARY KEY CHECK(length(digest) = 32), + push_id BLOB NOT NULL UNIQUE REFERENCES pushes(id), + actor TEXT NOT NULL, + signer TEXT NOT NULL, + key TEXT NOT NULL, + size INTEGER NOT NULL CHECK(size > 0), + recorded_at_ms INTEGER NOT NULL CHECK(recorded_at_ms >= 0) +) WITHOUT ROWID; + +CREATE TABLE push_certificate_chunks ( + push_id BLOB NOT NULL REFERENCES pushes(id), + part INTEGER NOT NULL CHECK(part >= 0), + body BLOB NOT NULL CHECK(length(body) BETWEEN 1 AND 524288), + PRIMARY KEY(push_id, part) +) WITHOUT ROWID; + +CREATE TABLE push_responses ( + id BLOB PRIMARY KEY CHECK(length(id) = 16), + push_id BLOB NOT NULL REFERENCES pushes(id), + status INTEGER NOT NULL CHECK(status BETWEEN 100 AND 599), + headers TEXT NOT NULL, + size INTEGER NOT NULL CHECK(size BETWEEN 0 AND 67108864), + digest BLOB NOT NULL CHECK(length(digest) = 32) +) WITHOUT ROWID; + +CREATE TABLE push_response_chunks ( + response_id BLOB NOT NULL REFERENCES push_responses(id), + part INTEGER NOT NULL CHECK(part >= 0), + body BLOB NOT NULL CHECK(length(body) BETWEEN 1 AND 524288), + PRIMARY KEY(response_id, part) +) WITHOUT ROWID; + +CREATE TABLE push_plan_chunks ( + response_id BLOB NOT NULL REFERENCES push_responses(id), + part INTEGER NOT NULL CHECK(part >= 0), + body BLOB NOT NULL CHECK(length(body) BETWEEN 1 AND 65536), + PRIMARY KEY(response_id, part) +) WITHOUT ROWID; + +CREATE TABLE issues ( + number INTEGER PRIMARY KEY AUTOINCREMENT, + id BLOB NOT NULL UNIQUE CHECK(length(id) = 16), + creation_digest BLOB NOT NULL CHECK(length(creation_digest) = 32), + author TEXT NOT NULL, + title TEXT NOT NULL CHECK(length(CAST(title AS BLOB)) BETWEEN 1 AND 256), + body TEXT NOT NULL CHECK(length(CAST(body AS BLOB)) <= 16384), + state TEXT NOT NULL CHECK(state IN ('open', 'closed')), + version INTEGER NOT NULL CHECK(typeof(version) = 'integer' AND version > 0), + created_ms INTEGER NOT NULL CHECK(created_ms >= 0), + updated_ms INTEGER NOT NULL CHECK(updated_ms >= created_ms) +); +CREATE INDEX issues_by_state ON issues(state, number); + +CREATE TABLE issue_comments ( + number INTEGER PRIMARY KEY AUTOINCREMENT, + issue_number INTEGER NOT NULL REFERENCES issues(number), + id BLOB NOT NULL UNIQUE CHECK(length(id) = 16), + creation_digest BLOB NOT NULL CHECK(length(creation_digest) = 32), + author TEXT NOT NULL, + body TEXT NOT NULL CHECK(length(CAST(body AS BLOB)) BETWEEN 1 AND 16384), + version INTEGER NOT NULL CHECK(typeof(version) = 'integer' AND version > 0), + created_ms INTEGER NOT NULL CHECK(created_ms >= 0), + updated_ms INTEGER NOT NULL CHECK(updated_ms >= created_ms) +); +CREATE INDEX comments_by_issue ON issue_comments(issue_number, number); + +CREATE TABLE check_contexts ( + name TEXT PRIMARY KEY, + reporter TEXT NOT NULL, + enabled INTEGER NOT NULL CHECK(enabled IN (0, 1)), + version INTEGER NOT NULL CHECK(typeof(version) = 'integer' AND version > 0) +) WITHOUT ROWID; + +CREATE INDEX check_contexts_by_enabled ON check_contexts(enabled, name); + +CREATE TABLE check_runs ( + number INTEGER PRIMARY KEY AUTOINCREMENT, + id BLOB NOT NULL UNIQUE CHECK(length(id) = 16), + oid BLOB NOT NULL CHECK(length(oid) IN (20, 32)), + context TEXT NOT NULL REFERENCES check_contexts(name), + context_version INTEGER NOT NULL CHECK(context_version > 0), + reporter TEXT NOT NULL, + state TEXT NOT NULL CHECK(state IN ('queued', 'in_progress', 'success', 'failure', 'cancelled')), + version INTEGER NOT NULL CHECK(typeof(version) = 'integer' AND version > 0), + summary TEXT NOT NULL CHECK(length(CAST(summary AS BLOB)) <= 4096), + created_ms INTEGER NOT NULL CHECK(created_ms >= 0), + updated_ms INTEGER NOT NULL CHECK(updated_ms >= created_ms) +); +CREATE INDEX check_runs_by_commit ON check_runs(oid, context, context_version, number); + + + + + +CREATE TABLE branch_rules ( + reference TEXT PRIMARY KEY, + version INTEGER NOT NULL CHECK(typeof(version) = 'integer' AND version > 0), + enabled INTEGER NOT NULL CHECK(enabled IN (0, 1)), + deny_deletions INTEGER NOT NULL CHECK(deny_deletions IN (0, 1)), + fast_forward INTEGER NOT NULL CHECK(fast_forward IN (0, 1)), + require_pull_request INTEGER NOT NULL CHECK(require_pull_request IN (0, 1)), + required_approvals INTEGER NOT NULL CHECK(required_approvals BETWEEN 0 AND 16 AND (require_pull_request = 1 OR required_approvals = 0)) +) WITHOUT ROWID; +CREATE TABLE branch_required_checks ( + reference TEXT NOT NULL REFERENCES branch_rules(reference), + context TEXT NOT NULL, + PRIMARY KEY(reference, context) +) WITHOUT ROWID; +CREATE INDEX branch_rules_by_enabled ON branch_rules(enabled, reference); + +CREATE TABLE pull_requests ( + number INTEGER PRIMARY KEY AUTOINCREMENT, + id BLOB NOT NULL UNIQUE CHECK(length(id) = 16), + creation_digest BLOB NOT NULL CHECK(length(creation_digest) = 32), + author TEXT NOT NULL, + title TEXT NOT NULL CHECK(length(CAST(title AS BLOB)) BETWEEN 1 AND 256), + body TEXT NOT NULL CHECK(length(CAST(body AS BLOB)) <= 16384), + state TEXT NOT NULL CHECK(state IN ('open', 'closed', 'merged')), + draft INTEGER NOT NULL CHECK(draft IN (0, 1)), + version INTEGER NOT NULL CHECK(typeof(version) = 'integer' AND version > 0), + source_ref TEXT NOT NULL REFERENCES refs(name), + base_ref TEXT NOT NULL REFERENCES refs(name) CHECK(source_ref != base_ref), + initial_source_oid BLOB NOT NULL CHECK(length(initial_source_oid) IN (20, 32)), + initial_base_oid BLOB NOT NULL CHECK(length(initial_base_oid) IN (20, 32)), + created_ms INTEGER NOT NULL CHECK(created_ms >= 0), + updated_ms INTEGER NOT NULL CHECK(updated_ms >= created_ms) +); +CREATE INDEX pulls_by_state ON pull_requests(state, number); + +CREATE TABLE pull_reviews ( + number INTEGER PRIMARY KEY AUTOINCREMENT, + id BLOB NOT NULL UNIQUE CHECK(length(id) = 16), + creation_digest BLOB NOT NULL CHECK(length(creation_digest) = 32), + pull_number INTEGER NOT NULL REFERENCES pull_requests(number), + reviewer TEXT NOT NULL, + membership_version INTEGER NOT NULL CHECK(membership_version >= 0), + kind TEXT NOT NULL CHECK(kind IN ('comment', 'approve', 'request_changes')), + body TEXT NOT NULL CHECK(length(CAST(body AS BLOB)) <= 16384), + pull_version INTEGER NOT NULL CHECK(pull_version > 0), + source_oid BLOB NOT NULL CHECK(length(source_oid) IN (20, 32)), + source_version INTEGER NOT NULL CHECK(source_version > 0), + base_oid BLOB NOT NULL CHECK(length(base_oid) IN (20, 32)), + base_version INTEGER NOT NULL CHECK(base_version > 0), + created_ms INTEGER NOT NULL CHECK(created_ms >= 0) +); +CREATE INDEX reviews_by_pull ON pull_reviews(pull_number, number); + +CREATE TABLE pull_review_heads ( + pull_number INTEGER NOT NULL REFERENCES pull_requests(number), + reviewer TEXT NOT NULL, + review_number INTEGER NOT NULL REFERENCES pull_reviews(number), + PRIMARY KEY(pull_number, reviewer) +) WITHOUT ROWID; + +CREATE TABLE pull_merges ( + id BLOB PRIMARY KEY CHECK(length(id) = 16), + binding BLOB NOT NULL CHECK(length(binding) = 32), + pull_number INTEGER NOT NULL UNIQUE REFERENCES pull_requests(number), + oid BLOB NOT NULL CHECK(length(oid) IN (20, 32)), + merged_ms INTEGER NOT NULL CHECK(merged_ms >= 0), + pull_version INTEGER NOT NULL CHECK(pull_version > 0), + source_oid BLOB NOT NULL CHECK(length(source_oid) IN (20, 32)), + source_version INTEGER NOT NULL CHECK(source_version > 0), + base_oid BLOB NOT NULL CHECK(length(base_oid) IN (20, 32)), + base_version INTEGER NOT NULL CHECK(base_version > 0) +) WITHOUT ROWID; + +CREATE TABLE merge_candidates ( + id BLOB PRIMARY KEY CHECK(length(id) = 16), + binding BLOB NOT NULL CHECK(length(binding) = 32), + pull_number INTEGER NOT NULL REFERENCES pull_requests(number), + actor TEXT NOT NULL, + request TEXT NOT NULL CHECK(length(CAST(request AS BLOB)) <= 131072), + created_ms INTEGER NOT NULL CHECK(created_ms >= 0), + result TEXT NOT NULL CHECK(length(CAST(result AS BLOB)) <= 262144), + source_oid BLOB NOT NULL CHECK(length(source_oid) IN (20, 32)), + base_oid BLOB NOT NULL CHECK(length(base_oid) IN (20, 32)), + oid BLOB CHECK(oid IS NULL OR length(oid) IN (20, 32)) +) WITHOUT ROWID; + +CREATE TABLE pull_threads ( + number INTEGER PRIMARY KEY AUTOINCREMENT, + id BLOB NOT NULL UNIQUE CHECK(length(id) = 16), + creation_digest BLOB NOT NULL CHECK(length(creation_digest) = 32), + pull_number INTEGER NOT NULL REFERENCES pull_requests(number), + author TEXT NOT NULL, + body TEXT NOT NULL CHECK(length(CAST(body AS BLOB)) BETWEEN 1 AND 16384), + resolved INTEGER NOT NULL CHECK(resolved IN (0, 1)), + version INTEGER NOT NULL CHECK(typeof(version) = 'integer' AND version > 0), + pull_version INTEGER NOT NULL CHECK(pull_version > 0), + source_oid BLOB NOT NULL CHECK(length(source_oid) IN (20, 32)), + source_version INTEGER NOT NULL CHECK(source_version > 0), + base_oid BLOB NOT NULL CHECK(length(base_oid) IN (20, 32)), + base_version INTEGER NOT NULL CHECK(base_version > 0), + merge_base BLOB NOT NULL CHECK(length(merge_base) IN (20, 32)), + path BLOB NOT NULL CHECK(length(path) BETWEEN 1 AND 4096), + side TEXT NOT NULL CHECK(side IN ('before', 'after')), + line INTEGER NOT NULL CHECK(line BETWEEN 1 AND 20000), + blob_oid BLOB NOT NULL CHECK(length(blob_oid) IN (20, 32)), + created_ms INTEGER NOT NULL CHECK(created_ms >= 0), + updated_ms INTEGER NOT NULL CHECK(updated_ms >= created_ms) +); +CREATE INDEX threads_by_pull ON pull_threads(pull_number, number); + +CREATE TABLE pull_thread_comments ( + number INTEGER PRIMARY KEY AUTOINCREMENT, + id BLOB NOT NULL UNIQUE CHECK(length(id) = 16), + creation_digest BLOB NOT NULL CHECK(length(creation_digest) = 32), + thread_number INTEGER NOT NULL REFERENCES pull_threads(number), + author TEXT NOT NULL, + body TEXT NOT NULL CHECK(length(CAST(body AS BLOB)) BETWEEN 1 AND 16384), + created_ms INTEGER NOT NULL CHECK(created_ms >= 0) +); +CREATE INDEX comments_by_thread ON pull_thread_comments(thread_number, number); + + +CREATE TABLE catalog_generations ( + generation INTEGER PRIMARY KEY CHECK(typeof(generation) = 'integer' AND generation >= 0), + catalog BLOB, + certificate BLOB, + refs BLOB CHECK(refs IS NULL OR (typeof(refs) = 'blob' AND length(refs) BETWEEN 1 AND 128)), + CHECK((generation = 0 AND catalog IS NULL AND certificate IS NULL AND refs IS NULL) + OR (generation > 0 AND catalog IS NOT NULL AND certificate IS NOT NULL AND length(catalog) BETWEEN 1 AND 256 AND length(certificate) = 32)) +) WITHOUT ROWID; +INSERT INTO catalog_generations VALUES(0, NULL, NULL, NULL); +CREATE TRIGGER catalog_generations_immutable BEFORE UPDATE ON catalog_generations +BEGIN SELECT RAISE(ABORT, 'catalog generations are immutable'); END; +CREATE TRIGGER catalog_generations_not_replaced BEFORE INSERT ON catalog_generations +WHEN EXISTS(SELECT 1 FROM catalog_generations WHERE generation=NEW.generation) +BEGIN SELECT RAISE(ABORT, 'catalog generations cannot be replaced'); END; +CREATE TRIGGER catalog_generations_bounded BEFORE INSERT ON catalog_generations +WHEN (SELECT count(*) FROM (SELECT generation FROM catalog_generations LIMIT 8192)) >= 8192 +BEGIN SELECT RAISE(ABORT, 'catalog generation quota exceeded'); END; +CREATE TABLE catalog_state ( + singleton INTEGER PRIMARY KEY CHECK(singleton = 1), + generation INTEGER NOT NULL REFERENCES catalog_generations(generation) +) WITHOUT ROWID; +INSERT INTO catalog_state VALUES(1, 0); + +-- One immutable initialization outcome per repository. Its shared generation +-- fact retains the small empty catalog/ref metadata for exact logical recovery. +CREATE TABLE catalog_initialization ( + singleton INTEGER PRIMARY KEY CHECK(singleton=1), + id BLOB NOT NULL UNIQUE CHECK(length(id)=16), + actor TEXT NOT NULL, + request_digest BLOB NOT NULL CHECK(length(request_digest)=32), + verification_digest BLOB NOT NULL CHECK(length(verification_digest)=32), + result BLOB NOT NULL CHECK(length(result) BETWEEN 1 AND 512) +) WITHOUT ROWID; +CREATE TRIGGER catalog_initialization_immutable BEFORE UPDATE ON catalog_initialization +BEGIN SELECT RAISE(ABORT, 'initialization outcome is immutable'); END; +CREATE TRIGGER catalog_initialization_not_replaced BEFORE INSERT ON catalog_initialization +WHEN EXISTS(SELECT 1 FROM catalog_initialization WHERE singleton=NEW.singleton) +BEGIN SELECT RAISE(ABORT, 'initialization outcome cannot be replaced'); END; +CREATE TRIGGER catalog_initialization_retained BEFORE DELETE ON catalog_initialization +BEGIN SELECT RAISE(ABORT, 'initialization outcome must be retained'); END; + +-- Exact logical outcomes for catalog-only maintenance. Kept separately from +-- network push response/product facts; no per-object outcome rows are created. +CREATE TABLE catalog_compactions ( + id BLOB PRIMARY KEY CHECK(length(id)=16), + actor TEXT NOT NULL, + request_digest BLOB NOT NULL CHECK(length(request_digest)=32), + verification_digest BLOB NOT NULL CHECK(length(verification_digest)=32), + result BLOB NOT NULL CHECK(length(result) BETWEEN 1 AND 128) +) WITHOUT ROWID; +CREATE TRIGGER catalog_compactions_immutable BEFORE UPDATE ON catalog_compactions +BEGIN SELECT RAISE(ABORT, 'compaction outcomes are immutable'); END; +CREATE TRIGGER catalog_compactions_not_replaced BEFORE INSERT ON catalog_compactions +WHEN EXISTS(SELECT 1 FROM catalog_compactions WHERE id=NEW.id) +BEGIN SELECT RAISE(ABORT, 'compaction outcomes cannot be replaced'); END; + +-- Staging attempts retain their creating namespace with a NULL generation. +-- A one-way late bind pins a generation floor and every later generation. This permits +-- read-only frontier refresh without a new durable pin/Claim per publication. +-- Replacement/abort preserves the old floor until its independent pin is reaped. +CREATE TABLE catalog_leases ( + incarnation BLOB NOT NULL CHECK(length(incarnation) = 16), + admission_sequence INTEGER NOT NULL CHECK(typeof(admission_sequence) = 'integer' AND admission_sequence > 0), + operation BLOB NOT NULL CHECK(length(operation) = 16), + owner_epoch BLOB NOT NULL CHECK(length(owner_epoch) = 8 AND owner_epoch != zeroblob(8)), + artifact_operation BLOB NOT NULL CHECK(length(artifact_operation) = 16), + generation INTEGER REFERENCES catalog_generations(generation), + -- Normalize the unbound staging phase for the deferred exact binding. A + -- nullable FK alone would skip validation of every other identity field. + binding_generation INTEGER GENERATED ALWAYS AS (coalesce(generation, -1)) STORED, + expires_at_ms INTEGER NOT NULL CHECK(typeof(expires_at_ms) = 'integer' AND expires_at_ms >= 0), + attestation BLOB CHECK(attestation IS NULL OR length(attestation) BETWEEN 1 AND 1024), + attestation_digest BLOB CHECK(attestation_digest IS NULL OR length(attestation_digest) = 32), + input_checkpoint BLOB CHECK(input_checkpoint IS NULL OR length(input_checkpoint) BETWEEN 1 AND 1024), + input_checkpoint_digest BLOB CHECK(input_checkpoint_digest IS NULL OR length(input_checkpoint_digest) = 32), + input_checkpoint_previous_digest BLOB CHECK(input_checkpoint_previous_digest IS NULL OR length(input_checkpoint_previous_digest) = 32), + input_checkpoint_revision INTEGER NOT NULL DEFAULT 0 CHECK(typeof(input_checkpoint_revision)='integer' AND input_checkpoint_revision BETWEEN 0 AND 256), + CHECK((input_checkpoint IS NULL) = (input_checkpoint_digest IS NULL)), + CHECK(input_checkpoint IS NOT NULL OR (input_checkpoint_revision=0 AND input_checkpoint_previous_digest IS NULL)), + CHECK((input_checkpoint_revision=0) = (input_checkpoint_previous_digest IS NULL)), + CHECK((attestation IS NULL) = (attestation_digest IS NULL)), + CHECK(generation IS NOT NULL OR attestation IS NULL), + PRIMARY KEY(incarnation, admission_sequence) +) WITHOUT ROWID; +CREATE INDEX catalog_leases_by_expiry ON catalog_leases(expires_at_ms, incarnation, admission_sequence); +CREATE INDEX catalog_leases_by_generation ON catalog_leases(generation, expires_at_ms); +CREATE TRIGGER catalog_generations_retained BEFORE DELETE ON catalog_generations +WHEN OLD.generation=0 OR OLD.generation >= (SELECT min(generation) FROM catalog_leases) +BEGIN SELECT RAISE(ABORT, 'catalog generation is retained'); END; +CREATE UNIQUE INDEX catalog_leases_by_artifact ON catalog_leases(artifact_operation); +CREATE TRIGGER catalog_lease_not_replaced BEFORE INSERT ON catalog_leases +WHEN EXISTS(SELECT 1 FROM catalog_leases WHERE incarnation=NEW.incarnation AND admission_sequence=NEW.admission_sequence) + OR EXISTS(SELECT 1 FROM catalog_leases WHERE artifact_operation=NEW.artifact_operation) +BEGIN SELECT RAISE(ABORT, 'catalog attempt pin cannot be replaced'); END; +CREATE UNIQUE INDEX catalog_leases_binding ON catalog_leases(incarnation, admission_sequence, operation, owner_epoch, artifact_operation, binding_generation, expires_at_ms); +CREATE TRIGGER catalog_lease_identity_immutable BEFORE UPDATE OF incarnation, admission_sequence, operation, owner_epoch, artifact_operation, generation ON catalog_leases +WHEN NEW.incarnation != OLD.incarnation OR NEW.admission_sequence != OLD.admission_sequence + OR NEW.operation != OLD.operation OR NEW.owner_epoch != OLD.owner_epoch + OR NEW.artifact_operation != OLD.artifact_operation + OR (NEW.generation IS NOT OLD.generation AND NOT + (OLD.generation IS NULL AND NEW.generation IS NOT NULL AND OLD.attestation IS NULL)) +BEGIN SELECT RAISE(ABORT, 'catalog attempt identity is immutable'); END; +CREATE TRIGGER catalog_lease_attestation_immutable BEFORE UPDATE OF attestation, attestation_digest ON catalog_leases +WHEN OLD.attestation IS NOT NULL AND (NEW.attestation IS NOT OLD.attestation OR NEW.attestation_digest IS NOT OLD.attestation_digest) +BEGIN SELECT RAISE(ABORT, 'catalog attempt attestation is immutable'); END; +CREATE TRIGGER catalog_lease_inputs_immutable BEFORE UPDATE OF input_checkpoint, input_checkpoint_digest, input_checkpoint_revision, input_checkpoint_previous_digest ON catalog_leases +WHEN (OLD.input_checkpoint IS NULL AND (NEW.input_checkpoint_revision!=0 OR NEW.input_checkpoint_previous_digest IS NOT NULL)) + OR (OLD.input_checkpoint IS NOT NULL AND NOT ( + (NEW.input_checkpoint IS OLD.input_checkpoint AND NEW.input_checkpoint_digest IS OLD.input_checkpoint_digest AND NEW.input_checkpoint_revision=OLD.input_checkpoint_revision AND NEW.input_checkpoint_previous_digest IS OLD.input_checkpoint_previous_digest) + OR (OLD.generation IS NULL AND NEW.generation IS NULL AND NEW.input_checkpoint IS NOT NULL AND NEW.input_checkpoint_digest IS NOT NULL + AND NEW.input_checkpoint IS NOT OLD.input_checkpoint AND NEW.input_checkpoint_digest IS NOT OLD.input_checkpoint_digest + AND NEW.input_checkpoint_revision=OLD.input_checkpoint_revision+1 AND NEW.input_checkpoint_previous_digest IS OLD.input_checkpoint_digest))) +BEGIN SELECT RAISE(ABORT, 'creating input checkpoint requires exact append'); END; + +CREATE TABLE catalog_operations ( + id BLOB PRIMARY KEY CHECK(length(id) = 16), + actor TEXT NOT NULL CHECK(length(CAST(actor AS BLOB)) BETWEEN 1 AND 64), + request_digest BLOB NOT NULL CHECK(length(request_digest) = 32), + incarnation BLOB NOT NULL CHECK(length(incarnation) = 16), + owner_epoch BLOB NOT NULL CHECK(length(owner_epoch) = 8 AND owner_epoch != zeroblob(8)), + admission_sequence INTEGER NOT NULL CHECK(typeof(admission_sequence) = 'integer' AND admission_sequence > 0), + artifact_operation BLOB NOT NULL CHECK(length(artifact_operation) = 16), + generation INTEGER REFERENCES catalog_generations(generation), + -- Normalize the unbound staging phase for the deferred exact binding. A + -- nullable FK alone would skip validation of every other identity field. + binding_generation INTEGER GENERATED ALWAYS AS (coalesce(generation, -1)) STORED, + expires_at_ms INTEGER NOT NULL CHECK(typeof(expires_at_ms) = 'integer' AND expires_at_ms >= 0), + attestation BLOB CHECK(attestation IS NULL OR length(attestation) BETWEEN 1 AND 1024), + attestation_digest BLOB CHECK(attestation_digest IS NULL OR length(attestation_digest) = 32), + CHECK((attestation IS NULL) = (attestation_digest IS NULL)), + CHECK(generation IS NOT NULL OR attestation IS NULL), + FOREIGN KEY(incarnation, admission_sequence, id, owner_epoch, artifact_operation, binding_generation, expires_at_ms) + REFERENCES catalog_leases(incarnation, admission_sequence, operation, owner_epoch, artifact_operation, binding_generation, expires_at_ms) + DEFERRABLE INITIALLY DEFERRED +) WITHOUT ROWID; +CREATE INDEX catalog_operations_by_expiry ON catalog_operations(expires_at_ms, id); +CREATE UNIQUE INDEX catalog_operations_by_lease ON catalog_operations(incarnation, admission_sequence); +CREATE INDEX catalog_operations_by_generation ON catalog_operations(generation); + +CREATE TRIGGER push_responses_immutable BEFORE UPDATE ON push_responses +BEGIN SELECT RAISE(ABORT, 'push outcome bytes are immutable'); END; +CREATE TRIGGER push_responses_not_replaced BEFORE INSERT ON push_responses +WHEN EXISTS(SELECT 1 FROM push_responses WHERE id=NEW.id) +BEGIN SELECT RAISE(ABORT, 'push outcome bytes cannot be replaced'); END; + +CREATE TRIGGER push_response_chunks_immutable BEFORE UPDATE ON push_response_chunks +BEGIN SELECT RAISE(ABORT, 'push outcome bytes are immutable'); END; +CREATE TRIGGER push_response_chunks_not_replaced BEFORE INSERT ON push_response_chunks +WHEN EXISTS(SELECT 1 FROM push_response_chunks WHERE response_id=NEW.response_id AND part=NEW.part) +BEGIN SELECT RAISE(ABORT, 'push outcome bytes cannot be replaced'); END; + +CREATE TRIGGER push_certificates_immutable BEFORE UPDATE ON push_certificates +BEGIN SELECT RAISE(ABORT, 'push outcome bytes are immutable'); END; +CREATE TRIGGER push_certificates_not_replaced BEFORE INSERT ON push_certificates +WHEN EXISTS(SELECT 1 FROM push_certificates WHERE digest=NEW.digest) +BEGIN SELECT RAISE(ABORT, 'push outcome bytes cannot be replaced'); END; + +CREATE TRIGGER push_certificate_chunks_immutable BEFORE UPDATE ON push_certificate_chunks +BEGIN SELECT RAISE(ABORT, 'push outcome bytes are immutable'); END; +CREATE TRIGGER push_certificate_chunks_not_replaced BEFORE INSERT ON push_certificate_chunks +WHEN EXISTS(SELECT 1 FROM push_certificate_chunks WHERE push_id=NEW.push_id AND part=NEW.part) +BEGIN SELECT RAISE(ABORT, 'push outcome bytes cannot be replaced'); END; diff --git a/crates/canopy-server/src/packs/publication/session.rs b/crates/canopy-server/src/packs/publication/session.rs new file mode 100644 index 0000000..31db788 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/session.rs @@ -0,0 +1,168 @@ +//! Shared authoritative preparation lease; no artifact loads or scratch. +use super::*; +use cellule_runtime::{CellClient, CellTarget, MutationIdentity, Receipt}; +use std::{ + sync::{ + Arc, Mutex, + atomic::{AtomicBool, Ordering}, + }, + time::Duration, +}; +use tokio::time::Instant; + +#[derive(Clone)] +pub struct PreparationSession { + pub(super) client: CellClient, + pub(super) target: CellTarget, + pub(super) check: LeaseCheck, + pub(super) lease: PreparationLease, + pub(super) deadline: Arc>, + pub(super) ceiling: Option, + pub(super) fenced: Arc, +} +impl PreparationSession { + pub async fn open( + client: CellClient, + target: CellTarget, + check: LeaseCheck, + minimum: Option, + ) -> Result { + if crate::repository_target( + target.tenant(), + target.application(), + check.token.repository, + ) + .map_err(|_| PreparationBaseError::Context)? + != target + { + return Err(PreparationBaseError::Context); + } + let (lease, deadline) = probe(&client, &target, &check, minimum).await?; + Ok(Self { + client, + target, + check, + lease, + deadline: Arc::new(Mutex::new(deadline)), + ceiling: None, + fenced: Arc::new(AtomicBool::new(false)), + }) + } + pub(super) fn capability(&self) -> (&CellClient, &CellTarget, &LeaseCheck) { + (&self.client, &self.target, &self.check) + } + pub(super) fn live_lease(&self) -> Result<(PreparationLease, Instant), PreparationBaseError> { + let deadline = *self + .deadline + .lock() + .map_err(|_| PreparationBaseError::Context)?; + let deadline = self.ceiling.map_or(deadline, |limit| deadline.min(limit)); + if self.fenced.load(Ordering::Acquire) || Instant::now() >= deadline { + return Err(PreparationBaseError::Inactive); + } + Ok((self.lease, deadline)) + } + /// A recorded renewal result is never a fresh clock observation. Query + /// after the durability gate even when the command is exact-outcome replay. + pub async fn renew( + &self, + identity: MutationIdentity, + lease_ms: u64, + ) -> Result<(), PreparationBaseError> { + let result = self.renew_inner(identity, lease_ms).await; + if result.is_err() { + self.fenced.store(true, Ordering::Release); + } + result + } + async fn renew_inner( + &self, + identity: MutationIdentity, + lease_ms: u64, + ) -> Result<(), PreparationBaseError> { + if self.fenced.load(Ordering::Acquire) + || self.ceiling.is_some_and(|limit| Instant::now() >= limit) + { + return Err(PreparationBaseError::Inactive); + } + let committed = self + .client + .command::( + &self.target, + identity, + LeaseRequest { + check: self.check.clone(), + lease_ms, + }, + ) + .await + .map_err(|error| PreparationBaseError::Command(Box::new(error)))?; + self.refresh(committed.receipt).await + } + pub(super) fn fence(&self) { + self.fenced.store(true, Ordering::Release); + } + pub(super) async fn refresh(&self, minimum: Receipt) -> Result<(), PreparationBaseError> { + let result = self.refresh_inner(minimum).await; + if result.is_err() { + self.fence(); + } + result + } + async fn refresh_inner(&self, minimum: Receipt) -> Result<(), PreparationBaseError> { + if self.fenced.load(Ordering::Acquire) + || self.ceiling.is_some_and(|limit| Instant::now() >= limit) + { + return Err(PreparationBaseError::Inactive); + } + let (lease, deadline) = + probe(&self.client, &self.target, &self.check, Some(minimum)).await?; + if lease.token != self.lease.token + || lease.base != self.lease.base + || lease.format != self.lease.format + { + return Err(PreparationBaseError::Context); + } + let mut shared = self + .deadline + .lock() + .map_err(|_| PreparationBaseError::Context)?; + if self.fenced.load(Ordering::Acquire) + || self.ceiling.is_some_and(|limit| Instant::now() >= limit) + { + return Err(PreparationBaseError::Inactive); + } + *shared = deadline; + Ok(()) + } +} +async fn probe( + client: &CellClient, + target: &CellTarget, + check: &LeaseCheck, + minimum: Option, +) -> Result<(PreparationLease, Instant), PreparationBaseError> { + // Start before the query, not after its reply, so transport/queue time can + // only shorten the usable lease. Queries do not replay stored commands. + let started = Instant::now(); + let lease = client + .query::(target, minimum, check.clone()) + .await + .map_err(|error| PreparationBaseError::Query(Box::new(error)))? + .output + .ok_or(PreparationBaseError::Inactive)?; + if lease.token != check.token + || lease.observed_at_ms < 0 + || lease.expires_at_ms <= lease.observed_at_ms + { + return Err(PreparationBaseError::Context); + } + let remaining = (lease.expires_at_ms - lease.observed_at_ms) as u64; + let deadline = started + .checked_add(Duration::from_millis(remaining.min(MAX_LEASE_MS))) + .ok_or(PreparationBaseError::Context)?; + if Instant::now() >= deadline { + return Err(PreparationBaseError::Inactive); + } + Ok((lease, deadline)) +} diff --git a/crates/canopy-server/src/packs/publication/sql.rs b/crates/canopy-server/src/packs/publication/sql.rs new file mode 100644 index 0000000..b3ec03a --- /dev/null +++ b/crates/canopy-server/src/packs/publication/sql.rs @@ -0,0 +1,265 @@ +use super::*; +use std::time::{SystemTime, UNIX_EPOCH}; +pub(super) const OPERATION: &str = "SELECT actor,request_digest,incarnation,owner_epoch,admission_sequence,artifact_operation,generation,expires_at_ms FROM catalog_operations WHERE id=?1"; + +pub(super) fn statement(sql: &str, parameters: Vec) -> SqlBatch { + SqlBatch { + statements: vec![SqlStatement { + sql: sql.into(), + parameters, + }], + } +} +pub(super) fn blob(bytes: impl AsRef<[u8]>) -> SqlValue { + SqlValue::Blob(bytes.as_ref().to_vec()) +} +pub(super) fn number(value: u64) -> cellule_runtime::Result { + Ok(SqlValue::Integer( + i64::try_from(value).map_err(|_| Error::Command("catalog integer overflow"))?, + )) +} +pub(super) fn rows(sets: &[SqlResultSet]) -> cellule_runtime::Result<&[Vec]> { + Ok(&sets + .first() + .ok_or(Error::Command("missing catalog SQL result"))? + .rows) +} +pub(super) fn unsigned(value: &SqlValue) -> cellule_runtime::Result { + match value { + SqlValue::Integer(value) if *value >= 0 => Ok(*value as u64), + _ => Err(Error::Command("invalid catalog integer")), + } +} +pub(super) fn optional_generation(value: &SqlValue) -> cellule_runtime::Result> { + if *value == SqlValue::Null { + Ok(None) + } else { + unsigned(value).map(Some) + } +} +pub(super) fn fixed(value: &SqlValue) -> cellule_runtime::Result<[u8; N]> { + match value { + SqlValue::Blob(value) => value + .as_slice() + .try_into() + .map_err(|_| Error::Command("invalid catalog bytes")), + _ => Err(Error::Command("invalid catalog bytes")), + } +} +// Context time is sampled before the worker queue. Refresh once for lease +// decisions, clamping against logical time just like credential expiry checks. +pub(super) fn now(admitted: i64) -> cellule_runtime::Result { + let elapsed = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_err(|source| Error::Facility { + name: "catalog lease clock", + source: Box::new(source), + })?; + Ok(i64::try_from(elapsed.as_millis()) + .map_err(|_| Error::Command("catalog lease clock overflow"))? + .max(admitted)) +} +pub(super) fn expiry(now: i64, ms: u64) -> cellule_runtime::Result { + if ms == 0 || ms > MAX_LEASE_MS { + return Err(Error::Command("invalid catalog lease duration")); + } + now.checked_add(ms as i64) + .ok_or(Error::Command("catalog lease expiry overflow")) +} +pub(super) fn identity( + sets: &[SqlResultSet], + repository: [u8; 16], +) -> cellule_runtime::Result> { + let Some(row) = rows(sets)?.first() else { + return Ok(None); + }; + let [id, SqlValue::Text(format)] = row.as_slice() else { + return Err(Error::Command("invalid catalog repository identity")); + }; + if fixed::<16>(id)? != repository { + return Ok(None); + } + Ok(Some(ObjectFormat::parse(format).ok_or(Error::Command( + "invalid catalog repository format", + ))?)) +} +pub(super) fn generation( + sets: &[SqlResultSet], + repository: [u8; 16], + format: ObjectFormat, +) -> cellule_runtime::Result { + let Some([generation, catalog, certificate, refs]) = rows(sets)?.first().map(Vec::as_slice) + else { + return Err(Error::Command("missing catalog generation fact")); + }; + let generation = unsigned(generation)?; + let catalog = match catalog { + SqlValue::Null => None, + SqlValue::Blob(bytes) => { + let mut decoder = BoundedDecoder::new(bytes, 256)?; + let catalog = StoredCatalog::decode(&mut decoder)?; + decoder.finish()?; + if catalog.repository != repository || catalog.format != format { + return Err(Error::Command("catalog generation context differs")); + } + Some(catalog) + } + _ => return Err(Error::Command("invalid catalog generation descriptor")), + }; + let certificate = match certificate { + SqlValue::Null => None, + value => Some(fixed(value)?), + }; + let refs = match refs { + SqlValue::Null => None, + SqlValue::Blob(bytes) => { + let mut decoder = BoundedDecoder::new(bytes, 128)?; + let refs = RefStateSnapshotRoot::decode(&mut decoder)?; + decoder.finish()?; + Some(refs) + } + _ => return Err(Error::Command("invalid generation ref snapshot")), + }; + let fact = GenerationFact { + generation, + catalog, + refs, + certificate, + }; + fact.validate()?; + Ok(fact) +} +#[derive(Clone)] +pub(super) struct Operation { + pub actor: String, + pub token: PreparationToken, + pub generation: Option, + pub expires: i64, +} +pub(super) fn operation( + sets: &[SqlResultSet], + repository: [u8; 16], + operation: [u8; 16], +) -> cellule_runtime::Result> { + let Some(row) = rows(sets)?.first() else { + return Ok(None); + }; + let [ + SqlValue::Text(actor), + digest, + incarnation, + epoch, + sequence, + artifact_operation, + generation, + SqlValue::Integer(expires), + ] = row.as_slice() + else { + return Err(Error::Command("invalid catalog operation row")); + }; + let token = PreparationToken { + repository, + operation, + artifact_operation: fixed(artifact_operation)?, + request_digest: fixed(digest)?, + owner: OwnerFence { + incarnation: IncarnationId::from_bytes(fixed(incarnation)?), + epoch: u64::from_be_bytes(fixed(epoch)?), + }, + attempt: unsigned(sequence)?, + }; + if token.owner.epoch == 0 || token.attempt == 0 || *expires < 0 { + return Err(Error::Command("invalid catalog operation authority")); + } + codec::artifact_valid(token.artifact_operation)?; + Ok(Some(Operation { + actor: actor.clone(), + token, + generation: optional_generation(generation)?, + expires: *expires, + })) +} +pub(super) fn grant( + operation: &Operation, + format: ObjectFormat, + base: GenerationFact, + now: i64, +) -> cellule_runtime::Result { + if operation.generation != Some(base.generation) || operation.expires <= now { + return Err(Error::Command("catalog lease and generation differ")); + } + Ok(PreparationLease { + token: operation.token, + base, + format, + observed_at_ms: now, + expires_at_ms: operation.expires, + }) +} +pub(super) const IDENTITY: &str = + "SELECT repository_id,object_format FROM repository_identity WHERE singleton=1"; +pub(super) const CURRENT: &str = "SELECT g.generation,g.catalog,g.certificate,g.refs FROM catalog_state s JOIN catalog_generations g ON g.generation=s.generation WHERE s.singleton=1"; +pub(super) const GENERATION: &str = + "SELECT generation,catalog,certificate,refs FROM catalog_generations WHERE generation=?1"; +pub(super) fn quota( + context: &CommandContext<'_, '_>, + new_operation: bool, +) -> cellule_runtime::Result { + let leases = context.sql(&statement( + "SELECT count(*) FROM (SELECT admission_sequence FROM catalog_leases LIMIT ?1)", + vec![number(MAX_GENERATION_LEASES + 1)?], + ))?; + let Some([count]) = rows(&leases)?.first().map(Vec::as_slice) else { + return Err(Error::Command("missing catalog lease count")); + }; + if unsigned(count)? >= MAX_GENERATION_LEASES { + return Ok(false); + } + if new_operation { + let operations = context.sql(&statement( + "SELECT count(*) FROM (SELECT id FROM catalog_operations LIMIT ?1)", + vec![number(MAX_OPERATIONS + 1)?], + ))?; + let Some([count]) = rows(&operations)?.first().map(Vec::as_slice) else { + return Err(Error::Command("missing catalog operation count")); + }; + if unsigned(count)? >= MAX_OPERATIONS { + return Ok(false); + } + } + Ok(true) +} +pub(super) fn insert_lease( + context: &CommandContext<'_, '_>, + token: PreparationToken, + generation: Option, + expires: i64, +) -> cellule_runtime::Result<()> { + context.sql(&statement("INSERT INTO catalog_leases(incarnation,admission_sequence,operation,owner_epoch,artifact_operation,generation,expires_at_ms) VALUES(?1,?2,?3,?4,?5,?6,?7)",vec![blob(token.owner.incarnation.as_bytes()),number(token.attempt)?,blob(token.operation),blob(token.owner.epoch.to_be_bytes()),blob(token.artifact_operation),generation.map(number).transpose()?.unwrap_or(SqlValue::Null),SqlValue::Integer(expires)]))?; + Ok(()) +} + +/// Persistent monotonic namespace allocation inside the admitted transaction. +/// It is preserved by snapshots/recovery; isolated rollback restores must use +/// a new provider namespace, as the deployment restore contract requires. +pub(super) fn allocate_artifacts( + context: &CommandContext<'_, '_>, +) -> cellule_runtime::Result<[u8; 16]> { + let changed = context.sql(&statement("UPDATE repository_identity SET artifact_sequence=artifact_sequence+1 WHERE singleton=1 AND artifact_sequence<9223372036854775807", vec![]))?; + if changed.first().is_none_or(|set| set.rows_affected != 1) { + return Err(Error::Command( + "artifact namespace allocator is exhausted or absent", + )); + } + let result = context.sql(&statement( + "SELECT artifact_sequence FROM repository_identity WHERE singleton=1", + vec![], + ))?; + let Some([value]) = rows(&result)?.first().map(Vec::as_slice) else { + return Err(Error::Command("artifact namespace allocator is absent")); + }; + let sequence = unsigned(value)?; + let mut operation = *b"CANOPY01\0\0\0\0\0\0\0\0"; + operation[8..].copy_from_slice(&sequence.to_be_bytes()); + Ok(operation) +} diff --git a/crates/canopy-server/src/packs/publication/staging.rs b/crates/canopy-server/src/packs/publication/staging.rs new file mode 100644 index 0000000..d40d7e7 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/staging.rs @@ -0,0 +1,313 @@ +//! Long-running input custody without a catalog generation floor. Late binding +//! preserves the creating namespace and expiry and grants no verification proof. +use super::{commands::*, sql::*, *}; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct StagingLease { + pub token: PreparationToken, + pub format: ObjectFormat, + pub observed_at_ms: i64, + pub expires_at_ms: i64, +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum StagingReply { + Granted(Box), + Denied(PreparationDenial), +} +fn denied(reason: PreparationDenial) -> CommandResult { + CommandResult::Rejected(StagingReply::Denied(reason)) +} +fn granted( + row: &Operation, + format: ObjectFormat, + now: i64, +) -> cellule_runtime::Result { + if row.generation.is_some() || row.expires <= now { + return Err(Error::Command("input staging phase differs")); + } + Ok(StagingLease { + token: row.token, + format, + observed_at_ms: now, + expires_at_ms: row.expires, + }) +} +fn insert_operation( + context: &CommandContext<'_, '_>, + row: &Operation, +) -> cellule_runtime::Result<()> { + context.sql(&statement("INSERT INTO catalog_operations(id,actor,request_digest,incarnation,owner_epoch,admission_sequence,artifact_operation,generation,expires_at_ms) VALUES(?1,?2,?3,?4,?5,?6,?7,NULL,?8)", vec![blob(row.token.operation),SqlValue::Text(row.actor.clone()),blob(row.token.request_digest),blob(row.token.owner.incarnation.as_bytes()),blob(row.token.owner.epoch.to_be_bytes()),number(row.token.attempt)?,blob(row.token.artifact_operation),SqlValue::Integer(row.expires)]))?; + Ok(()) +} + +/// Allocate durable input custody using the same bounded attempt and pin rows. +/// A staging lease cannot open a PreparationBaseResolver or certify publication. +pub struct BeginStaging; +impl Command for BeginStaging { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 24; + const CODEC_VERSION: u32 = 1; + type Input = BeginRequest; + type Output = StagingReply; + fn execute( + context: &mut CommandContext<'_, '_>, + input: BeginRequest, + ) -> cellule_runtime::Result> { + let Some(format) = authorized(context, input.repository, &input.actor, TokenScope::Write)? + else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + if !logical_available(context, &input)? { + return Ok(denied(PreparationDenial::Conflict)); + } + let now = now(context.now_ms())?; + let expires = expiry(now, input.lease_ms)?; + if let Some(row) = operation( + &context.sql(&statement(OPERATION, vec![blob(input.operation)]))?, + input.repository, + input.operation, + )? { + if row.actor != input.actor + || row.token.request_digest != input.request_digest + || row.generation.is_some() + { + return Ok(denied(PreparationDenial::Conflict)); + } + if row.token.owner != context.owner_fence() { + return Ok(denied(PreparationDenial::Stale)); + } + if row.expires <= now { + return Ok(denied(PreparationDenial::Expired)); + } + check_pin(context, &row)?; + return Ok(CommandResult::Success(StagingReply::Granted(Box::new( + granted(&row, format, now)?, + )))); + } + if !quota(context, true)? { + return Ok(denied(PreparationDenial::Capacity)); + } + let next = token( + context, + input.repository, + input.operation, + input.request_digest, + )?; + insert_lease(context, next, None, expires)?; + let row = Operation { + actor: input.actor, + token: next, + generation: None, + expires, + }; + insert_operation(context, &row)?; + Ok(CommandResult::Success(StagingReply::Granted(Box::new( + granted(&row, format, now)?, + )))) + } +} + +/// Extend only artifact custody. A staging renewal never acquires a floor. +pub struct RenewStaging; +impl Command for RenewStaging { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 25; + const CODEC_VERSION: u32 = 1; + type Input = LeaseRequest; + type Output = StagingReply; + fn execute( + context: &mut CommandContext<'_, '_>, + input: LeaseRequest, + ) -> cellule_runtime::Result> { + let check = input.check; + let Some(format) = authorized( + context, + check.token.repository, + &check.actor, + TokenScope::Write, + )? + else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + if check.token.owner != context.owner_fence() { + return Ok(denied(PreparationDenial::Stale)); + } + let Some(mut row) = load(context, check.token)? else { + return Ok(denied(PreparationDenial::Missing)); + }; + if !matched(&row, &check) { + return Ok(denied(PreparationDenial::Stale)); + } + if row.generation.is_some() { + return Ok(denied(PreparationDenial::Conflict)); + } + check_pin(context, &row)?; + let now = now(context.now_ms())?; + if row.expires <= now { + return Ok(denied(PreparationDenial::Expired)); + } + row.expires = row.expires.max(expiry(now, input.lease_ms)?); + context.sql(&statement("UPDATE catalog_leases SET expires_at_ms=?1 WHERE incarnation=?2 AND admission_sequence=?3", vec![SqlValue::Integer(row.expires),blob(row.token.owner.incarnation.as_bytes()),number(row.token.attempt)?]))?; + context.sql(&statement( + "UPDATE catalog_operations SET expires_at_ms=?1 WHERE id=?2", + vec![SqlValue::Integer(row.expires), blob(row.token.operation)], + ))?; + Ok(CommandResult::Success(StagingReply::Granted(Box::new( + granted(&row, format, now)?, + )))) + } +} + +/// Bind once to the current generation, after input upload/normalization. This +/// command neither shortens existing input custody nor renews it. Repeating it +/// returns the original floor; further frontier selection is read-only. +pub struct BindStaging; +impl Command for BindStaging { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 26; + const CODEC_VERSION: u32 = 1; + type Input = LeaseCheck; + type Output = PreparationReply; + fn execute( + context: &mut CommandContext<'_, '_>, + check: LeaseCheck, + ) -> cellule_runtime::Result> { + let deny = |reason| CommandResult::Rejected(PreparationReply::Denied(reason)); + let Some(format) = authorized( + context, + check.token.repository, + &check.actor, + TokenScope::Write, + )? + else { + return Ok(deny(PreparationDenial::Unauthorized)); + }; + if check.token.owner != context.owner_fence() { + return Ok(deny(PreparationDenial::Stale)); + } + let Some(mut row) = load(context, check.token)? else { + return Ok(deny(PreparationDenial::Missing)); + }; + if !matched(&row, &check) { + return Ok(deny(PreparationDenial::Stale)); + } + check_pin(context, &row)?; + let now = now(context.now_ms())?; + if row.expires <= now { + return Ok(deny(PreparationDenial::Expired)); + } + let base = fact(context, check.token.repository, format, row.generation)?; + if row.generation.is_none() { + context.sql(&statement("UPDATE catalog_leases SET generation=?1 WHERE incarnation=?2 AND admission_sequence=?3 AND generation IS NULL", vec![number(base.generation)?,blob(row.token.owner.incarnation.as_bytes()),number(row.token.attempt)?]))?; + context.sql(&statement( + "UPDATE catalog_operations SET generation=?1 WHERE id=?2 AND generation IS NULL", + vec![number(base.generation)?, blob(row.token.operation)], + ))?; + row.generation = Some(base.generation); + } + Ok(CommandResult::Success(PreparationReply::Granted(Box::new( + grant(&row, format, base, now)?, + )))) + } +} + +/// Recover input admission under a new admitted attempt. The previous namespace +/// remains independently retained; no unchecked cross-attempt input adoption. +pub struct ClaimStaging; +impl Command for ClaimStaging { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 28; + const CODEC_VERSION: u32 = 1; + type Input = LeaseRequest; + type Output = StagingReply; + fn execute( + context: &mut CommandContext<'_, '_>, + input: LeaseRequest, + ) -> cellule_runtime::Result> { + let check = input.check; + let Some(format) = authorized( + context, + check.token.repository, + &check.actor, + TokenScope::Write, + )? + else { + return Ok(denied(PreparationDenial::Unauthorized)); + }; + let Some(row) = load(context, check.token)? else { + return Ok(denied(PreparationDenial::Missing)); + }; + if !matched(&row, &check) { + return Ok(denied(PreparationDenial::Stale)); + } + if row.generation.is_some() { + return Ok(denied(PreparationDenial::Conflict)); + } + check_pin(context, &row)?; + if !quota(context, false)? { + return Ok(denied(PreparationDenial::Capacity)); + } + let now = now(context.now_ms())?; + let expires = expiry(now, input.lease_ms)?; + let next = token( + context, + check.token.repository, + check.token.operation, + check.token.request_digest, + )?; + insert_lease(context, next, None, expires)?; + context.sql(&statement("UPDATE catalog_operations SET incarnation=?1,owner_epoch=?2,admission_sequence=?3,artifact_operation=?4,expires_at_ms=?5 WHERE id=?6", vec![blob(next.owner.incarnation.as_bytes()),blob(next.owner.epoch.to_be_bytes()),number(next.attempt)?,blob(next.artifact_operation),SqlValue::Integer(expires),blob(next.operation)]))?; + let row = Operation { + actor: check.actor, + token: next, + generation: None, + expires, + }; + Ok(CommandResult::Success(StagingReply::Granted(Box::new( + granted(&row, format, now)?, + )))) + } +} + +pub struct CheckStaging; +impl Query for CheckStaging { + const MODULE: &'static str = RepositoryModule::NAME; + const ID: u32 = 27; + const CODEC_VERSION: u32 = 1; + type Input = LeaseCheck; + type Output = Option; + fn execute( + context: &mut QueryContext<'_>, + check: LeaseCheck, + ) -> cellule_runtime::Result { + validate_component(&check.actor)?; + if !decode_access(&context.sql(&SqlBatch { + statements: vec![access_statement(&check.actor)], + })?)? + .is_some_and(|role| role >= TokenScope::Write) + { + return Ok(None); + } + let Some(format) = identity( + &context.sql(&statement(IDENTITY, vec![]))?, + check.token.repository, + )? + else { + return Ok(None); + }; + let Some(row) = operation( + &context.sql(&statement(OPERATION, vec![blob(check.token.operation)]))?, + check.token.repository, + check.token.operation, + )? + else { + return Ok(None); + }; + let now = now(context.now_ms())?; + if !matched(&row, &check) || row.generation.is_some() || row.expires <= now { + return Ok(None); + } + pin(&context.sql(&pin_query(row.token)?)?, &row)?; + Ok(Some(granted(&row, format, now)?)) + } +} diff --git a/crates/canopy-server/src/packs/publication/staging_service.rs b/crates/canopy-server/src/packs/publication/staging_service.rs new file mode 100644 index 0000000..d017cfb --- /dev/null +++ b/crates/canopy-server/src/packs/publication/staging_service.rs @@ -0,0 +1,1740 @@ +//! Service-owned, bounded input custody. Never infer a live deadline from a +//! replayed command, discard ambiguous evidence, or let an observer cancel work. +use super::*; +use crate::packs::catalog::{CatalogFiles, CatalogIndexes}; +use cellule_runtime::{ + CellClient, CellTarget, Committed, InvocationError, MutationIdentity, PreparedCommand, Receipt, +}; +use std::{ + any::Any, + collections::HashMap, + future::Future, + sync::{Arc, Mutex}, + time::Duration, +}; +use tokio::{ + sync::{Notify, OwnedSemaphorePermit, Semaphore, watch}, + time::{Instant, sleep_until}, +}; + +mod bound; +use bound::accept_bound; +mod publication; +pub use publication::{StagedPublicationFailure, StagedPublicationTicket}; + +const COMMAND_BYTES: u32 = 4096; +#[derive(Clone, Copy, Debug)] +pub struct StagingLimits { + pub operations: usize, + pub per_actor: usize, + pub workers: usize, + pub workers_per_actor: usize, + pub lease_ms: u64, + pub renew_before_ms: u64, + pub lifetime_ms: u64, + pub bound_lifetime_ms: u64, +} +impl Default for StagingLimits { + fn default() -> Self { + Self { + operations: 32, + per_actor: 8, + workers: 64, + workers_per_actor: 8, + lease_ms: DEFAULT_LEASE_MS, + renew_before_ms: DEFAULT_LEASE_MS / 2, + lifetime_ms: 4 * 60 * 60 * 1000, + bound_lifetime_ms: DEFAULT_LEASE_MS, + } + } +} +impl StagingLimits { + fn validate(self) -> Result<(), StagingError> { + if self.operations < 2 + || self.operations > MAX_OPERATIONS as usize + || self.per_actor == 0 + || self.per_actor >= self.operations + || self.workers == 0 + || self.workers > MAX_GENERATION_LEASES as usize + || self.workers_per_actor == 0 + || self.workers_per_actor > self.workers + || self.lease_ms == 0 + || self.lease_ms > MAX_LEASE_MS + || self.renew_before_ms == 0 + || self.renew_before_ms >= self.lease_ms + || self.lifetime_ms < self.lease_ms + || self.lifetime_ms > 24 * 60 * 60 * 1000 + || self.bound_lifetime_ms == 0 + || self.bound_lifetime_ms > MAX_LEASE_MS + { + return Err(StagingError::InvalidLimits); + } + Ok(()) + } +} +#[derive(Debug, thiserror::Error)] +pub enum StagingError { + #[error("invalid staging limits")] + InvalidLimits, + #[error("staging admission is closed")] + Closed, + #[error("staging admission capacity exceeded")] + Capacity, + #[error("staging request belongs to another coordinator")] + Foreign, + #[error("logical staging request already admitted")] + Duplicate, + #[error("staging has no live matching custody")] + Inactive, + #[error("staging is not ready for this action")] + NotReady, + #[error("staging context differs")] + Context, + #[error("staging clock failed")] + Clock, + #[error("staging worker panicked")] + Worker, + #[error("input preparation failed")] + Input(#[source] Box), + #[error("staging begin failed")] + Begin(#[source] Box>), + #[error("staging claim failed")] + Claim(#[source] Box>), + #[error("input checkpoint registration failed")] + Checkpoint(#[source] Box>), + #[error("staging renewal failed")] + Renew(#[source] Box>), + #[error("bound preparation renewal failed")] + BoundRenew(#[source] Box>), + #[error("bound preparation claim failed")] + BoundClaim(#[source] Box>), + #[error("staging bind failed")] + Bind(#[source] Box>), + #[error("staging query failed")] + Query(#[source] Box>>), + #[error("bound base failed")] + Base(#[from] PreparationBaseError), + #[error("final publication admission failed: {0}")] + PublicationAdmission(PublicationScheduleError), + #[error("final publication failed: {0}")] + Publication(#[source] Arc), +} +impl StagingError { + fn uncertain(&self) -> bool { + fn unknown(e: &InvocationError) -> bool { + matches!( + e, + InvocationError::Pending(_) | InvocationError::InvalidPublishedResult { .. } + ) + } + match self { + Self::Begin(e) | Self::Renew(e) | Self::Claim(e) | Self::Checkpoint(e) => unknown(e), + Self::Bind(e) | Self::BoundRenew(e) | Self::BoundClaim(e) => unknown(e), + _ => false, + } + } +} +#[derive(Clone, Copy, Debug)] +pub struct StagingBound { + pub lease: PreparationLease, + pub receipt: Receipt, +} +#[derive(Clone, Debug)] +pub enum StagingState { + Starting, + Active(StagingLease), + Draining(StagingLease), + Binding, + RegisteringInputs, + Resolving, + Uncertain(Arc), + Bound(Arc), + Finishing, + Publishing, + /// Original final outcome; no post-completion lease query can erase it. + Published(Result>), + Fenced(Arc), + Stopped, +} +impl StagingState { + fn terminal(&self) -> bool { + matches!( + self, + Self::Bound(_) | Self::Published(_) | Self::Fenced(_) | Self::Stopped + ) + } +} +#[must_use] +pub struct ReadyStaging { + inner: Box, +} +struct StagingRequest { + client: CellClient, + target: CellTarget, + request: BeginRequest, + command: Exact, + bound_source: Option, +} +impl ReadyStaging { + pub async fn new( + client: CellClient, + target: CellTarget, + request: BeginRequest, + identity: MutationIdentity, + ) -> Result { + if crate::repository_target(target.tenant(), target.application(), request.repository) + .map_err(|_| StagingError::Context)? + != target + { + return Err(StagingError::Context); + } + request + .encode(&mut BoundedEncoder::new(COMMAND_BYTES).map_err(|_| StagingError::Context)?) + .map_err(|_| StagingError::Context)?; + let command = client + .prepare_command::(&target, identity, request.clone()) + .await + .map_err(|e| StagingError::Begin(Box::new(e)))?; + Ok(Self { + inner: Box::new(StagingRequest { + client, + target, + request, + command: Exact::Begin(command), + bound_source: None, + }), + }) + } + /// Resume the same logical staging request through the authoritative Claim + /// command, with a new creating namespace and independent previous pin. + pub async fn claim( + client: CellClient, + target: CellTarget, + request: LeaseRequest, + identity: MutationIdentity, + ) -> Result { + let begin = BeginRequest { + repository: request.check.token.repository, + operation: request.check.token.operation, + request_digest: request.check.token.request_digest, + actor: request.check.actor.clone(), + lease_ms: request.lease_ms, + }; + if crate::repository_target(target.tenant(), target.application(), begin.repository) + .map_err(|_| StagingError::Context)? + != target + { + return Err(StagingError::Context); + } + request + .encode(&mut BoundedEncoder::new(COMMAND_BYTES).map_err(|_| StagingError::Context)?) + .map_err(|_| StagingError::Context)?; + let command = client + .prepare_command::(&target, identity, request) + .await + .map_err(|e| StagingError::Claim(Box::new(e)))?; + Ok(Self { + inner: Box::new(StagingRequest { + client, + target, + request: begin, + command: Exact::Claim(command), + bound_source: None, + }), + }) + } + /// Admit a bound takeover into the same operation/worker lifecycle. The + /// exact old token is checked at execution; no prior live session is needed. + pub async fn claim_bound( + client: CellClient, + target: CellTarget, + request: LeaseRequest, + identity: MutationIdentity, + ) -> Result { + let begin = BeginRequest { + repository: request.check.token.repository, + operation: request.check.token.operation, + request_digest: request.check.token.request_digest, + actor: request.check.actor.clone(), + lease_ms: request.lease_ms, + }; + if crate::repository_target(target.tenant(), target.application(), begin.repository) + .map_err(|_| StagingError::Context)? + != target + { + return Err(StagingError::Context); + } + request + .encode(&mut BoundedEncoder::new(COMMAND_BYTES).map_err(|_| StagingError::Context)?) + .map_err(|_| StagingError::Context)?; + let source = request.check.clone(); + let command = client + .prepare_command::(&target, identity, request) + .await + .map_err(|e| StagingError::BoundClaim(Box::new(e)))?; + Ok(Self { + inner: Box::new(StagingRequest { + client, + target, + request: begin, + command: Exact::BoundClaim(command), + bound_source: Some(source), + }), + }) + } +} +struct ActorAdmission { + operations: usize, + workers: Arc, +} +#[derive(Default)] +struct Admission { + closed: bool, + jobs: HashMap<[u8; 16], Arc>, + actors: HashMap, +} +struct Inner { + target: CellTarget, + limits: StagingLimits, + admission: Mutex, + workers: Arc, + drained: Notify, + #[cfg(test)] + fault: std::sync::atomic::AtomicU8, +} +struct Local { + lease: Option, + bound: Option>, + bound_source: Option, + bound_started: Option, + bound_result: Option>, + bound_renewal: Option>, + finishing: bool, + deadline: Instant, + lifetime: Instant, + workers: usize, + seal: bool, + stop: bool, + fenced: bool, + recovery: bool, + renew: bool, +} +trait RetainedWork: Any + Send + Sync { + fn fence_completed(&self); + fn erased(self: Arc) -> Arc; +} +#[derive(Default)] +struct WorkSlots { + next: u64, + slots: HashMap>, +} +struct Job { + client: CellClient, + target: CellTarget, + actor: String, + operation: [u8; 16], + actor_workers: Arc, + local: Mutex, + work: Mutex, + exact: Mutex>, + checkpoint: Mutex>>, + publication: Mutex>, + status: watch::Sender, + changed: Notify, +} +#[derive(Clone)] +enum Exact { + Begin(PreparedCommand), + Claim(PreparedCommand), + Checkpoint(PreparedCommand), + BoundCheckpoint(PreparedCommand), + Renew(PreparedCommand), + Bind(PreparedCommand), + BoundClaim(PreparedCommand), + BoundRenew(PreparedCommand), +} +enum Outcome { + Stage(Committed), + Bound(Committed), + BoundClaim(Committed), + BoundRenew(Committed), + Checkpoint(Committed), + BoundCheckpoint(Committed), +} +impl Exact { + fn pending(&self) -> StagingError { + match self { + Self::Begin(c) => StagingError::Begin(Box::new(InvocationError::Pending(Box::new( + c.evidence().clone(), + )))), + Self::Claim(c) => StagingError::Claim(Box::new(InvocationError::Pending(Box::new( + c.evidence().clone(), + )))), + Self::Checkpoint(c) | Self::BoundCheckpoint(c) => StagingError::Checkpoint(Box::new( + InvocationError::Pending(Box::new(c.evidence().clone())), + )), + Self::Renew(c) => StagingError::Renew(Box::new(InvocationError::Pending(Box::new( + c.evidence().clone(), + )))), + Self::BoundClaim(c) => StagingError::BoundClaim(Box::new(InvocationError::Pending( + Box::new(c.evidence().clone()), + ))), + Self::BoundRenew(c) => StagingError::BoundRenew(Box::new(InvocationError::Pending( + Box::new(c.evidence().clone()), + ))), + Self::Bind(c) => StagingError::Bind(Box::new(InvocationError::Pending(Box::new( + c.evidence().clone(), + )))), + } + } + async fn execute( + self, + client: CellClient, + recover: bool, + fault: u8, + ) -> Result { + match self { + Self::Begin(c) => super::exact::invoke(&client, c, recover, COMMAND_BYTES, fault) + .await + .map(Outcome::Stage) + .map_err(|e| StagingError::Begin(Box::new(e))), + Self::Claim(c) => super::exact::invoke(&client, c, recover, COMMAND_BYTES, fault) + .await + .map(Outcome::Stage) + .map_err(|e| StagingError::Claim(Box::new(e))), + Self::Checkpoint(c) => super::exact::invoke(&client, c, recover, COMMAND_BYTES, fault) + .await + .map(Outcome::Checkpoint) + .map_err(|e| StagingError::Checkpoint(Box::new(e))), + Self::BoundCheckpoint(c) => { + super::exact::invoke(&client, c, recover, COMMAND_BYTES, fault) + .await + .map(Outcome::BoundCheckpoint) + .map_err(|e| StagingError::Checkpoint(Box::new(e))) + } + Self::Renew(c) => super::exact::invoke(&client, c, recover, COMMAND_BYTES, fault) + .await + .map(Outcome::Stage) + .map_err(|e| StagingError::Renew(Box::new(e))), + Self::BoundClaim(c) => super::exact::invoke(&client, c, recover, COMMAND_BYTES, fault) + .await + .map(Outcome::BoundClaim) + .map_err(|e| StagingError::BoundClaim(Box::new(e))), + Self::BoundRenew(c) => super::exact::invoke(&client, c, recover, COMMAND_BYTES, fault) + .await + .map(Outcome::BoundRenew) + .map_err(|e| StagingError::BoundRenew(Box::new(e))), + Self::Bind(c) => super::exact::invoke(&client, c, recover, COMMAND_BYTES, fault) + .await + .map(Outcome::Bound) + .map_err(|e| StagingError::Bind(Box::new(e))), + } + } +} + +/// Keep one instance per repository in the service. Local ownership is not a +/// durable outbox, a public ACL endpoint, or permission for remote deletion. +#[derive(Clone)] +pub struct StagingCoordinator { + inner: Arc, +} +#[derive(Clone)] +#[must_use] +pub struct StagingTicket { + inner: Arc, + job: Arc, +} +#[derive(Clone, Copy, Debug)] +pub struct StagingStats { + pub admitted: usize, + pub accounts: usize, + pub workers: usize, + pub uncertain: usize, + pub command_bytes: u64, + pub closed: bool, +} +impl StagingCoordinator { + pub fn new(target: CellTarget, limits: StagingLimits) -> Result { + limits.validate()?; + Ok(Self { + inner: Arc::new(Inner { + target, + limits, + admission: Mutex::new(Admission::default()), + workers: Arc::new(Semaphore::new(limits.workers)), + drained: Notify::new(), + #[cfg(test)] + fault: std::sync::atomic::AtomicU8::new(0), + }), + }) + } + /// Admission is synchronous; a canceled observer never owns the command. + pub fn submit( + &self, + ready: ReadyStaging, + ) -> Result { + let mut admission = self.inner.admission.lock().expect("staging admission"); + let error = if ready.inner.target != self.inner.target { + Some(StagingError::Foreign) + } else if admission.closed { + Some(StagingError::Closed) + } else if ready.inner.request.lease_ms != self.inner.limits.lease_ms { + Some(StagingError::Context) + } else if admission.jobs.contains_key(&ready.inner.request.operation) { + Some(StagingError::Duplicate) + } else if admission.jobs.len() >= self.inner.limits.operations + || admission + .actors + .get(&ready.inner.request.actor) + .map(|actor| actor.operations) + .unwrap_or(0) + >= self.inner.limits.per_actor + { + Some(StagingError::Capacity) + } else { + None + }; + if let Some(error) = error { + return Err((error, ready)); + } + let actor = admission + .actors + .entry(ready.inner.request.actor.clone()) + .or_insert_with(|| ActorAdmission { + operations: 0, + workers: Arc::new(Semaphore::new(self.inner.limits.workers_per_actor)), + }); + actor.operations += 1; + let actor_workers = Arc::clone(&actor.workers); + let now = Instant::now(); + let job = Arc::new(Job { + client: ready.inner.client, + target: ready.inner.target, + actor: ready.inner.request.actor, + operation: ready.inner.request.operation, + actor_workers, + local: Mutex::new(Local { + lease: None, + bound: None, + bound_source: ready.inner.bound_source.clone(), + bound_started: ready.inner.bound_source.as_ref().map(|_| now), + bound_result: None, + bound_renewal: None, + finishing: false, + deadline: now, + lifetime: now + Duration::from_millis(self.inner.limits.lifetime_ms), + workers: 0, + seal: false, + stop: false, + fenced: false, + recovery: false, + renew: false, + }), + work: Mutex::new(WorkSlots::default()), + exact: Mutex::new(Some(ready.inner.command)), + checkpoint: Mutex::new(None), + publication: Mutex::new(None), + status: watch::channel(StagingState::Starting).0, + changed: Notify::new(), + }); + admission.jobs.insert(job.operation, Arc::clone(&job)); + tokio::spawn(supervise(Arc::clone(&self.inner), Arc::clone(&job))); + Ok(StagingTicket { + inner: Arc::clone(&self.inner), + job, + }) + } + pub fn pending(&self, operation: [u8; 16]) -> Option { + self.inner + .admission + .lock() + .expect("staging admission") + .jobs + .get(&operation) + .map(|job| StagingTicket { + inner: Arc::clone(&self.inner), + job: Arc::clone(job), + }) + } + pub fn recover(&self, ticket: &StagingTicket) -> Result<(), StagingError> { + if !Arc::ptr_eq(&self.inner, &ticket.inner) { + return Err(StagingError::Foreign); + } + if !matches!(ticket.state(), StagingState::Uncertain(_)) { + return Err(StagingError::NotReady); + } + ticket.job.local.lock().expect("staging local").recovery = true; + ticket.job.status.send_replace(StagingState::Resolving); + ticket.job.changed.notify_one(); + Ok(()) + } + pub fn stats(&self) -> StagingStats { + let a = self.inner.admission.lock().expect("staging admission"); + StagingStats { + admitted: a.jobs.len(), + accounts: a.actors.len(), + workers: self.inner.limits.workers - self.inner.workers.available_permits(), + uncertain: a + .jobs + .values() + .filter(|j| matches!(*j.status.borrow(), StagingState::Uncertain(_))) + .count(), + command_bytes: a + .jobs + .values() + .map(|job| { + let copies = + 2 + u64::from(job.checkpoint.lock().expect("staging checkpoint").is_some()); + copies * COMMAND_BYTES as u64 + }) + .sum(), + closed: a.closed, + } + } + /// Stop admission and renew while accepted workers drain. Uncertain exact + /// commands remain charged and returned; explicit recovery remains possible. + pub async fn close_and_drain(&self) -> Vec { + loop { + let wake = self.inner.drained.notified(); + tokio::pin!(wake); + wake.as_mut().enable(); + { + let mut a = self.inner.admission.lock().expect("staging admission"); + a.closed = true; + for job in a.jobs.values() { + job.local.lock().expect("staging local").stop = true; + job.changed.notify_one(); + } + if a.jobs.values().all(|j| { + matches!(*j.status.borrow(), StagingState::Uncertain(_)) + && j.local.lock().expect("staging local").workers == 0 + }) { + return a + .jobs + .values() + .map(|job| StagingTicket { + inner: Arc::clone(&self.inner), + job: Arc::clone(job), + }) + .collect(); + } + } + wake.await; + } + } + #[cfg(test)] + pub(super) fn fault_for_test(&self, fault: u8) { + self.inner + .fault + .store(fault, std::sync::atomic::Ordering::Release); + } +} +struct InputRegistration { + request: Mutex>, + result: watch::Sender>>>, + bound_digest: Option<[u8; 32]>, + checkpoint_digest: [u8; 32], +} +impl InputRegistration { + fn finish(&self, result: Result>) { + self.request + .lock() + .expect("staging checkpoint request") + .take(); + self.result.send_if_modified(|old| { + if old.is_some() { + false + } else { + *old = Some(result); + true + } + }); + } +} +#[derive(Clone)] +#[must_use] +pub struct StagedInputsTicket { + job: Arc, + registration: Arc, +} +impl StagedInputsTicket { + /// Observe the original durable registration receipt. An uncertain error + /// retains the exact command in the coordinator; recover and wait again. + /// A receipt is not a fresh authority or lease observation. + pub async fn wait(&self) -> Result> { + let mut result = self.registration.result.subscribe(); + let mut state = self.job.status.subscribe(); + loop { + if let Some(value) = result.borrow_and_update().clone() { + return value; + } + match state.borrow_and_update().clone() { + StagingState::Uncertain(error) | StagingState::Fenced(error) => return Err(error), + _ => {} + } + tokio::select! { + value = result.changed() => { if value.is_err() { return Err(Arc::new(StagingError::Worker)); } }, + value = state.changed() => { if value.is_err() { return Err(Arc::new(StagingError::Worker)); } }, + } + } + } +} +impl StagingTicket { + /// Synchronously transfer one bounded checkpoint request into service + /// custody. A dropped observer cannot cancel or replace its exact identity. + pub fn register_inputs( + &self, + proof: NativeInputCertificate, + identity: MutationIdentity, + ) -> Result)> { + let check = match proof.scoped_check(&self.job.target) { + Ok(check) => check, + Err(_) => return Err((StagingError::Context, Box::new(proof))), + }; + let (checkpoint_digest, predecessor) = match proof.checkpoint_lineage() { + Ok(value) => value, + Err(_) => return Err((StagingError::Context, Box::new(proof))), + }; + let local = self.job.local.lock().expect("staging local"); + let bound = local.bound.is_some(); + if local.fenced + || local.finishing + || local.stop + || (local.seal && !bound) + || Instant::now() >= local.deadline.min(local.lifetime) + || !(matches!(self.state(), StagingState::Active(_)) && !bound + || matches!(self.state(), StagingState::Bound(_)) && bound) + { + return Err((StagingError::Inactive, Box::new(proof))); + } + let expected = local + .bound + .as_ref() + .map(|s| s.lease.token) + .or_else(|| local.lease.map(|s| s.token)); + if expected != Some(check.token) || check.actor != self.job.actor { + return Err((StagingError::Context, Box::new(proof))); + } + let digest = if let Some(session) = &local.bound { + if session.live_lease().is_err() { + return Err((StagingError::Inactive, Box::new(proof))); + } + match proof.bound_digest(session) { + Ok(digest) => Some(digest), + Err(_) => return Err((StagingError::Context, Box::new(proof))), + } + } else { + None + }; + let mut checkpoint = self.job.checkpoint.lock().expect("staging checkpoint"); + if let Some(old) = checkpoint.as_ref() + && (bound + || predecessor != Some(old.checkpoint_digest) + || !old.result.borrow().as_ref().is_some_and(Result::is_ok)) + { + return Err((StagingError::Duplicate, Box::new(proof))); + } + let registration = Arc::new(InputRegistration { + request: Mutex::new(Some((proof, identity))), + result: watch::channel(None).0, + bound_digest: digest, + checkpoint_digest, + }); + *checkpoint = Some(Arc::clone(®istration)); + self.job.changed.notify_one(); + Ok(StagedInputsTicket { + job: Arc::clone(&self.job), + registration, + }) + } + /// Retrieve the accepted checkpoint observer after cancellation. + pub fn pending_inputs(&self) -> Option { + self.job + .checkpoint + .lock() + .expect("staging checkpoint") + .as_ref() + .map(|registration| StagedInputsTicket { + job: Arc::clone(&self.job), + registration: Arc::clone(registration), + }) + } + pub fn state(&self) -> StagingState { + self.job.status.borrow().clone() + } + pub async fn wait(&self) -> StagingState { + let mut status = self.job.status.subscribe(); + loop { + let state = status.borrow_and_update().clone(); + if !matches!( + state, + StagingState::Starting + | StagingState::Binding + | StagingState::RegisteringInputs + | StagingState::Resolving + | StagingState::Draining(_) + | StagingState::Finishing + | StagingState::Publishing + ) { + return state; + } + if status.changed().await.is_err() { + return status.borrow().clone(); + } + } + } + pub async fn wait_terminal(&self) -> StagingState { + let mut status = self.job.status.subscribe(); + loop { + let state = status.borrow_and_update().clone(); + if state.terminal() || matches!(state, StagingState::Uncertain(_)) { + return state; + } + if status.changed().await.is_err() { + return status.borrow().clone(); + } + } + } + pub fn seal(&self) -> Result<(), StagingError> { + let mut l = self.job.local.lock().expect("staging local"); + if l.lease.is_none() || l.stop || l.fenced || self.state().terminal() { + return Err(StagingError::NotReady); + } + if l.seal { + return Ok(()); + } + l.seal = true; + if let Some(lease) = l.lease + && !matches!(self.state(), StagingState::Uncertain(_)) + { + self.job.status.send_replace(StagingState::Draining(lease)); + } + self.job.changed.notify_one(); + Ok(()) + } + pub fn stop(&self) { + self.job.local.lock().expect("staging local").stop = true; + self.job.changed.notify_one(); + } + pub async fn open_base( + &self, + indexes: Arc, + files: Arc, + ) -> Result { + let session = self.bound_session()?; + let receipt = self.bound_result().ok_or(StagingError::NotReady)?.receipt; + session.refresh(receipt).await?; + Ok(PreparationBaseResolver::from_session((*session).clone(), indexes, files).await?) + } + /// A fresh live shared session, never authority reconstructed from Bound's + /// recorded timestamps. Shutdown and lifetime fences also cover every base. + pub fn bound_session(&self) -> Result, StagingError> { + let l = self.job.local.lock().expect("staging local"); + if l.stop + || l.finishing + || l.fenced + || Instant::now() >= l.lifetime + || !matches!(self.state(), StagingState::Bound(_)) + { + return Err(StagingError::Inactive); + } + let session = l.bound.clone().ok_or(StagingError::NotReady)?; + session.live_lease()?; + Ok(session) + } + /// Known Bind/Claim outcome survives failed fresh custody and graceful stop. + pub fn bound_result(&self) -> Option> { + self.job + .local + .lock() + .expect("staging local") + .bound_result + .clone() + } + pub fn bound_renewal(&self) -> Option> { + self.job + .local + .lock() + .expect("staging local") + .bound_renewal + .clone() + } + + /// Own input work in a task, separately from its observer. A producer error, + /// panic, expired custody or lost authority fences the session before Bind. + pub fn spawn(&self, producer: F) -> Result, StagingError> + where + F: FnOnce(StagingContext) -> Fut + Send + 'static, + Fut: Future> + Send + 'static, + T: Send + 'static, + { + self.spawn_in_phase(false, producer) + } + /// Bound verification, reconciliation and publication use the same worker + /// slots, cancellation/drain and typed result ownership as staging inputs. + pub fn spawn_bound(&self, producer: F) -> Result, StagingError> + where + F: FnOnce(Arc) -> Fut + Send + 'static, + Fut: Future> + Send + 'static, + T: Send + 'static, + { + let session = self.bound_session()?; + self.spawn_in_phase(true, move |_| producer(session)) + } + fn spawn_in_phase( + &self, + bound: bool, + producer: F, + ) -> Result, StagingError> + where + F: FnOnce(StagingContext) -> Fut + Send + 'static, + Fut: Future> + Send + 'static, + T: Send + 'static, + { + let permit = Arc::clone(&self.inner.workers) + .try_acquire_owned() + .map_err(|_| StagingError::Capacity)?; + let actor_permit = Arc::clone(&self.job.actor_workers) + .try_acquire_owned() + .map_err(|_| StagingError::Capacity)?; + let context = { + let mut l = self.job.local.lock().expect("staging local"); + if (l.seal && !bound) + || l.finishing + || l.bound.is_some() != bound + || l.stop + || l.fenced + || l.deadline <= Instant::now() + || matches!( + self.state(), + StagingState::Uncertain(_) | StagingState::Resolving + ) + { + return Err(StagingError::Inactive); + } + let (token, format) = if bound { + let (lease, _) = l + .bound + .as_ref() + .ok_or(StagingError::NotReady)? + .live_lease()?; + (lease.token, lease.format) + } else { + let lease = l.lease.ok_or(StagingError::NotReady)?; + (lease.token, lease.format) + }; + l.workers += 1; + StagingContext { + job: Arc::clone(&self.job), + token, + format, + bound, + } + }; + let guard = Activity { + inner: Arc::clone(&self.inner), + job: Arc::clone(&self.job), + permit: Some(permit), + actor_permit: Some(actor_permit), + }; + let mut work = self.job.work.lock().expect("staging work"); + let id = work.next.checked_add(1).ok_or(StagingError::Capacity)?; + work.next = id; + let slot = Arc::new(WorkSlot { + result: Mutex::new(None), + ready: watch::channel(false).0, + guard: Mutex::new(Some(guard)), + }); + work.slots.insert(id, slot.clone()); + drop(work); + let owned = Arc::clone(&slot); + let job = Arc::clone(&self.job); + tokio::spawn(async move { + let observe = context.clone(); + let mut task = tokio::spawn(async move { producer(context).await }); + let result = tokio::select! { result = &mut task => result.unwrap_or(Err(StagingError::Worker)), _ = observe.fenced() => { task.abort(); let _ = task.await; Err(StagingError::Inactive) } }; + let failed = { + let mut local = job.local.lock().expect("staging local"); + let result = if local.fenced + || local + .bound + .as_ref() + .is_some_and(|s| s.live_lease().is_err()) + || Instant::now() >= local.deadline.min(local.lifetime) + { + drop(result); + Err(StagingError::Inactive) + } else { + result + }; + let failed = result.is_err(); + if failed { + local.fenced = true; + } + *owned.result.lock().expect("staging result") = Some(result.map_err(Arc::new)); + owned.ready.send_replace(true); + failed + }; + if failed { + job.changed.notify_one(); + job.work.lock().expect("staging work").slots.remove(&id); + owned.guard.lock().expect("staging guard").take(); + } + }); + Ok(StagingTask { + id, + job: Arc::clone(&self.job), + slot, + }) + } + /// Recover a service-owned result after its observer was dropped. A wrong + /// result type fails; it cannot reinterpret a physical witness or descriptor. + pub fn pending_task(&self, id: u64) -> Option> { + let slot = self + .job + .work + .lock() + .expect("staging work") + .slots + .get(&id)? + .clone() + .erased() + .downcast::>() + .ok()?; + Some(StagingTask { + id, + job: Arc::clone(&self.job), + slot, + }) + } + #[cfg(test)] + pub(super) fn renew_for_test(&self) { + self.job.local.lock().expect("staging local").renew = true; + self.job.changed.notify_one(); + } +} +struct Activity { + inner: Arc, + job: Arc, + permit: Option, + actor_permit: Option, +} +impl Drop for Activity { + fn drop(&mut self) { + drop(self.actor_permit.take()); + drop(self.permit.take()); + self.job.local.lock().expect("staging local").workers -= 1; + self.job.changed.notify_one(); + self.inner.drained.notify_waiters(); + } +} +#[derive(Clone)] +pub struct StagingContext { + job: Arc, + token: PreparationToken, + format: ObjectFormat, + bound: bool, +} +impl StagingContext { + pub(super) fn capability(&self) -> (&CellClient, &CellTarget, LeaseCheck) { + ( + &self.job.client, + &self.job.target, + LeaseCheck { + token: self.token, + actor: self.job.actor.clone(), + }, + ) + } + pub fn token(&self) -> Result { + self.ensure_live()?; + Ok(self.token) + } + pub fn format(&self) -> ObjectFormat { + self.format + } + pub fn ensure_live(&self) -> Result<(), StagingError> { + let l = self.job.local.lock().expect("staging local"); + if l.fenced + || l.bound.is_some() != self.bound + || l.deadline <= Instant::now() + || l.lifetime <= Instant::now() + { + Err(StagingError::Inactive) + } else { + Ok(()) + } + } + async fn fenced(&self) { + let mut status = self.job.status.subscribe(); + loop { + let deadline = { + let l = self.job.local.lock().expect("staging local"); + if l.fenced || l.bound.is_some() != self.bound { + return; + } + l.deadline.min(l.lifetime) + }; + if Instant::now() >= deadline { + return; + } + tokio::select! { _ = sleep_until(deadline) => {}, result = status.changed() => { if result.is_err() { return; } } } + } + } +} +struct WorkSlot { + result: Mutex>>>, + ready: watch::Sender, + guard: Mutex>, +} +impl RetainedWork for WorkSlot { + fn erased(self: Arc) -> Arc { + self + } + fn fence_completed(&self) { + if *self.ready.borrow() { + let value = self.result.lock().expect("staging result").take(); + // Drop all owned input resources before returning their credit. + drop(value); + *self.result.lock().expect("staging result") = + Some(Err(Arc::new(StagingError::Inactive))); + self.guard.lock().expect("staging guard").take(); + } + } +} +#[must_use] +pub struct StagingTask { + id: u64, + job: Arc, + slot: Arc>, +} +impl StagingTask { + pub fn id(&self) -> u64 { + self.id + } + /// Transfer the result once. Cancellation before readiness preserves it in + /// the service. Physical resources move before their worker credit releases. + pub async fn wait(&self) -> Result> { + let mut ready = self.slot.ready.subscribe(); + while !*ready.borrow_and_update() { + if ready.changed().await.is_err() { + return Err(Arc::new(StagingError::Worker)); + } + } + let result = self + .slot + .result + .lock() + .expect("staging result") + .take() + .ok_or_else(|| Arc::new(StagingError::NotReady))?; + self.job + .work + .lock() + .expect("staging work") + .slots + .remove(&self.id); + self.slot.guard.lock().expect("staging guard").take(); + result + } +} + +async fn probe(job: &Job, minimum: Receipt) -> Result<(StagingLease, Instant), StagingError> { + let started = Instant::now(); + let token = match job.local.lock().expect("staging local").lease { + Some(l) => l.token, + None => return Err(StagingError::Context), + }; + let lease = job + .client + .query::( + &job.target, + Some(minimum), + LeaseCheck { + token, + actor: job.actor.clone(), + }, + ) + .await + .map_err(|e| StagingError::Query(Box::new(e)))? + .output + .ok_or(StagingError::Inactive)?; + if lease.token != token + || lease.observed_at_ms < 0 + || lease.expires_at_ms <= lease.observed_at_ms + { + return Err(StagingError::Context); + } + let deadline = started + + Duration::from_millis((lease.expires_at_ms - lease.observed_at_ms) as u64) + .min(Duration::from_millis(MAX_LEASE_MS)); + if deadline <= Instant::now() { + return Err(StagingError::Inactive); + } + Ok((lease, deadline)) +} +async fn supervise(inner: Arc, job: Arc) { + let mut recover = false; + loop { + let task = tokio::spawn(run(Arc::clone(&inner), Arc::clone(&job), recover)); + if task.await.is_ok() { + return; + } + { + let mut local = job.local.lock().expect("staging local"); + local.fenced = true; + if let Some(session) = &local.bound { + session.fence(); + } + } + let exact = job.exact.lock().expect("staging exact").clone(); + let Some(exact) = exact else { + let publication = job.publication.lock().expect("staging publication").clone(); + if let Some(ticket) = publication + && !matches!( + ticket.state(), + PublicationState::Held | PublicationState::Discarded + ) + { + // The other coordinator owns execution and exact evidence. + // Observe it even if this supervisor lost its local authority. + publication::observe(&inner, &job, &ticket).await; + return; + } + fence_and_drain(&inner, &job, StagingError::Worker).await; + return; + }; + job.status + .send_replace(StagingState::Uncertain(Arc::new(exact.pending()))); + inner.drained.notify_waiters(); + await_recovery(&job).await; + recover = true; + } +} +async fn await_recovery(job: &Job) { + loop { + let wake = job.changed.notified(); + tokio::pin!(wake); + wake.as_mut().enable(); + let ready = { + let mut l = job.local.lock().expect("staging local"); + if (l.lease.is_some() || l.bound.is_some()) + && Instant::now() >= l.deadline.min(l.lifetime) + { + l.fenced = true; + } + let ready = l.recovery; + l.recovery = false; + ready + }; + if ready { + return; + } + wake.await; + } +} +async fn run(inner: Arc, job: Arc, mut recover: bool) { + loop { + let publication = job.publication.lock().expect("staging publication").clone(); + if let Some(ticket) = publication + && !matches!( + ticket.state(), + PublicationState::Held | PublicationState::Discarded + ) + { + publication::observe(&inner, &job, &ticket).await; + return; + } + let command = job.exact.lock().expect("staging exact").clone(); + if let Some(command) = command { + if recover { + job.status.send_replace(StagingState::Resolving); + } + #[cfg(test)] + let fault = inner.fault.swap(0, std::sync::atomic::Ordering::AcqRel); + #[cfg(not(test))] + let fault = 0; + let pending = command.pending(); + let result = tokio::spawn(command.execute(job.client.clone(), recover, fault)) + .await + .unwrap_or(Err(pending)); + match result { + Err(error) if error.uncertain() => { + job.status + .send_replace(StagingState::Uncertain(Arc::new(error))); + inner.drained.notify_waiters(); + await_recovery(&job).await; + recover = true; + continue; + } + Err(error) => { + job.exact.lock().expect("staging exact").take(); + fence_and_drain(&inner, &job, error).await; + return; + } + Ok(Outcome::Bound(value)) => { + job.exact.lock().expect("staging exact").take(); + if !accept_bound(&inner, &job, value, false).await { + return; + } + recover = false; + } + Ok(Outcome::BoundClaim(value)) => { + job.exact.lock().expect("staging exact").take(); + if !accept_bound(&inner, &job, value, true).await { + return; + } + recover = false; + } + Ok(Outcome::BoundCheckpoint(value)) => { + job.exact.lock().expect("staging exact").take(); + let session = job + .local + .lock() + .expect("staging local") + .bound + .clone() + .expect("bound checkpoint session"); + let registration = job + .checkpoint + .lock() + .expect("staging checkpoint") + .clone() + .expect("bound checkpoint slot"); + registration.finish(Ok(value.receipt)); + let matched = matches!(&value.output, StagingReply::Granted(lease) + if lease.token == session.lease.token && lease.format == session.lease.format); + let result = if matched { + super::inputs::observe_bound_registration( + &session, + registration.bound_digest.expect("bound checkpoint digest"), + value.receipt, + ) + .await + } else { + Err(PreparationBaseError::Context.into()) + }; + if let Err(error) = result { + fence_and_drain(&inner, &job, StagingError::Input(Box::new(error))).await; + return; + } + let mut local = job.local.lock().expect("staging local"); + local.deadline = *session.deadline.lock().expect("bound deadline"); + job.status.send_replace(bound_state(&local)); + recover = false; + } + Ok(Outcome::BoundRenew(value)) => { + job.exact.lock().expect("staging exact").take(); + let session = { + let mut local = job.local.lock().expect("staging local"); + local.bound_renewal = Some(value.clone()); + local.bound.clone().expect("bound renewal session") + }; + let matched = matches!(&value.output, PreparationReply::Granted(lease) + if lease.token == session.lease.token && lease.base == session.lease.base && lease.format == session.lease.format); + let result = if matched { + session.refresh(value.receipt).await + } else { + Err(PreparationBaseError::Context) + }; + if let Err(error) = result { + fence_and_drain(&inner, &job, StagingError::Base(error)).await; + return; + } + let (_, _) = match session.live_lease() { + Ok(lease) => lease, + Err(error) => { + fence_and_drain(&inner, &job, StagingError::Base(error)).await; + return; + } + }; + let mut local = job.local.lock().expect("staging local"); + local.deadline = *session.deadline.lock().expect("bound deadline"); + job.status.send_replace(bound_state(&local)); + recover = false; + } + Ok(outcome @ (Outcome::Stage(_) | Outcome::Checkpoint(_))) => { + let (value, checkpoint) = match outcome { + Outcome::Stage(value) => (value, false), + Outcome::Checkpoint(value) => (value, true), + Outcome::Bound(_) + | Outcome::BoundClaim(_) + | Outcome::BoundRenew(_) + | Outcome::BoundCheckpoint(_) => { + unreachable!() + } + }; + let StagingReply::Granted(lease) = value.output else { + job.exact.lock().expect("staging exact").take(); + fence_and_drain(&inner, &job, StagingError::Context).await; + return; + }; + if checkpoint { + let expected = job.local.lock().expect("staging local").lease; + if !expected.is_some_and(|old| { + old.token == lease.token && old.format == lease.format + }) { + fence_and_drain(&inner, &job, StagingError::Context).await; + return; + } + job.checkpoint + .lock() + .expect("staging checkpoint") + .as_ref() + .expect("accepted input checkpoint") + .finish(Ok(value.receipt)); + } + job.exact.lock().expect("staging exact").take(); + { + let mut l = job.local.lock().expect("staging local"); + if l.lease.is_none() { + l.lease = Some(*lease); + } + } + match probe(&job, value.receipt).await { + Ok((lease, deadline)) => { + let differs = job + .local + .lock() + .expect("staging local") + .lease + .is_some_and(|old| { + old.token != lease.token || old.format != lease.format + }); + if differs { + fence_and_drain(&inner, &job, StagingError::Context).await; + return; + } + let mut l = job.local.lock().expect("staging local"); + l.lease = Some(lease); + l.deadline = deadline; + job.status.send_replace(if l.seal || l.stop { + StagingState::Draining(lease) + } else { + StagingState::Active(lease) + }); + } + Err(error) => { + fence_and_drain(&inner, &job, error).await; + return; + } + } + recover = false; + } + } + } + enum Next { + Stop, + Fence, + Bind(LeaseCheck), + Renew(LeaseCheck), + BoundRenew(LeaseCheck), + Checkpoint(Arc), + Publish, + Wait(Instant), + } + let wake = job.changed.notified(); + tokio::pin!(wake); + wake.as_mut().enable(); + let next = { + let mut l = job.local.lock().expect("staging local"); + let token = l + .bound + .as_ref() + .map(|s| s.lease.token) + .or_else(|| l.lease.map(|s| s.token)) + .expect("active preparation lease"); + let check = LeaseCheck { + token, + actor: job.actor.clone(), + }; + let now = Instant::now(); + let due = l + .deadline + .checked_sub(Duration::from_millis(inner.limits.renew_before_ms)) + .unwrap_or(now); + let checkpoint = job + .checkpoint + .lock() + .expect("staging checkpoint") + .as_ref() + .filter(|slot| { + slot.request + .lock() + .expect("staging checkpoint request") + .is_some() + }) + .cloned(); + if l.fenced || now >= l.deadline.min(l.lifetime) { + Next::Fence + } else if l.stop && !l.finishing && l.workers == 0 && checkpoint.is_none() { + Next::Stop + } else if l.bound.is_none() + && l.seal + && l.workers == 0 + && !l.stop + && checkpoint.is_none() + { + Next::Bind(check) + } else if l.renew || now >= due { + l.renew = false; + if l.bound.is_some() { + Next::BoundRenew(check) + } else { + Next::Renew(check) + } + } else if let Some(registration) = checkpoint { + Next::Checkpoint(registration) + } else if l.finishing && l.workers == 0 { + Next::Publish + } else { + Next::Wait(due.min(l.lifetime)) + } + }; + match next { + Next::Publish => { + let ticket = job + .publication + .lock() + .expect("staging publication") + .clone() + .expect("accepted final publication"); + let (session, receipt) = { + let l = job.local.lock().expect("staging local"); + let mut receipt = l.bound_result.as_ref().expect("bound receipt").receipt; + if let Some(renewal) = &l.bound_renewal + && renewal.receipt.commit_sequence > receipt.commit_sequence + { + receipt = renewal.receipt; + } + if let Some(registration) = + job.checkpoint.lock().expect("staging checkpoint").as_ref() + && let Some(Ok(registered)) = registration.result.borrow().as_ref() + && registered.commit_sequence > receipt.commit_sequence + { + receipt = *registered; + } + (l.bound.clone().expect("bound publication session"), receipt) + }; + if let Err(error) = session.refresh(receipt).await { + fence_and_drain(&inner, &job, StagingError::Base(error)).await; + return; + } + if job.local.lock().expect("staging local").fenced { + fence_and_drain(&inner, &job, StagingError::Inactive).await; + return; + } + if let Err(error) = ticket.activate().await { + fence_and_drain(&inner, &job, StagingError::PublicationAdmission(error)).await; + return; + } + publication::observe(&inner, &job, &ticket).await; + return; + } + Next::Stop => { + let mut local = job.local.lock().expect("staging local"); + local.fenced = true; + if let Some(session) = &local.bound { + session.fence(); + } + // Keep the original binding receipt observable after graceful stop. + job.status.send_replace( + local + .bound_result + .clone() + .map(StagingState::Bound) + .unwrap_or(StagingState::Stopped), + ); + drop(local); + remove(&inner, &job); + return; + } + Next::Fence => { + fence_and_drain(&inner, &job, StagingError::Inactive).await; + return; + } + Next::Wait(deadline) => { + tokio::select! { _ = wake => {}, _ = sleep_until(deadline) => {} } + } + Next::Checkpoint(registration) => { + job.status.send_replace(StagingState::RegisteringInputs); + let (proof, identity) = registration + .request + .lock() + .expect("staging checkpoint request") + .take() + .expect("queued checkpoint"); + match job + .client + .prepare_command::(&job.target, identity, proof) + .await + { + Ok(command) => { + *job.exact.lock().expect("staging exact") = + Some(if registration.bound_digest.is_some() { + Exact::BoundCheckpoint(command) + } else { + Exact::Checkpoint(command) + }); + } + Err(error) => { + fence_and_drain(&inner, &job, StagingError::Checkpoint(Box::new(error))) + .await; + return; + } + } + } + Next::Bind(check) => { + job.local.lock().expect("staging local").bound_started = Some(Instant::now()); + job.status.send_replace(StagingState::Binding); + let identity = crate::server::mutation_identity().map_err(|_| StagingError::Clock); + let result = match identity { + Ok(id) => job + .client + .prepare_command::(&job.target, id, check) + .await + .map(Exact::Bind) + .map_err(|e| StagingError::Bind(Box::new(e))), + Err(e) => Err(e), + }; + match result { + Ok(c) => { + *job.exact.lock().expect("staging exact") = Some(c); + } + Err(e) => { + fence_and_drain(&inner, &job, e).await; + return; + } + } + } + Next::BoundRenew(check) => { + let identity = crate::server::mutation_identity().map_err(|_| StagingError::Clock); + let result = match identity { + Ok(id) => job + .client + .prepare_command::( + &job.target, + id, + LeaseRequest { + check, + lease_ms: inner.limits.lease_ms, + }, + ) + .await + .map(Exact::BoundRenew) + .map_err(|e| StagingError::BoundRenew(Box::new(e))), + Err(e) => Err(e), + }; + match result { + Ok(command) => { + *job.exact.lock().expect("staging exact") = Some(command); + } + Err(error) => { + fence_and_drain(&inner, &job, error).await; + return; + } + } + } + Next::Renew(check) => { + let identity = crate::server::mutation_identity().map_err(|_| StagingError::Clock); + let result = match identity { + Ok(id) => job + .client + .prepare_command::( + &job.target, + id, + LeaseRequest { + check, + lease_ms: inner.limits.lease_ms, + }, + ) + .await + .map(Exact::Renew) + .map_err(|e| StagingError::Renew(Box::new(e))), + Err(e) => Err(e), + }; + match result { + Ok(c) => { + *job.exact.lock().expect("staging exact") = Some(c); + } + Err(e) => { + fence_and_drain(&inner, &job, e).await; + return; + } + } + } + } + } +} +async fn fence_and_drain(inner: &Inner, job: &Job, error: StagingError) { + { + let mut local = job.local.lock().expect("staging local"); + local.fenced = true; + if let Some(session) = &local.bound { + session.fence(); + } + } + let publication = job.publication.lock().expect("staging publication").clone(); + if let Some(ticket) = publication { + match ticket.state() { + PublicationState::Held => { + if ticket.discard_held().await.is_err() + && !matches!(ticket.state(), PublicationState::Discarded) + { + // A service-internal activation raced fencing. Execution + // owns exact evidence now; preserve its original outcome. + publication::observe(inner, job, &ticket).await; + return; + } + } + PublicationState::Discarded => {} + _ => { + publication::observe(inner, job, &ticket).await; + return; + } + } + } + let error = Arc::new(error); + job.status + .send_replace(StagingState::Fenced(Arc::clone(&error))); + if let Some(registration) = job.checkpoint.lock().expect("staging checkpoint").as_ref() { + registration.finish(Err(error)); + } + drain_work(job).await; + remove(inner, job); +} +async fn drain_work(job: &Job) { + // Release the service's completed-result ownership. In-flight supervisors + // still own their slots until abort/join and retain their admission guards. + let slots = std::mem::take(&mut job.work.lock().expect("staging work").slots); + for slot in slots.values() { + slot.fence_completed(); + } + drop(slots); + loop { + let wake = job.changed.notified(); + tokio::pin!(wake); + wake.as_mut().enable(); + if job.local.lock().expect("staging local").workers == 0 { + return; + } + wake.await; + } +} +fn bound_state(local: &Local) -> StagingState { + if local.finishing { + StagingState::Finishing + } else { + StagingState::Bound(local.bound_result.clone().expect("bound original result")) + } +} +fn remove(inner: &Inner, job: &Job) { + let mut a = inner.admission.lock().expect("staging admission"); + a.jobs.remove(&job.operation); + let count = a.actors.get_mut(&job.actor).expect("staging actor"); + count.operations -= 1; + if count.operations == 0 { + a.actors.remove(&job.actor); + } + inner.drained.notify_waiters(); +} diff --git a/crates/canopy-server/src/packs/publication/staging_service/bound.rs b/crates/canopy-server/src/packs/publication/staging_service/bound.rs new file mode 100644 index 0000000..8aab4f1 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/staging_service/bound.rs @@ -0,0 +1,85 @@ +//! Bound handoff reuses the parent lifecycle job, admission and worker slots. +use super::*; + +pub(super) async fn accept_bound( + inner: &Inner, + job: &Job, + value: Committed, + claimed: bool, +) -> bool { + let PreparationReply::Granted(lease) = value.output else { + fence_and_drain(inner, job, StagingError::Context).await; + return false; + }; + let original = Arc::new(StagingBound { + lease: *lease, + receipt: value.receipt, + }); + let (matched, ceiling) = { + let mut local = job.local.lock().expect("staging local"); + // A known result is retained before fresh probes, even after revocation. + local.bound_result = Some(original.clone()); + let matched = if claimed { + local.bound_source.as_ref().is_some_and(|source| { + source.token.repository == lease.token.repository + && source.token.operation == lease.token.operation + && source.token.request_digest == lease.token.request_digest + && source.token != lease.token + && source.actor == job.actor + }) + } else { + local.lease.is_some_and(|staged| { + staged.token == lease.token + && staged.format == lease.format + && lease.expires_at_ms >= staged.expires_at_ms + }) + }; + let started = local.bound_started.expect("bound command admission time"); + let ceiling = + (started + Duration::from_millis(inner.limits.bound_lifetime_ms)).min(local.lifetime); + local.lifetime = ceiling; + ( + matched && !local.fenced && Instant::now() < ceiling, + ceiling, + ) + }; + if !matched { + fence_and_drain(inner, job, StagingError::Context).await; + return false; + } + let session = PreparationSession::open( + job.client.clone(), + job.target.clone(), + LeaseCheck { + token: lease.token, + actor: job.actor.clone(), + }, + Some(value.receipt), + ) + .await; + let mut session = match session { + Ok(session) => session, + Err(error) => { + fence_and_drain(inner, job, StagingError::Base(error)).await; + return false; + } + }; + if session.lease.base != lease.base || session.lease.format != lease.format { + fence_and_drain(inner, job, StagingError::Context).await; + return false; + } + session.ceiling = Some(ceiling); + let _ = match session.live_lease() { + Ok((_, deadline)) => deadline, + Err(error) => { + fence_and_drain(inner, job, StagingError::Base(error)).await; + return false; + } + }; + let deadline = *session.deadline.lock().expect("bound deadline"); + let mut local = job.local.lock().expect("staging local"); + local.bound = Some(Arc::new(session)); + local.deadline = deadline; + job.status.send_replace(StagingState::Bound(original)); + true +} diff --git a/crates/canopy-server/src/packs/publication/staging_service/publication.rs b/crates/canopy-server/src/packs/publication/staging_service/publication.rs new file mode 100644 index 0000000..e1403ef --- /dev/null +++ b/crates/canopy-server/src/packs/publication/staging_service/publication.rs @@ -0,0 +1,151 @@ +//! The lifecycle retains only a small ticket. The fair coordinator owns and +//! charges the original private proof, command and exact transport recovery. +use super::*; + +pub struct StagedPublicationFailure { + pub reason: StagingError, + pub ready: ReadyPublication, +} +impl std::fmt::Debug for StagedPublicationFailure { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("StagedPublicationFailure") + .field("reason", &self.reason) + .finish_non_exhaustive() + } +} +impl std::fmt::Display for StagedPublicationFailure { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + self.reason.fmt(f) + } +} +impl std::error::Error for StagedPublicationFailure {} + +/// Observation cannot activate/discard the lifecycle's held final command. +#[derive(Clone)] +#[must_use] +pub struct StagedPublicationTicket { + ticket: PublicationTicket, +} +impl StagedPublicationTicket { + pub fn state(&self) -> PublicationState { + self.ticket.state() + } + pub async fn wait(&self) -> PublicationState { + self.ticket.wait().await + } + pub async fn response( + &self, + ) -> Result { + self.ticket.response().await + } +} +impl StagingTicket { + /// Seal the bound phase and synchronously transfer the exact final command + /// into fair dispatch admission. Existing workers/results, due renewal and + /// a queued checkpoint drain before activation. Returns the original ready + /// value on refusal. Call after retrieving the producer's typed result; + /// awaiting publication inside an owned producer would block its own drain. + pub fn publish( + &self, + coordinator: &PublicationCoordinator, + ready: impl Into, + ) -> Result> { + let ready = ready.into(); + let mut local = self.job.local.lock().expect("staging local"); + let reason = + if local.fenced || local.stop || Instant::now() >= local.deadline.min(local.lifetime) { + Some(StagingError::Inactive) + } else if local.finishing { + Some(StagingError::Duplicate) + } else if !matches!( + self.state(), + StagingState::Bound(_) | StagingState::RegisteringInputs + ) { + Some(StagingError::NotReady) + } else { + match local.bound.as_ref() { + Some(session) if session.live_lease().is_err() => Some(StagingError::Inactive), + Some(session) if ready.belongs_to(session) => None, + _ => Some(StagingError::Context), + } + }; + if let Some(reason) = reason { + return Err(Box::new(StagedPublicationFailure { reason, ready })); + } + let ticket = coordinator.try_reserve(ready).map_err(|failure| { + Box::new(StagedPublicationFailure { + reason: StagingError::PublicationAdmission(failure.reason), + ready: failure.ready, + }) + })?; + *self.job.publication.lock().expect("staging publication") = Some(ticket.clone()); + local.finishing = true; + self.job.status.send_replace(StagingState::Finishing); + self.job.changed.notify_one(); + Ok(StagedPublicationTicket { ticket }) + } + /// Retrieve an observer after caller cancellation, including after a known + /// final result removes the operation from lifecycle admission. + pub fn pending_publication(&self) -> Option { + self.job + .publication + .lock() + .expect("staging publication") + .clone() + .map(|ticket| StagedPublicationTicket { ticket }) + } +} + +pub(super) async fn observe(inner: &Inner, job: &Job, ticket: &PublicationTicket) { + loop { + job.status.send_replace(StagingState::Publishing); + match ticket.wait().await { + PublicationState::Uncertain(error) => { + job.status.send_replace(StagingState::Uncertain(Arc::new( + StagingError::Publication(error), + ))); + inner.drained.notify_waiters(); + tokio::select! { + _ = await_recovery(job) => {}, + _ = ticket.wait_recovered() => {}, + } + job.local.lock().expect("staging local").recovery = false; + // Another service-internal observer may already have scheduled + // exact recovery. Never replace or reconstruct that command. + if matches!(ticket.state(), PublicationState::Uncertain(_)) { + let _ = ticket.recover().await; + } + } + PublicationState::Finished(outcome) => { + { + let mut local = job.local.lock().expect("staging local"); + local.fenced = true; + if let Some(session) = &local.bound { + session.fence(); + } + } + drain_work(job).await; + job.status.send_replace(StagingState::Published(outcome)); + remove(inner, job); + return; + } + PublicationState::Discarded => { + // Only service-internal premature discard can reach this path; + // there is no durable final result to acknowledge. + { + let mut local = job.local.lock().expect("staging local"); + local.fenced = true; + if let Some(session) = &local.bound { + session.fence(); + } + } + drain_work(job).await; + job.status + .send_replace(StagingState::Fenced(Arc::new(StagingError::Inactive))); + remove(inner, job); + return; + } + _ => unreachable!("publication wait observes an outcome"), + } + } +} diff --git a/crates/canopy-server/src/packs/publication/tests.rs b/crates/canopy-server/src/packs/publication/tests.rs new file mode 100644 index 0000000..48df6f8 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests.rs @@ -0,0 +1,1169 @@ +use super::*; +mod attestation; +mod compaction; +mod completion; +mod coordinator; +mod frontier; +mod initialization; +mod inputs; +mod namespaces; +mod native_capture; +mod prepare; +mod publishing; +mod reconcile; +mod ref_policy; +mod ref_snapshot; +mod refs; +mod root_completion; +mod staging; +mod staging_service; +use cellule_ltx::{CellReplica, CellStorageLayout, Limits}; +use cellule_runtime::{ + ApplicationId, BuildDescriptor, CellClient, CellRuntime, CellTarget, Digest, InvocationError, + MigrationDescriptor, ModuleDescriptor, MutationIdentity, NamespaceDescriptor, Registry, + SessionId, SqlWorkerPool, TenantId, + cell::{ + actor::CellHandle, + catalog::{CatalogEntry, CatalogRole, CellCatalog}, + }, + control::{Owner, authority::CellAuthority}, + identity::RequestId, +}; +use cellule_store::Store; +use object_store::{memory::InMemory, path::Path}; +use std::{ + sync::{Arc, OnceLock}, + time::{SystemTime, UNIX_EPOCH}, +}; +type Result = std::result::Result>; +struct Module; +impl CellModule for Module { + const NAME: &'static str = "repository"; + fn descriptor(&self) -> &'static ModuleDescriptor { + static DESCRIPTOR: OnceLock = OnceLock::new(); + DESCRIPTOR.get_or_init(|| { + let descriptor = |id| cellule_runtime::registry::OperationDescriptor { + id, + codec_version: 1, + schema_min: 1, + schema_max: 1, + input_limit: 4096, + output_limit: 4096, + }; + let mut ref_descriptor = descriptor(90); + ref_descriptor.input_limit = 64 << 10; + let mut publish_descriptor = descriptor(18); + publish_descriptor.input_limit = 4 << 20; + let mut complete_descriptor = descriptor(19); + complete_descriptor.input_limit = 4 << 20; + let mut initial_descriptor = descriptor(31); + initial_descriptor.input_limit = INITIALIZATION_BYTES; + initial_descriptor.output_limit = 512; + let mut initial_query = descriptor(32); + initial_query.output_limit = 512; + let mut policy_page = descriptor(33); + policy_page.input_limit = REF_POLICY_PAGE_BYTES; + policy_page.output_limit = 128; + let mut policy_query = descriptor(34); + policy_query.output_limit = 128; + let mut policy_reap = descriptor(35); + policy_reap.output_limit = 128; + // Match the existing production SQL transport contract exactly. + // The generic 4 KiB fixture limit cannot encode even one valid + // 65 KiB ref name; policy construction has its own smaller bound. + let sql_query = crate::operation(2); + ModuleDescriptor { + name: Self::NAME, + source_digest: Digest::from_bytes([11; 32]), + retained_codes: &[], + schema_min: 1, + schema_max: 1, + migrations: Box::leak(Box::new([MigrationDescriptor { + version: 1, + sql: SCHEMA, + digest: Digest::from_bytes(*blake3::hash(SCHEMA.as_bytes()).as_bytes()), + }])), + commands: Box::leak(Box::new([ + descriptor(11), + descriptor(12), + descriptor(13), + descriptor(14), + descriptor(16), + descriptor(17), + publish_descriptor, + complete_descriptor, + descriptor(22), + descriptor(24), + descriptor(25), + descriptor(26), + descriptor(28), + descriptor(29), + initial_descriptor, + policy_page, + policy_reap, + ref_descriptor, + ])), + queries: Box::leak(Box::new([ + sql_query, + descriptor(15), + descriptor(20), + descriptor(21), + descriptor(23), + descriptor(27), + descriptor(30), + initial_query, + policy_query, + ])), + workflow_definitions: &[], + activity_types: &[], + namespaces: Box::leak(Box::new([NamespaceDescriptor { + id: crate::REPOSITORIES, + name: "repository", + role: CatalogRole::Sql, + shards: 1, + effect_targets: &[], + dead_letter: None, + }])), + } + }) + } + fn register(self, registry: &mut RegistryBuilder) -> cellule_runtime::Result<()> { + super::register(registry)?; + registry.bind_command::()?; + registry.bind_query::>() + } +} +struct Fixture { + root: tempfile::TempDir, + target: CellTarget, + repository: [u8; 16], + format: ObjectFormat, + layout: CellStorageLayout, + replica: CellReplica, + registry: Arc, + runtime: CellRuntime, + handle: CellHandle, +} +impl Fixture { + async fn new(format: ObjectFormat) -> Result { + Self::with_artifact_sequence(format, 0).await + } + async fn with_artifact_sequence(format: ObjectFormat, artifact_sequence: i64) -> Result { + let mut builder = RegistryBuilder::new(BuildDescriptor { + source_revision: "packed-publication-test".into(), + cargo_lock_digest: Digest::from_bytes([10; 32]), + }); + builder.register(Module)?; + let registry = Arc::new(builder.finish()?); + let repository = *uuid::Uuid::new_v4().as_bytes(); + let tenant = TenantId::from_bytes([12; 16]); + let application = ApplicationId::from_bytes([13; 16]); + let target = crate::repository_target(tenant, application, repository)?; + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + Path::from("packed-publication"), + *application.as_bytes(), + ); + let incarnation = IncarnationId::from_bytes([14; 16]); + let replica = CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + Limits::default(), + )?; + let session = SessionId::from_bytes([15; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, session)?; + let catalog = CellCatalog::new(layout.clone(), tenant); + let proof = catalog + .provision(CatalogEntry::new( + &target, + CatalogRole::Sql, + registry.module_code("repository").ok_or("code")?, + 1, + )?) + .await?; + let authority = CellAuthority::new(layout.clone()); + let control = authority + .create_initial( + &proof, + incarnation, + Owner { + session, + endpoint: "https://owner-a.invalid".into(), + }, + ) + .await?; + let root = tempfile::TempDir::new()?; + let handle=runtime.bootstrap(proof,replica.clone(),authority,control,root.path().join("a.sqlite"),move|tx|{ + tx.execute_batch(SCHEMA)?; + tx.execute("INSERT INTO repository_identity(singleton,repository_id,object_format,owner,push_cert_seed,artifact_sequence) VALUES(1,?1,?2,'owner',?3,?4)",rusqlite::params![repository.as_slice(),format.as_str(),[16u8;32].as_slice(),artifact_sequence])?;Ok(()) + }).await?; + Ok(Self { + root, + target, + repository, + format, + layout, + replica, + registry, + runtime, + handle, + }) + } + fn client(&self) -> CellClient { + CellClient::local(Arc::clone(&self.registry), self.handle.clone()) + } + fn begin(&self, operation: [u8; 16]) -> BeginRequest { + BeginRequest { + repository: self.repository, + operation, + request_digest: [17; 32], + actor: "owner".into(), + lease_ms: DEFAULT_LEASE_MS, + } + } + async fn counts(&self) -> Result<(u64, u64)> { + counts(&self.handle).await + } + // Trusted fixture injection only. Production generation facts require the + // complete catalog verifier and fenced publisher; a digest is not a proof. + async fn install_catalog(&self, generation: u64, catalog: StoredCatalog) -> Result<()> { + self.install_generation(generation, catalog, None).await + } + async fn install_generation( + &self, + generation: u64, + catalog: StoredCatalog, + refs: Option, + ) -> Result<()> { + let mut encoder = BoundedEncoder::new(256)?; + catalog.encode(&mut encoder)?; + let bytes = encoder.finish(); + let refs = refs + .map(|refs| { + let mut e = BoundedEncoder::new(128)?; + refs.encode(&mut e)?; + Ok::<_, CodecError>(e.finish()) + }) + .transpose()?; + self.handle + .execute( + identity()?, + Digest::from_bytes([64; 32]), + sql::now(0)?, + bytes.len(), + 0, + move |tx| { + tx.execute( + "INSERT INTO catalog_generations(generation,catalog,certificate,refs) VALUES(?1,?2,?3,?4)", + rusqlite::params![generation as i64, bytes, [42u8; 32].as_slice(), refs], + )?; + tx.execute( + "UPDATE catalog_state SET generation=?1 WHERE singleton=1", + [generation as i64], + )?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + Ok(()) + } + async fn install_empty_root(&self, generation: u64) -> Result { + use crate::packs::{catalog::CatalogSnapshot, directory::snapshot::DirectorySnapshot}; + use canopy_object_storage::artifact::ArtifactStore; + let store = ArtifactStore::new(Arc::new(InMemory::new()), self.repository); + let directory = DirectorySnapshot::empty(self.repository, self.format) + .upload(&store, [generation as u8 + 40; 16]) + .await?; + let catalog = CatalogSnapshot { + directory, + sources: None, + } + .upload(&store, [generation as u8 + 50; 16]) + .await?; + self.install_catalog(generation, catalog).await?; + Ok(catalog) + } +} +fn identity() -> Result { + let now = i64::try_from(SystemTime::now().duration_since(UNIX_EPOCH)?.as_millis())?; + Ok(MutationIdentity { + request_id: RequestId::from_bytes(uuid::Uuid::new_v4().into_bytes()), + issued_at_ms: now, + expires_at_ms: now + 60_000, + }) +} +fn lease(reply: PreparationReply) -> Result { + match reply { + PreparationReply::Granted(lease) => Ok(*lease), + PreparationReply::Denied(reason) => Err(format!("denied: {reason:?}").into()), + } +} +fn check(token: PreparationToken) -> LeaseCheck { + LeaseCheck { + token, + actor: "owner".into(), + } +} +fn request(token: PreparationToken) -> LeaseRequest { + LeaseRequest { + check: check(token), + lease_ms: DEFAULT_LEASE_MS, + } +} +fn rejected( + result: std::result::Result< + cellule_runtime::Committed, + InvocationError, + >, + reason: PreparationDenial, +) { + assert!( + matches!(result,Err(InvocationError::Rejected(ref outcome)) if outcome.output==PreparationReply::Denied(reason)) + ); +} +async fn counts(handle: &CellHandle) -> Result<(u64, u64)> { + let bytes = handle + .query(0, 16, |connection| { + let operations: u64 = + connection.query_row("SELECT count(*) FROM catalog_operations", [], |row| { + row.get(0) + })?; + let leases: u64 = + connection + .query_row("SELECT count(*) FROM catalog_leases", [], |row| row.get(0))?; + Ok([operations.to_be_bytes(), leases.to_be_bytes()].concat()) + }) + .await?; + Ok(( + u64::from_be_bytes(bytes[..8].try_into()?), + u64::from_be_bytes(bytes[8..].try_into()?), + )) +} + +#[tokio::test] +async fn preparation_replay_renewal_abort_and_record_recreation_keep_attempt_identity() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let client = fixture.client(); + let input = fixture.begin([20; 16]); + let mutation = identity()?; + let first = client + .command::(&fixture.target, mutation, input.clone()) + .await?; + let started = lease(first.output.clone())?; + assert_eq!(started.token.owner, fixture.handle.owner_fence()); + assert_eq!(started.token.artifact_operation, artifact_number(1)); + assert_eq!(started.base.generation, 0); + assert!(started.base.catalog.is_none()); + let replay = client + .command::(&fixture.target, mutation, input.clone()) + .await?; + assert_eq!(replay.output, first.output); + assert_eq!(replay.receipt, first.receipt); + assert_eq!(fixture.counts().await?, (1, 1)); + let duplicate = lease( + client + .command::(&fixture.target, identity()?, input.clone()) + .await? + .output, + )?; + assert_eq!(duplicate.token, started.token); + assert_eq!(duplicate.expires_at_ms, started.expires_at_ms); + let mut conflict = input.clone(); + conflict.request_digest[0] ^= 1; + rejected( + client + .command::(&fixture.target, identity()?, conflict) + .await, + PreparationDenial::Conflict, + ); + let mut renewal = request(started.token); + renewal.lease_ms = 1; + let renewed = lease( + client + .command::(&fixture.target, identity()?, renewal) + .await? + .output, + )?; + assert_eq!(renewed.token, started.token); + assert!(renewed.expires_at_ms >= started.expires_at_ms); + assert!( + client + .query::(&fixture.target, None, check(started.token)) + .await? + .output + .is_some() + ); + assert!( + client + .command::(&fixture.target, identity()?, check(started.token)) + .await? + .output + ); + assert_eq!(fixture.counts().await?, (0, 1)); + assert!( + client + .query::(&fixture.target, None, check(started.token)) + .await? + .output + .is_none() + ); + let next = lease( + client + .command::(&fixture.target, identity()?, input) + .await? + .output, + )?; + assert_eq!(next.token.owner, started.token.owner); + assert!(next.token.attempt > started.token.attempt); + assert_eq!(next.token.artifact_operation, artifact_number(2)); + rejected( + client + .command::(&fixture.target, identity()?, request(started.token)) + .await, + PreparationDenial::Stale, + ); + rejected( + client + .command::(&fixture.target, identity()?, request(started.token)) + .await, + PreparationDenial::Stale, + ); + assert_eq!(fixture.counts().await?, (1, 2)); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn successor_claim_rejects_the_old_owner_and_preserves_exact_replayed_result() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let input = fixture.begin([21; 16]); + let mutation = identity()?; + let first = fixture + .client() + .command::(&fixture.target, mutation, input.clone()) + .await?; + let old = lease(first.output.clone())?; + fixture.handle.drain().await?; + fixture.runtime.shutdown().await?; + let session = SessionId::from_bytes([22; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, session)?; + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await? + .ok_or("idle")?; + let proof = CellCatalog::new(fixture.layout.clone(), fixture.target.tenant()) + .lookup(fixture.target.cell_id()) + .await? + .ok_or("proof")?; + let handle = runtime + .acquire_idle_restored( + proof, + fixture.replica.clone(), + authority, + idle, + fixture.root.path().join("b.sqlite"), + Owner { + session, + endpoint: "https://owner-b.invalid".into(), + }, + ) + .await?; + let client = CellClient::local(Arc::clone(&fixture.registry), handle.clone()); + assert!(handle.owner_fence().epoch > old.token.owner.epoch); + let replay = client + .command::(&fixture.target, mutation, input) + .await?; + assert_eq!(replay.output, first.output); + assert_eq!(replay.receipt, first.receipt); + rejected( + client + .command::(&fixture.target, identity()?, request(old.token)) + .await, + PreparationDenial::Stale, + ); + let claimed = lease( + client + .command::(&fixture.target, identity()?, request(old.token)) + .await? + .output, + )?; + assert_eq!(claimed.token.owner, handle.owner_fence()); + assert_eq!(old.token.artifact_operation, artifact_number(1)); + assert_eq!(claimed.token.artifact_operation, artifact_number(2)); + assert!(claimed.token.attempt > old.token.attempt); + assert_eq!(counts(&handle).await?, (1, 2)); + assert!( + client + .query::(&fixture.target, None, check(old.token)) + .await? + .output + .is_none() + ); + assert!( + client + .query::(&fixture.target, None, check(claimed.token)) + .await? + .output + .is_some() + ); + let mut wrong = request(claimed.token); + wrong.check.token.owner.incarnation = IncarnationId::from_bytes([99; 16]); + rejected( + client + .command::(&fixture.target, identity()?, wrong) + .await, + PreparationDenial::Stale, + ); + rejected( + client + .command::(&fixture.target, identity()?, request(old.token)) + .await, + PreparationDenial::Stale, + ); + runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn rebase_claim_keeps_the_previous_generation_pinned_and_revocation_stops_renewal() -> Result +{ + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let first_catalog = fixture.install_empty_root(1).await?; + let client = fixture.client(); + let first = lease( + client + .command::(&fixture.target, identity()?, fixture.begin([23; 16])) + .await? + .output, + )?; + assert_eq!(first.base.catalog, Some(first_catalog)); + let second_catalog = fixture.install_empty_root(2).await?; + let next = lease( + client + .command::(&fixture.target, identity()?, request(first.token)) + .await? + .output, + )?; + assert_eq!(next.base.generation, 2); + assert_eq!(next.base.catalog, Some(second_catalog)); + assert_eq!(fixture.counts().await?, (1, 2)); + // The old generation is neither current nor the operation's current base, + // but its independent lease still forbids deleting its authoritative fact. + assert!( + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([64; 32]), + sql::now(0)?, + 1, + 0, + |tx| { + tx.execute("DELETE FROM catalog_generations WHERE generation=1", [])?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + } + ) + .await + .is_err() + ); + let empty = fixture + .handle + .query(0, 8, |connection| { + let count: u64 = connection.query_row( + "SELECT count(*) FROM catalog_leases WHERE generation=1", + [], + |row| row.get(0), + )?; + Ok(count.to_be_bytes().to_vec()) + }) + .await?; + assert_eq!( + u64::from_be_bytes(empty.try_into().map_err(|_| "count")?), + 1 + ); + let maintenance = MaintenanceRequest { + repository: fixture.repository, + actor: "owner".into(), + owner: fixture.handle.owner_fence(), + }; + assert_eq!( + client + .command::(&fixture.target, identity()?, maintenance.clone()) + .await? + .output, + 0 + ); + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([64; 32]), + sql::now(0)?, + 1, + 0, + |tx| { + tx.execute( + "INSERT INTO repository_members(account,role) VALUES('writer','write')", + [], + )?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + let mut input = fixture.begin([24; 16]); + input.actor = "writer".into(); + let writer = lease( + client + .command::(&fixture.target, identity()?, input) + .await? + .output, + )?; + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([64; 32]), + sql::now(0)?, + 1, + 0, + |tx| { + tx.execute("DELETE FROM repository_members WHERE account='writer'", [])?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + let mut renew = request(writer.token); + renew.check.actor = "writer".into(); + rejected( + client + .command::(&fixture.target, identity()?, renew) + .await, + PreparationDenial::Unauthorized, + ); + let probe = LeaseCheck { + token: writer.token, + actor: "writer".into(), + }; + assert!( + client + .query::(&fixture.target, None, probe) + .await? + .output + .is_none() + ); + // Expire just the superseded pin. The live operation and its new base + // remain intact; the reaper may now remove only the obsolete SQL fact. + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([65; 32]), + sql::now(0)?, + 128, + 0, + |tx| { + tx.execute( + "UPDATE catalog_leases SET expires_at_ms=0 WHERE generation=1", + [], + )?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + assert_eq!( + client + .command::(&fixture.target, identity()?, maintenance) + .await? + .output, + 2 + ); + let generations = fixture + .handle + .query(0, 16, |connection| { + let old: u64 = connection.query_row( + "SELECT count(*) FROM catalog_generations WHERE generation=1", + [], + |row| row.get(0), + )?; + let current: u64 = connection.query_row( + "SELECT generation FROM catalog_state WHERE singleton=1", + [], + |row| row.get(0), + )?; + Ok([old.to_be_bytes(), current.to_be_bytes()].concat()) + }) + .await?; + assert_eq!( + generations, + [0u64.to_be_bytes(), 2u64.to_be_bytes()].concat() + ); + assert!( + client + .query::(&fixture.target, None, check(next.token)) + .await? + .output + .is_some() + ); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn expired_attempts_cannot_renew_and_reaping_respects_bounded_indexed_work() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let client = fixture.client(); + let mut original_id = [0; 16]; + original_id[15] = 1; + let started = lease( + client + .command::(&fixture.target, identity()?, fixture.begin(original_id)) + .await? + .output, + )?; + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([64; 32]), + sql::now(0)?, + 1, + 0, + |tx| { + tx.execute("UPDATE catalog_operations SET expires_at_ms=0", [])?; + tx.execute("UPDATE catalog_leases SET expires_at_ms=0", [])?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + rejected( + client + .command::(&fixture.target, identity()?, request(started.token)) + .await, + PreparationDenial::Expired, + ); + assert!( + client + .query::(&fixture.target, None, check(started.token)) + .await? + .output + .is_none() + ); + let owner = fixture.handle.owner_fence(); + // Fixture preparation is bounded too: reuse parsed statements and avoid a + // monolithic setup competing with unrelated native verifier test workers. + // The full quota and the publisher/reaper deadlines below remain unchanged. + for first in (1..MAX_OPERATIONS).step_by(REAP_ROWS as usize) { + let last = (first + REAP_ROWS).min(MAX_OPERATIONS); + fixture.handle.execute( + identity()?, Digest::from_bytes([65; 32]), sql::now(0)?, 128, 0, + move |tx| { + let mut pin = tx.prepare("INSERT INTO catalog_leases(incarnation,admission_sequence,operation,owner_epoch,artifact_operation,generation,expires_at_ms) VALUES(?1,?2,?3,?4,?5,0,0)")?; + let mut operation = tx.prepare("INSERT INTO catalog_operations(id,actor,request_digest,incarnation,owner_epoch,admission_sequence,artifact_operation,generation,expires_at_ms) VALUES(?1,'owner',?2,?3,?4,?5,?6,0,0)")?; + for n in first..last { + let mut id = [0u8; 16]; + id[..8].copy_from_slice(&n.to_be_bytes()); + let seq = 1_000_000 + n as i64; + pin.execute(rusqlite::params![owner.incarnation.as_bytes().as_slice(), seq, id.as_slice(), owner.epoch.to_be_bytes().as_slice(), artifact_number(seq as u64).as_slice()])?; + operation.execute(rusqlite::params![id.as_slice(), [17u8; 32].as_slice(), owner.incarnation.as_bytes().as_slice(), owner.epoch.to_be_bytes().as_slice(), seq, artifact_number(seq as u64).as_slice()])?; + } + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success(Vec::new())) + }, + ).await?; + } + rejected( + client + .command::(&fixture.target, identity()?, fixture.begin([26; 16])) + .await, + PreparationDenial::Capacity, + ); + assert_eq!(fixture.counts().await?, (MAX_OPERATIONS, MAX_OPERATIONS)); + let removed = client + .command::( + &fixture.target, + identity()?, + MaintenanceRequest { + repository: fixture.repository, + actor: "owner".into(), + owner, + }, + ) + .await? + .output; + assert_eq!(removed, REAP_ROWS * 2); + assert_eq!( + fixture.counts().await?, + (MAX_OPERATIONS - REAP_ROWS, MAX_OPERATIONS - REAP_ROWS) + ); + let fresh = lease( + client + .command::(&fixture.target, identity()?, fixture.begin(original_id)) + .await? + .output, + )?; + assert!(fresh.token.attempt > started.token.attempt); + let plans=fixture.handle.query(0,4096,|connection|{ + let mut text=String::new();for sql in ["SELECT id FROM catalog_operations WHERE expires_at_ms<=0 ORDER BY expires_at_ms,id LIMIT 512","SELECT l.incarnation,l.admission_sequence FROM catalog_leases l WHERE l.expires_at_ms<=0 AND NOT EXISTS(SELECT 1 FROM catalog_operations o WHERE o.incarnation=l.incarnation AND o.admission_sequence=l.admission_sequence) ORDER BY l.expires_at_ms,l.incarnation,l.admission_sequence LIMIT 512"] { + let mut statement=connection.prepare(&format!("EXPLAIN QUERY PLAN {sql}"))?;for row in statement.query_map([],|row|row.get::<_,String>(3))? {text.push_str(&row?);text.push('\n');} + }Ok(text.into_bytes()) + }).await?; + let plans = String::from_utf8(plans)?; + assert!( + plans.contains("catalog_operations_by_expiry") + && plans.contains("catalog_leases_by_expiry") + && plans.contains("catalog_operations_by_lease") + ); + assert!(!plans.contains("TEMP B-TREE")); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[test] +fn fresh_schema_and_codecs_reject_incomplete_facts_and_support_full_owner_epochs() -> Result { + let mut connection = rusqlite::Connection::open_in_memory()?; + connection.execute_batch("PRAGMA foreign_keys=ON;")?; + connection.execute_batch(SCHEMA)?; + let legacy:u64=connection.query_row("SELECT count(*) FROM sqlite_schema WHERE name IN ('objects','object_edges','object_closure','object_uploads','object_chunks','git_packs','commit_parents','commit_ancestry')",[],|row|row.get(0))?; + assert_eq!(legacy, 0); + assert!( + connection + .execute( + "INSERT INTO catalog_generations(generation,catalog,certificate) VALUES(1,NULL,zeroblob(32))", + [] + ) + .is_err() + ); + assert!( + connection + .execute( + "INSERT INTO catalog_generations(generation,catalog,certificate) VALUES(1,zeroblob(1),NULL)", + [] + ) + .is_err() + ); + assert!( + connection + .execute( + "UPDATE catalog_generations SET certificate=zeroblob(32) WHERE generation=0", + [] + ) + .is_err() + ); + for refs in [ + SqlValue::Text("x".into()), + SqlValue::Blob(Vec::new()), + SqlValue::Blob(vec![0; 129]), + ] { + let value = match refs { + SqlValue::Text(value) => rusqlite::types::Value::Text(value), + SqlValue::Blob(value) => rusqlite::types::Value::Blob(value), + _ => unreachable!(), + }; + assert!(connection.execute("INSERT INTO catalog_generations(generation,catalog,certificate,refs) VALUES(1,zeroblob(1),zeroblob(32),?1)", [value]).is_err()); + } + connection.execute( + "INSERT INTO catalog_generations(generation,catalog,certificate) VALUES(1,zeroblob(1),zeroblob(32))", + [], + )?; + connection.execute( + "INSERT INTO catalog_leases(incarnation,admission_sequence,operation,owner_epoch,artifact_operation,generation,expires_at_ms) VALUES(zeroblob(16),1,zeroblob(16),x'0000000000000001',x'43414e4f505930310000000000000001',0,100)", + [], + )?; + let inconsistent = connection.transaction()?; + inconsistent.execute("INSERT INTO catalog_operations(id,actor,request_digest,incarnation,owner_epoch,admission_sequence,artifact_operation,generation,expires_at_ms) VALUES(zeroblob(16),'owner',zeroblob(32),zeroblob(16),x'0000000000000001',1,x'43414e4f505930310000000000000001',1,100)",[])?; + assert!(inconsistent.commit().is_err()); + // Retained SQL roots cannot grow with the complete publication history. + // These opaque bytes deliberately test SQL constraints, not certification. + let roots = connection.transaction()?; + for generation in 2..MAX_RETAINED_GENERATIONS { + roots.execute( + "INSERT INTO catalog_generations(generation,catalog,certificate) VALUES(?1,zeroblob(1),zeroblob(32))", + [generation as i64], + )?; + } + roots.commit()?; + assert!( + connection + .execute( + "INSERT INTO catalog_generations(generation,catalog,certificate) VALUES(?1,zeroblob(1),zeroblob(32))", + [MAX_RETAINED_GENERATIONS as i64], + ) + .is_err() + ); + let retained: u64 = + connection.query_row("SELECT count(*) FROM catalog_generations", [], |row| { + row.get(0) + })?; + assert_eq!(retained, MAX_RETAINED_GENERATIONS); + // Even an expired independent floor protects every later fact until the + // pin itself is removed; reaching capacity cannot bypass that retention. + assert!( + connection + .execute("DELETE FROM catalog_generations WHERE generation=2", []) + .is_err() + ); + connection.execute("DELETE FROM catalog_leases", [])?; + connection.execute("DELETE FROM catalog_generations WHERE generation=2", [])?; + connection.execute( + "INSERT INTO catalog_generations(generation,catalog,certificate) VALUES(?1,zeroblob(1),zeroblob(32))", + [MAX_RETAINED_GENERATIONS as i64], + )?; + let token = PreparationToken { + repository: uuid::Uuid::new_v4().into_bytes(), + operation: [31; 16], + artifact_operation: artifact_number(1), + request_digest: [32; 32], + owner: OwnerFence { + incarnation: IncarnationId::from_bytes([33; 16]), + epoch: u64::MAX, + }, + attempt: 1, + }; + let mut encoder = BoundedEncoder::new(4096)?; + token.encode(&mut encoder)?; + let bytes = encoder.finish(); + let mut decoder = BoundedDecoder::new(&bytes, 4096)?; + assert_eq!(PreparationToken::decode(&mut decoder)?, token); + decoder.finish()?; + for length in 0..bytes.len() { + let mut decoder = BoundedDecoder::new(&bytes[..length], 4096)?; + assert!(PreparationToken::decode(&mut decoder).is_err()); + } + let mut bad = token; + bad.attempt = 0; + assert!(bad.encode(&mut BoundedEncoder::new(4096)?).is_err()); + bad = token; + bad.owner.epoch = 0; + assert!(bad.encode(&mut BoundedEncoder::new(4096)?).is_err()); + for namespace in [[0; 16], artifact_number(0), artifact_number(u64::MAX)] { + bad = token; + bad.artifact_operation = namespace; + assert!(bad.encode(&mut BoundedEncoder::new(4096)?).is_err()); + } + Ok(()) +} + +#[tokio::test] +async fn foreign_catalog_facts_fail_before_a_preparation_or_pin_is_created() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let catalog = fixture.install_empty_root(1).await?; + for foreign in [ + StoredCatalog { + repository: uuid::Uuid::new_v4().into_bytes(), + ..catalog + }, + StoredCatalog { + format: if format == ObjectFormat::Sha1 { + ObjectFormat::Sha256 + } else { + ObjectFormat::Sha1 + }, + ..catalog + }, + ] { + // Each immutable fact is inserted separately. Trusted production + // publication must reject it even earlier, before inserting it. + let generation = if foreign.repository != fixture.repository { + 2 + } else { + 3 + }; + fixture.install_catalog(generation, foreign).await?; + assert!( + fixture + .client() + .command::( + &fixture.target, + identity()?, + fixture.begin([35; 16]), + ) + .await + .is_err() + ); + assert_eq!(fixture.counts().await?, (0, 0)); + } + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn authoritative_base_resolution_uses_live_queried_facts_and_fences_failed_renewal_replay() +-> Result { + use crate::packs::{ + catalog::{CatalogFileLimits, CatalogFiles}, + closure::{BaseResolver, ClosureError}, + }; + use cellule_ltx::DiskBudget; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let native = + crate::packs::catalog::tests::prepared_for_repository(format, fixture.repository) + .await?; + fixture.install_catalog(1, native.stored).await?; + let client = fixture.client(); + let started = client + .command::(&fixture.target, identity()?, fixture.begin([36; 16])) + .await?; + let granted = lease(started.output)?; + let budget = DiskBudget::new(128 << 20); + let files = Arc::new(CatalogFiles::new( + fixture.root.path(), + budget.clone(), + Arc::clone(&native.store), + format, + CatalogFileLimits { + open_files: 1, + cached_files: 1, + ..CatalogFileLimits::default() + }, + )?); + let resolver = PreparationBaseResolver::open( + client.clone(), + fixture.target.clone(), + check(granted.token), + Arc::clone(&native.indexes), + Arc::clone(&files), + Some(started.receipt), + ) + .await?; + let base = resolver.context().base.ok_or("base")?; + assert_eq!(base.catalog, native.stored); + assert_eq!(base.generation, 1); + let native_ids: Vec<_> = native.fixture.objects.keys().copied().collect(); + let ids: Vec<_> = (0..512).map(|n| native_ids[n % native_ids.len()]).collect(); + let batch = resolver.resolve(base, &ids).await?; + assert_eq!(batch.base, base); + assert_eq!(batch.objects.len(), ids.len()); + for (oid, object) in ids.iter().zip(batch.objects) { + let object = object.ok_or("base object")?; + assert!(object.certified); + assert_eq!(object.header.object, native.fixture.objects[oid].0); + } + let mut foreign = base; + foreign.generation += 1; + assert!(matches!( + resolver.resolve(foreign, &ids).await, + Err(ClosureError::Integrity) + )); + assert!(matches!( + resolver.resolve(base, &vec![ids[0]; 513]).await, + Err(ClosureError::Integrity) + )); + let renewal = identity()?; + resolver.renew(renewal, DEFAULT_LEASE_MS).await?; + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([64; 32]), + sql::now(0)?, + 128, + 0, + |tx| { + tx.execute("UPDATE catalog_operations SET expires_at_ms=0", [])?; + tx.execute("UPDATE catalog_leases SET expires_at_ms=0", [])?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + // The exact renewal RPC replays success, but the subsequent fresh query + // sees expiry. It cannot restart a local deadline from the old reply. + assert!(matches!( + resolver.renew(renewal, DEFAULT_LEASE_MS).await, + Err(PreparationBaseError::Inactive) + )); + assert!(matches!( + resolver.resolve(base, &ids).await, + Err(ClosureError::LeaseExpired) + )); + assert!(matches!( + PreparationBaseResolver::open( + client, + fixture.target.clone(), + check(granted.token), + Arc::clone(&native.indexes), + Arc::clone(&files), + None + ) + .await, + Err(PreparationBaseError::Inactive) + )); + drop(resolver); + drop(files); + assert_eq!(budget.used(), 0); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn lease_quota_rejects_claim_without_mutating_the_existing_attempt() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let client = fixture.client(); + let started = lease( + client + .command::(&fixture.target, identity()?, fixture.begin([37; 16])) + .await? + .output, + )?; + let owner = fixture.handle.owner_fence(); + let expires = started.expires_at_ms; + for first in (1..MAX_GENERATION_LEASES).step_by(REAP_ROWS as usize) { + let last = (first + REAP_ROWS).min(MAX_GENERATION_LEASES); + fixture.handle.execute( + identity()?, Digest::from_bytes([64; 32]), sql::now(0)?, 128, 0, + move |tx| { + let mut insert = tx.prepare("INSERT INTO catalog_leases(incarnation,admission_sequence,operation,owner_epoch,artifact_operation,generation,expires_at_ms) VALUES(?1,?2,zeroblob(16),?4,?5,0,?3)")?; + for n in first..last { + insert.execute(rusqlite::params![ + owner.incarnation.as_bytes().as_slice(), + 1_000_000 + n as i64, + expires, + owner.epoch.to_be_bytes().as_slice(), + artifact_number(1_000_000 + n).as_slice() + ])?; + } + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success(Vec::new())) + }, + ).await?; + } + rejected( + client + .command::(&fixture.target, identity()?, request(started.token)) + .await, + PreparationDenial::Capacity, + ); + assert_eq!(fixture.counts().await?, (1, MAX_GENERATION_LEASES)); + let active = client + .query::(&fixture.target, None, check(started.token)) + .await? + .output + .ok_or("active")?; + assert_eq!(active.token, started.token); + assert_eq!(active.expires_at_ms, started.expires_at_ms); + fixture.runtime.shutdown().await?; + Ok(()) +} + +fn artifact_number(sequence: u64) -> [u8; 16] { + let mut value = *b"CANOPY01\0\0\0\0\0\0\0\0"; + value[8..].copy_from_slice(&sequence.to_be_bytes()); + value +} diff --git a/crates/canopy-server/src/packs/publication/tests/attestation.rs b/crates/canopy-server/src/packs/publication/tests/attestation.rs new file mode 100644 index 0000000..5feeec6 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/attestation.rs @@ -0,0 +1,543 @@ +use super::prepare::{cleaned, opened_native, physical}; +use super::*; +use crate::packs::metadata::tests::limits; +use cellule_ltx::DiskBudget; + +async fn prepared( + fixture: &Fixture, + operation: [u8; 16], +) -> Result<(PreparedCatalog, tempfile::TempDir, DiskBudget)> { + let (native, base, _, _) = opened_native(fixture, operation, 32).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(256 << 20); + let mut assembler = + CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + let (witness, segments) = physical(&native, root.path(), budget.clone()).await?; + assembler.begin_pack(witness)?; + for segment in segments { + assembler.add_segment(segment).await?; + } + assembler.finish_pack().await?; + Ok((assembler.finish().await?, root, budget)) +} +async fn saved(handle: &CellHandle, operation: [u8; 16]) -> Result> { + let bytes = handle + .query(0, CERTIFICATE_BYTES as usize, move |connection| { + use rusqlite::OptionalExtension; + let body: Option>> = connection + .query_row( + "SELECT attestation FROM catalog_operations WHERE id=?1", + [operation.as_slice()], + |row| row.get(0), + ) + .optional()?; + Ok(body.flatten().unwrap_or_default()) + }) + .await?; + if bytes.is_empty() { + return Ok(None); + } + let mut decoder = BoundedDecoder::new(&bytes, CERTIFICATE_BYTES)?; + let certificate = CatalogCertificate::decode(&mut decoder)?; + decoder.finish()?; + Ok(Some(certificate)) +} +async fn retained( + handle: &CellHandle, + token: PreparationToken, +) -> Result> { + let bytes = handle.query(0, CERTIFICATE_BYTES as usize, move |connection| { + use rusqlite::OptionalExtension; + let bytes = connection.query_row( + "SELECT operation,owner_epoch,artifact_operation,attestation,attestation_digest FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2", + rusqlite::params![token.owner.incarnation.as_bytes().as_slice(),token.attempt as i64], + |row| { + let operation: Vec = row.get(0)?; + let epoch: Vec = row.get(1)?; + let artifact: Vec = row.get(2)?; + let body: Option> = row.get(3)?; + let digest: Option> = row.get(4)?; + assert_eq!(operation, token.operation); + assert_eq!(epoch, token.owner.epoch.to_be_bytes()); + assert_eq!(artifact, token.artifact_operation); + assert_eq!(digest, body.as_ref().map(|b| blake3::hash(b).as_bytes().to_vec())); + Ok(body.unwrap_or_default()) + }, + ).optional()?; + Ok(bytes.unwrap_or_default()) + }).await?; + if bytes.is_empty() { + return Ok(None); + } + let mut decoder = BoundedDecoder::new(&bytes, CERTIFICATE_BYTES)?; + let certificate = CatalogCertificate::decode(&mut decoder)?; + decoder.finish()?; + Ok(Some(certificate)) +} +fn registered(outcome: AttestationOutcome) -> Result { + match outcome { + AttestationOutcome::Registered(value) => Ok(value), + AttestationOutcome::Denied(reason) => Err(format!("denied: {reason:?}").into()), + } +} +fn denied( + result: std::result::Result< + cellule_runtime::Committed, + InvocationError, + >, + reason: PreparationDenial, +) { + assert!( + matches!(result,Err(InvocationError::Rejected(ref value)) if value.output==AttestationOutcome::Denied(reason)) + ); +} +async fn catalog_state(handle: &CellHandle) -> Result> { + Ok(handle + .query(0, 32, |connection| { + let generation: u64 = connection.query_row( + "SELECT generation FROM catalog_state WHERE singleton=1", + [], + |row| row.get(0), + )?; + let refs: u64 = + connection.query_row("SELECT count(*) FROM refs", [], |row| row.get(0))?; + let generations: u64 = + connection.query_row("SELECT count(*) FROM catalog_generations", [], |row| { + row.get(0) + })?; + Ok([ + generation.to_be_bytes(), + refs.to_be_bytes(), + generations.to_be_bytes(), + ] + .concat()) + }) + .await?) +} + +#[tokio::test] +async fn native_catalog_attestation_is_bounded_durable_idempotent_and_not_publication() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let (proof, root, budget) = prepared(&fixture, [52; 16]).await?; + let before = catalog_state(&fixture.handle).await?; + let inline = proof.certificate().await?; + assert!( + saved(&fixture.handle, proof.token().operation) + .await? + .is_none() + ); + assert_eq!(catalog_state(&fixture.handle).await?, before); + let mutation = identity()?; + let first = proof.attest(mutation).await?; + let binding = registered(first.output)?; + assert_eq!(binding.token, proof.token()); + let certificate = saved(&fixture.handle, proof.token().operation) + .await? + .ok_or("certificate")?; + let bytes = certificate.bytes()?; + assert_eq!( + retained(&fixture.handle, proof.token()).await?, + Some(certificate.clone()) + ); + assert_eq!(inline, certificate); + assert!(bytes.len() <= CERTIFICATE_BYTES as usize); + assert_eq!(binding.certificate_digest, *blake3::hash(&bytes).as_bytes()); + let facts = certificate.data()?; + assert_eq!(facts.base, proof.base()); + assert_eq!(facts.catalog, proof.catalog()); + assert_eq!(facts.object_count, proof.object_count()); + assert_eq!(facts.inputs_digest, proof.inputs_digest()); + assert_eq!(facts.inventory_digest, proof.inventory_digest()); + let replay = proof.attest(mutation).await?; + assert_eq!(replay.output, first.output); + assert_eq!(replay.receipt, first.receipt); + assert_eq!(proof.attest(identity()?).await?.output, first.output); + assert_eq!(catalog_state(&fixture.handle).await?, before); + assert_eq!(fixture.counts().await?, (1, 1)); + // No extra renewal or staged proof record is needed to retain it. + fixture + .client() + .command::(&fixture.target, identity()?, request(proof.token())) + .await?; + assert_eq!( + saved(&fixture.handle, proof.token().operation).await?, + Some(certificate.clone()) + ); + for length in 0..bytes.len() { + let mut decoder = BoundedDecoder::new(&bytes[..length], CERTIFICATE_BYTES)?; + assert!(CatalogCertificate::decode(&mut decoder).is_err()); + } + let mut extra = bytes.clone(); + extra.push(0); + let mut decoder = BoundedDecoder::new(&extra, CERTIFICATE_BYTES)?; + CatalogCertificate::decode(&mut decoder)?; + assert!(decoder.finish().is_err()); + drop(proof); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn edited_facts_wrong_scope_and_conflicting_attestations_do_not_replace_the_record() -> Result +{ + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let (proof, root, budget) = prepared(&fixture, [53; 16]).await?; + proof.attest(identity()?).await?; + let original = saved(&fixture.handle, proof.token().operation) + .await? + .ok_or("certificate")?; + for scope in [false, true] { + let mut certificate = original.clone(); + let mut facts = certificate.data()?; + if scope { + facts.application[0] ^= 1; + } else { + facts.inventory_digest[0] ^= 1; + } + let mut encoder = BoundedEncoder::new(960)?; + facts.encode(&mut encoder)?; + certificate.0.body = encoder.finish(); + denied( + fixture + .client() + .command::(&fixture.target, identity()?, certificate) + .await, + if scope { + PreparationDenial::Stale + } else { + PreparationDenial::Unauthorized + }, + ); + } + // A privileged test signer cannot overwrite a prior valid registration + // with a different valid certificate for the same attempt either. + let mut conflicting = original.data()?; + conflicting.inputs_digest[0] ^= 1; + let conflicting = CatalogCertificate::seal(&conflicting, &[16; 32])?; + denied( + fixture + .client() + .command::(&fixture.target, identity()?, conflicting) + .await, + PreparationDenial::Conflict, + ); + assert_eq!( + saved(&fixture.handle, proof.token().operation).await?, + Some(original) + ); + assert_eq!( + catalog_state(&fixture.handle).await?, + [0u64.to_be_bytes(), 0u64.to_be_bytes(), 1u64.to_be_bytes()].concat() + ); + drop(proof); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn revoked_actor_cannot_issue_or_register_but_recorded_outcome_remains_exact() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let (proof, root, budget) = prepared(&fixture, [57; 16]).await?; + let certificate = proof.certificate().await?; + let mutation = identity()?; + let first = fixture + .client() + .command::(&fixture.target, mutation, certificate.clone()) + .await?; + let before = catalog_state(&fixture.handle).await?; + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([64; 32]), + sql::now(0)?, + 1, + 0, + |tx| { + tx.execute( + "UPDATE repository_identity SET owner='successor' WHERE singleton=1", + [], + )?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + assert!(proof.certificate().await.is_err()); + assert!(proof.attest(identity()?).await.is_err()); + denied( + fixture + .client() + .command::( + &fixture.target, + identity()?, + certificate.clone(), + ) + .await, + PreparationDenial::Unauthorized, + ); + let replay = fixture + .client() + .command::(&fixture.target, mutation, certificate.clone()) + .await?; + assert_eq!(replay.output, first.output); + assert_eq!(replay.receipt, first.receipt); + assert_eq!( + saved(&fixture.handle, proof.token().operation).await?, + Some(certificate) + ); + assert_eq!(catalog_state(&fixture.handle).await?, before); + drop(proof); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn claim_clears_the_attestation_and_recorded_replay_cannot_recreate_it() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let (proof, root, budget) = prepared(&fixture, [54; 16]).await?; + let mutation = identity()?; + let first = proof.attest(mutation).await?; + let certificate = saved(&fixture.handle, proof.token().operation) + .await? + .ok_or("certificate")?; + let claimed = lease( + fixture + .client() + .command::(&fixture.target, identity()?, request(proof.token())) + .await? + .output, + )?; + assert_ne!(claimed.token.attempt, proof.token().attempt); + assert_ne!( + claimed.token.artifact_operation, + proof.token().artifact_operation + ); + assert_eq!( + retained(&fixture.handle, proof.token()).await?, + Some(certificate.clone()) + ); + assert!(retained(&fixture.handle, claimed.token).await?.is_none()); + assert!( + saved(&fixture.handle, proof.token().operation) + .await? + .is_none() + ); + denied( + fixture + .client() + .command::( + &fixture.target, + identity()?, + certificate.clone(), + ) + .await, + PreparationDenial::Stale, + ); + let replay = fixture + .client() + .command::(&fixture.target, mutation, certificate) + .await?; + assert_eq!(replay.output, first.output); + assert_eq!(replay.receipt, first.receipt); + assert!( + saved(&fixture.handle, proof.token().operation) + .await? + .is_none() + ); + assert!(proof.attest(identity()?).await.is_err()); + assert_eq!(fixture.counts().await?, (1, 2)); + drop(proof); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn expired_attestations_are_not_reissued_and_reaping_removes_the_bounded_record() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let (proof, root, budget) = prepared(&fixture, [55; 16]).await?; + let mutation = identity()?; + let first = proof.attest(mutation).await?; + let certificate = saved(&fixture.handle, proof.token().operation) + .await? + .ok_or("certificate")?; + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([69; 32]), + sql::now(0)?, + 128, + 0, + |tx| { + tx.execute("UPDATE catalog_operations SET expires_at_ms=0", [])?; + tx.execute("UPDATE catalog_leases SET expires_at_ms=0", [])?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + assert!(proof.attest(identity()?).await.is_err()); + denied( + fixture + .client() + .command::( + &fixture.target, + identity()?, + certificate.clone(), + ) + .await, + PreparationDenial::Expired, + ); + let replay = fixture + .client() + .command::(&fixture.target, mutation, certificate.clone()) + .await?; + assert_eq!(replay.output, first.output); + assert_eq!(replay.receipt, first.receipt); + assert_eq!( + fixture + .client() + .command::( + &fixture.target, + identity()?, + MaintenanceRequest { + repository: fixture.repository, + actor: "owner".into(), + owner: fixture.handle.owner_fence() + } + ) + .await? + .output, + 2 + ); + assert!( + saved(&fixture.handle, proof.token().operation) + .await? + .is_none() + ); + denied( + fixture + .client() + .command::( + &fixture.target, + identity()?, + certificate.clone(), + ) + .await, + PreparationDenial::Missing, + ); + let fresh = lease( + fixture + .client() + .command::( + &fixture.target, + identity()?, + fixture.begin(proof.token().operation), + ) + .await? + .output, + )?; + assert!(fresh.token.attempt > proof.token().attempt); + assert_eq!(fresh.token.artifact_operation, artifact_number(2)); + assert!(retained(&fixture.handle, proof.token()).await?.is_none()); + denied( + fixture + .client() + .command::(&fixture.target, identity()?, certificate) + .await, + PreparationDenial::Stale, + ); + drop(proof); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn successor_owner_cannot_register_old_proofs_but_preserves_exact_outcome_replay() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let (proof, root, budget) = prepared(&fixture, [56; 16]).await?; + let mutation = identity()?; + let first = proof.attest(mutation).await?; + let certificate = saved(&fixture.handle, proof.token().operation) + .await? + .ok_or("certificate")?; + fixture.handle.drain().await?; + fixture.runtime.shutdown().await?; + let session = SessionId::from_bytes([57; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, session)?; + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await? + .ok_or("idle")?; + let provision = CellCatalog::new(fixture.layout.clone(), fixture.target.tenant()) + .lookup(fixture.target.cell_id()) + .await? + .ok_or("provision")?; + let handle = runtime + .acquire_idle_restored( + provision, + fixture.replica.clone(), + authority, + idle, + fixture.root.path().join("attestation-b.sqlite"), + Owner { + session, + endpoint: "https://attestation-b.invalid".into(), + }, + ) + .await?; + let client = CellClient::local(Arc::clone(&fixture.registry), handle.clone()); + assert!(handle.owner_fence().epoch > proof.token().owner.epoch); + denied( + client + .command::( + &fixture.target, + identity()?, + certificate.clone(), + ) + .await, + PreparationDenial::Stale, + ); + let replay = client + .command::(&fixture.target, mutation, certificate.clone()) + .await?; + assert_eq!(replay.output, first.output); + assert_eq!(replay.receipt, first.receipt); + assert_eq!( + saved(&handle, proof.token().operation).await?, + Some(certificate.clone()) + ); + assert_eq!( + retained(&handle, proof.token()).await?, + Some(certificate.clone()) + ); + let claimed = lease( + client + .command::(&fixture.target, identity()?, request(proof.token())) + .await? + .output, + )?; + assert_eq!(claimed.token.artifact_operation, artifact_number(2)); + assert_eq!(retained(&handle, proof.token()).await?, Some(certificate)); + assert!(retained(&handle, claimed.token).await?.is_none()); + assert!(saved(&handle, proof.token().operation).await?.is_none()); + assert_eq!( + catalog_state(&handle).await?, + [0u64.to_be_bytes(), 0u64.to_be_bytes(), 1u64.to_be_bytes()].concat() + ); + drop(proof); + cleaned(root.path(), &budget).await?; + runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/compaction.rs b/crates/canopy-server/src/packs/publication/tests/compaction.rs new file mode 100644 index 0000000..79fae6d --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/compaction.rs @@ -0,0 +1,506 @@ +use super::*; +use super::{ + prepare::{cleaned, opened}, + publishing::{edit, plan, state, update}, + reconcile::graph, +}; +use crate::packs::{ + catalog::{CatalogFiles, CatalogIndexes, CatalogReader, CatalogSnapshot}, + directory::snapshot::DirectorySnapshot, + metadata::tests::limits, +}; +use canopy_object_storage::artifact::ArtifactStore; +use cellule_ltx::DiskBudget; + +mod coordinator; +mod range; +mod recovery; +mod schedule; + +#[tokio::test] +async fn compaction_carries_the_same_authenticated_ref_snapshot_into_the_next_joint_generation() +-> Result { + use crate::packs::ref_state::{RefStateIndex, RefStateSnapshot, RefStateSnapshotRoot}; + use crate::{ObjectId, RefUpdate}; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let inventory = seed(&fixture, 2).await?; + let bytes = refs(&fixture.handle).await?; + type Rows = Vec<(String, Option>, i64)>; + let (rows, generation): (Rows, u64) = serde_json::from_slice(&bytes)?; + let changes = plan( + rows.into_iter() + .map(|(name, oid, version)| { + assert_eq!(version, 1); + Ok(RefUpdate { + name, + expected: None, + new_oid: oid + .map(|bytes| ObjectId::try_from(bytes.as_slice())) + .transpose()?, + }) + }) + .collect::>>()?, + ); + let index = RefStateIndex::new(Arc::clone(&inventory.store), format); + let refs = index.prepare(None, artifact_number(900), &changes).await?; + let snapshot = RefStateSnapshotRoot::upload( + &inventory.store, + artifact_number(900), + RefStateSnapshot { + repository: fixture.repository, + format, + generation, + default_branch: "refs/heads/main".into(), + root: Some(refs.root()), + }, + ) + .await?; + let first = prepare_compaction(&fixture, &inventory, 210, &[0, 1]).await?; + fixture + .install_generation( + 3, + first.compact.base().catalog.ok_or("catalog")?, + Some(snapshot), + ) + .await?; + let prepared = prepare_compaction(&fixture, &inventory, 211, &[0, 1]).await?; + assert_eq!(prepared.compact.base().refs, Some(snapshot)); + let certificate = prepared.compact.certificate().await?; + assert_eq!(certificate.data()?.base.refs, Some(snapshot)); + assert!(certificate.bytes()?.len() <= CERTIFICATE_BYTES as usize); + let committed = fixture + .client() + .command::(&fixture.target, identity()?, certificate.clone()) + .await?; + assert!( + matches!(committed.output, CompactionReply::Published(value) if value.generation==4) + ); + let observed = fixture + .client() + .command::(&fixture.target, identity()?, fixture.begin([212; 16])) + .await?; + assert_eq!(lease(observed.output)?.base.refs, Some(snapshot)); + assert_eq!( + snapshot.read(&inventory.store).await?.generation, + generation + ); + assert_eq!( + index + .read(Some(refs.root()), &changes.updates[0].name) + .await? + .and_then(|r| r.oid), + changes.updates[0].new_oid + ); + assert_eq!( + fixture + .client() + .command::(&fixture.target, identity()?, certificate) + .await? + .output, + committed.output + ); + drop(first.compact); + drop(prepared.compact); + cleaned(first.root.path(), &first.budget).await?; + cleaned(prepared.root.path(), &prepared.budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +struct Inventory { + provider: Arc, + store: Arc, +} +async fn seed(fixture: &Fixture, roots: usize) -> Result { + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new( + Arc::clone(&provider), + fixture.repository, + )); + for n in 0..roots { + push( + fixture, + &Inventory { + provider: Arc::clone(&provider), + store: Arc::clone(&store), + }, + n as u8 + 20, + 4 + n % 3, + ) + .await?; + } + Ok(Inventory { provider, store }) +} +async fn push(fixture: &Fixture, inventory: &Inventory, operation: u8, blobs: usize) -> Result { + let graph = graph( + fixture, + Arc::clone(&inventory.provider), + Arc::clone(&inventory.store), + [operation; 16], + blobs, + ) + .await?; + let proof = Box::pin(graph.prepared.ref_proof( + plan(vec![update( + &format!("refs/heads/b-{operation}"), + None, + Some(graph.initial), + )]), + graph.root.path(), + graph.budget.clone(), + limits(), + )) + .await?; + let result = fixture + .client() + .command::(&fixture.target, identity()?, proof) + .await?; + assert!(matches!(result.output, PublicationReply::Published(_))); + Ok(()) +} +struct Prepared { + compact: PreparedCompaction, + root: tempfile::TempDir, + budget: DiskBudget, + files: Arc, + indexes: Arc, +} +async fn prepare_compaction( + fixture: &Fixture, + inventory: &Inventory, + operation: u8, + selected: &[usize], +) -> Result { + let (base, files, indexes) = + opened(fixture, [operation; 16], Arc::clone(&inventory.store)).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let compact = PreparedCompaction::prepare( + root.path(), + budget.clone(), + base, + selected, + CompactionLimits { + spool: limits(), + output: crate::packs::metadata::MetadataLimits { + max_file_bytes: 16 << 10, + cache_kib: 16, + }, + ..CompactionLimits::default() + }, + ) + .await?; + Ok(Prepared { + compact, + root, + budget, + files, + indexes, + }) +} +async fn refs(handle: &CellHandle) -> Result> { + Ok(handle + .query(0, 64 << 10, |connection| { + let mut statement = + connection.prepare("SELECT name,oid,version FROM refs ORDER BY name")?; + let rows = statement + .query_map([], |row| { + Ok(( + row.get::<_, String>(0)?, + row.get::<_, Option>>(1)?, + row.get::<_, i64>(2)?, + )) + })? + .collect::>>()?; + let generation: u64 = + connection.query_row("SELECT generation FROM ref_generation", [], |row| { + row.get(0) + })?; + serde_json::to_vec(&(rows, generation)).map_err(|_| Error::Command("fixture ref state")) + }) + .await?) +} +async fn outcomes(handle: &CellHandle) -> Result { + let bytes = handle + .query(0, 64, |connection| { + let count: u64 = + connection.query_row("SELECT count(*) FROM catalog_compactions", [], |row| { + row.get(0) + })?; + Ok(count.to_be_bytes().to_vec()) + }) + .await?; + Ok(u64::from_be_bytes( + bytes.try_into().map_err(|_| "invalid outcome count")?, + )) +} +async fn reject( + fixture: &Fixture, + certificate: CatalogCertificate, + reason: PreparationDenial, +) -> Result { + let before = state(&fixture.handle).await?; + let count = outcomes(&fixture.handle).await?; + let result = fixture + .client() + .command::(&fixture.target, identity()?, certificate) + .await; + assert!( + matches!(result, Err(InvocationError::Rejected(ref value)) if value.output == CompactionReply::Denied(reason)), + "{result:?}" + ); + assert_eq!(state(&fixture.handle).await?, before); + assert_eq!(outcomes(&fixture.handle).await?, count); + Ok(()) +} + +#[tokio::test] +async fn full_ingress_compaction_preserves_native_inventory_refs_and_exact_replay() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let inventory = seed(&fixture, 32).await?; + let prepared = + prepare_compaction(&fixture, &inventory, 180, &(0..32).collect::>()).await?; + let original_catalog = prepared.compact.base().catalog.ok_or("base")?; + let original = CatalogSnapshot::download(&inventory.store, original_catalog).await?; + let directory = DirectorySnapshot::download(&inventory.store, original.directory).await?; + assert_eq!(directory.level_zero.len(), 32); + let mut ids = std::collections::BTreeSet::new(); + for root in &directory.level_zero { + let mut cursor = prepared.indexes.ranges().cursor(Some(*root), None)?; + while let Some(stored) = cursor.next().await? { + use crate::packs::directory::snapshot::RunLoader; + let run = prepared.files.load(stored).await?; + let mut after = None; + loop { + let entries = run.entries_after(after)?; + if entries.is_empty() { + break; + } + after = entries.last().map(|entry| entry.header.object.oid); + ids.extend(entries.iter().map(|entry| entry.header.object.oid)); + } + } + } + assert_eq!(prepared.compact.object_count(), ids.len() as u64); + let old_reader = + CatalogReader::open(Arc::clone(&prepared.indexes), original_catalog).await?; + let new_reader = + CatalogReader::open(Arc::clone(&prepared.indexes), prepared.compact.catalog()).await?; + for oid in &ids { + assert_eq!( + old_reader + .lookup(*oid, &*prepared.files, &*prepared.files) + .await? + .ok_or("old")? + .entry, + new_reader + .lookup(*oid, &*prepared.files, &*prepared.files) + .await? + .ok_or("new")? + .entry + ); + } + assert!( + fixture + .client() + .query::(&fixture.target, None, fixture.begin([180; 16])) + .await? + .output + .is_none() + ); + let before = refs(&fixture.handle).await?; + let certificate = prepared.compact.certificate().await?; + assert!(certificate.bytes()?.len() <= CERTIFICATE_BYTES as usize); + let mutation = identity()?; + let committed = fixture + .client() + .command::(&fixture.target, mutation, certificate.clone()) + .await?; + assert!( + matches!(committed.output,CompactionReply::Published(value) if value.generation==33) + ); + assert_eq!(refs(&fixture.handle).await?, before); + let current = + CatalogSnapshot::download(&inventory.store, prepared.compact.catalog()).await?; + assert_eq!(current.sources, original.sources); + assert_eq!( + DirectorySnapshot::download(&inventory.store, current.directory) + .await? + .level_zero + .len(), + 1 + ); + // Old pinned roots remain usable after publication; compaction deletes no bytes. + for oid in ids { + assert!( + old_reader + .lookup(oid, &*prepared.files, &*prepared.files) + .await? + .is_some() + ); + } + let replay = fixture + .client() + .command::(&fixture.target, mutation, certificate.clone()) + .await?; + assert_eq!(replay.receipt, committed.receipt); + assert_eq!(replay.output, committed.output); + assert_eq!( + fixture + .client() + .command::(&fixture.target, identity()?, certificate) + .await? + .output, + committed.output + ); + assert_eq!( + fixture + .client() + .query::(&fixture.target, None, fixture.begin([180; 16])) + .await? + .output, + Some(committed.output) + ); + let counts = fixture.counts().await?; + rejected( + fixture + .client() + .command::(&fixture.target, identity()?, fixture.begin([180; 16])) + .await, + PreparationDenial::Conflict, + ); + assert_eq!(fixture.counts().await?, counts); + assert_eq!(outcomes(&fixture.handle).await?, 1); + drop(prepared.compact); + cleaned(prepared.root.path(), &prepared.budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn compaction_reconciles_new_ingress_but_cannot_resurrect_replaced_roots() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let inventory = seed(&fixture, 2).await?; + let prepared = prepare_compaction(&fixture, &inventory, 180, &[1, 0]).await?; + let competitor = prepare_compaction(&fixture, &inventory, 181, &[0, 1]).await?; + let certificate = prepared.compact.certificate().await?; + fixture + .client() + .command::(&fixture.target, identity()?, certificate.clone()) + .await?; + push(&fixture, &inventory, 99, 8).await?; + reject(&fixture, certificate, PreparationDenial::Conflict).await?; + let selected = prepared.compact.reconcile().await?; + assert_eq!(selected.token(), prepared.compact.token()); + assert_eq!( + selected.inventory_digest(), + prepared.compact.inventory_digest() + ); + let before = refs(&fixture.handle).await?; + let result = fixture + .client() + .command::( + &fixture.target, + identity()?, + selected.certificate().await?, + ) + .await?; + assert!(matches!(result.output,CompactionReply::Published(value) if value.generation==4)); + assert_eq!(refs(&fixture.handle).await?, before); + let snapshot = CatalogSnapshot::download(&inventory.store, selected.catalog()).await?; + assert_eq!( + DirectorySnapshot::download(&inventory.store, snapshot.directory) + .await? + .level_zero + .len(), + 2 + ); + assert!(matches!( + competitor.compact.reconcile().await, + Err(CatalogPreparationError::Catalog( + crate::packs::directory::index::IndexError::Stale + )) + )); + assert_eq!(outcomes(&fixture.handle).await?, 1); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn maintenance_certificate_purpose_authority_tampering_and_late_rollback_are_enforced() +-> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let inventory = seed(&fixture, 3).await?; + let prepared = prepare_compaction(&fixture, &inventory, 180, &[0, 1]).await?; + let certificate = prepared.compact.certificate().await?; + let mut data = certificate.data()?; + data.compaction = false; + reject( + &fixture, + CatalogCertificate::seal(&data, &[16; 32])?, + PreparationDenial::Unauthorized, + ) + .await?; + data.compaction = true; + data.refs_digest = Some([1; 32]); + assert!(CatalogCertificate::seal(&data, &[16; 32]).is_err()); + data.refs_digest = None; + data.object_count += 1; + let mut tampered = certificate.clone(); + let mut e = BoundedEncoder::new(960)?; + data.encode(&mut e)?; + tampered.0.body = e.finish(); + reject(&fixture, tampered, PreparationDenial::Unauthorized).await?; + edit(&fixture,"UPDATE repository_identity SET owner='other'; INSERT INTO repository_members VALUES('owner','write');").await?; + reject( + &fixture, + certificate.clone(), + PreparationDenial::Unauthorized, + ) + .await?; + assert!(prepared.compact.certificate().await.is_err()); + assert_eq!( + fixture + .client() + .query::(&fixture.target, None, fixture.begin([180; 16])) + .await? + .output, + Some(CompactionReply::Denied(PreparationDenial::Unauthorized)) + ); + edit(&fixture,"UPDATE repository_identity SET owner='owner'; DELETE FROM repository_members WHERE account='owner'; CREATE TRIGGER forced_compaction_failure BEFORE INSERT ON catalog_compactions BEGIN SELECT RAISE(ABORT,'forced late compaction failure'); END;").await?; + let before = state(&fixture.handle).await?; + assert!( + fixture + .client() + .command::(&fixture.target, identity()?, certificate.clone()) + .await + .is_err() + ); + assert_eq!(state(&fixture.handle).await?, before); + assert_eq!(outcomes(&fixture.handle).await?, 0); + edit(&fixture, "DROP TRIGGER forced_compaction_failure;").await?; + fixture + .client() + .command::(&fixture.target, identity()?, certificate) + .await?; + assert!( + edit(&fixture, "UPDATE catalog_compactions SET actor='other';") + .await + .is_err() + ); + assert!( + edit( + &fixture, + "INSERT OR REPLACE INTO catalog_compactions SELECT * FROM catalog_compactions;" + ) + .await + .is_err() + ); + assert_eq!(outcomes(&fixture.handle).await?, 1); + fixture.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/compaction/coordinator.rs b/crates/canopy-server/src/packs/publication/tests/compaction/coordinator.rs new file mode 100644 index 0000000..35c5650 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/compaction/coordinator.rs @@ -0,0 +1,465 @@ +use super::super::coordinator::{empty_in_store, finished, refused, request}; +use super::*; +use tokio::time::{Duration, timeout}; + +fn compacted(state: PublicationState) -> Result> { + match state { + PublicationState::Finished(Ok(PublicationOutcome::Compaction(value))) => Ok(value), + other => Err(format!("unexpected {other:?}").into()), + } +} + +#[tokio::test] +async fn maintenance_final_publication_uses_shared_bound_lifecycle_and_reserved_fair_dispatch() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let inventory = seed(&fixture, 2).await?; + let before_refs = refs(&fixture.handle).await?; + let stages = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let ready = ReadyStaging::new( + fixture.client(), + fixture.target.clone(), + fixture.begin([184; 16]), + identity()?, + ) + .await?; + let ticket = stages.submit(ready).map_err(|(e, _)| e)?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait()).await?, + StagingState::Active(_) + )); + ticket.seal()?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait_terminal()).await?, + StagingState::Bound(_) + )); + let session = ticket.bound_session()?; + let root = Arc::new(tempfile::TempDir::new()?); + let budget = DiskBudget::new(128 << 20); + let indexes = Arc::new(CatalogIndexes::new(inventory.store.clone(), format)); + let files = Arc::new(CatalogFiles::new( + fixture.root.path(), + DiskBudget::new(64 << 20), + inventory.store.clone(), + format, + crate::packs::catalog::CatalogFileLimits::default(), + )?); + let base = Arc::new(ticket.open_base(indexes, files).await?); + let owned_root = root.clone(); + let owned_budget = budget.clone(); + let mutation = identity()?; + let work = ticket.spawn_bound(move |_| async move { + let prepared = Arc::new( + PreparedCompaction::prepare( + owned_root.path(), + owned_budget, + base, + &[0, 1], + CompactionLimits { + spool: limits(), + output: crate::packs::metadata::MetadataLimits { + max_file_bytes: 16 << 10, + cache_kib: 16, + }, + ..CompactionLimits::default() + }, + ) + .await + .map_err(|e| StagingError::Input(Box::new(e)))?, + ); + let weak = Arc::downgrade(&prepared); + let ready = prepared + .ready_compaction(mutation) + .await + .map_err(|e| StagingError::Input(Box::new(e)))?; + Ok((ready, weak)) + })?; + let (ready, weak) = work.wait().await.map_err(|e| e.to_string())?; + let publications = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let (release, entered) = publications.pause_for_test().await; + let observer = ticket.publish(&publications, ready)?; + timeout(Duration::from_secs(10), entered).await??; + let stats = publications.stats().await; + assert_eq!( + (stats.foreground, stats.maintenance, stats.command_bytes), + (0, 1, 8 << 10) + ); + assert!(weak.upgrade().is_some()); + assert_eq!(outcomes(&fixture.handle).await?, 0); + release + .send(()) + .map_err(|_| "maintenance transport disappeared")?; + let completed = compacted(timeout(Duration::from_secs(10), observer.wait()).await?)?; + assert!(matches!( + completed.output, + CompactionReply::Published(PublishedCompaction { generation: 3, .. }) + )); + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait_terminal()).await?, + StagingState::Published(Ok(PublicationOutcome::Compaction(_))) + )); + assert!(observer.response().await.is_err()); + assert!(session.live_lease().is_err()); + assert_eq!(refs(&fixture.handle).await?, before_refs); + assert_eq!(outcomes(&fixture.handle).await?, 1); + assert!(weak.upgrade().is_none()); + cleaned(root.path(), &budget).await?; + assert!(stages.close_and_drain().await.is_empty()); + assert!(publications.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn uncertain_compaction_retains_exact_command_and_recovers_original_receipt() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in [1, 2, 3] { + let fixture = Fixture::new(format).await?; + let inventory = seed(&fixture, 2).await?; + let before_refs = refs(&fixture.handle).await?; + let prepared = prepare_compaction(&fixture, &inventory, 180, &[0, 1]).await?; + let compact = Arc::new(prepared.compact); + let weak = Arc::downgrade(&compact); + let ready = compact.ready_compaction(identity()?).await?; + let coordinator = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + coordinator.fault_for_test(fault); + let ticket = coordinator.try_reserve(ready)?; + assert_eq!(ticket.class(), PublicationClass::Maintenance); + assert_eq!(coordinator.stats().await.held, 1); + assert_eq!(outcomes(&fixture.handle).await?, 0); + drop(compact); + assert!(weak.upgrade().is_some()); + ticket.activate().await?; + let uncertain = timeout(Duration::from_secs(10), ticket.wait()).await?; + let PublicationState::Uncertain(error) = uncertain else { + return Err("uncertainty lost".into()); + }; + let PublicationError::Compaction(InvocationError::Pending(evidence)) = error.as_ref() + else { + return Err("wrong evidence kind".into()); + }; + let original = match fixture.client().resolve(evidence).await? { + cellule_runtime::Resolution::Committed(value) => Some(value.commit_sequence()), + cellule_runtime::Resolution::Absent => None, + other => return Err(format!("unexpected resolution {other:?}").into()), + }; + assert_eq!(original.is_some(), fault != 1); + let stats = coordinator.stats().await; + assert_eq!( + ( + stats.foreground, + stats.maintenance, + stats.uncertain, + stats.command_bytes + ), + (0, 1, 1, 8 << 10) + ); + assert!(weak.upgrade().is_some()); + assert_eq!( + fixture + .client() + .query::( + &fixture.target, + None, + fixture.begin([180; 16]) + ) + .await? + .output + .is_some(), + fault != 1 + ); + edit( + &fixture, + "UPDATE ref_generation SET visibility='public' WHERE singleton=1", + ) + .await?; + drop(ticket); + let retained = coordinator.pending([180; 16]).await.ok_or("input lost")?; + let drained = coordinator.close_and_drain().await; + assert_eq!(drained.len(), 1); + assert_eq!(drained[0].class(), PublicationClass::Maintenance); + retained.recover().await?; + let completed = compacted(timeout(Duration::from_secs(10), retained.wait()).await?)?; + if let Some(sequence) = original { + assert_eq!(completed.receipt.commit_sequence, sequence); + } + assert!(matches!( + completed.output, + CompactionReply::Published(PublishedCompaction { generation: 3, .. }) + )); + assert_eq!( + fixture + .client() + .query::( + &fixture.target, + None, + fixture.begin([180; 16]) + ) + .await? + .output, + Some(completed.output) + ); + assert_eq!(refs(&fixture.handle).await?, before_refs); + assert_eq!(outcomes(&fixture.handle).await?, 1); + assert!(retained.response().await.is_err()); + assert!(weak.upgrade().is_none()); + assert_eq!(coordinator.reservations_for_test().await, (0, 0, 0)); + assert!(coordinator.close_and_drain().await.is_empty()); + cleaned(prepared.root.path(), &prepared.budget).await?; + fixture.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn reserved_classes_and_actor_quotas_keep_mixed_admission_bounded() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let inventory = seed(&fixture, 2).await?; + edit( + &fixture, + "INSERT INTO repository_members VALUES('writer','write')", + ) + .await?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits { + operations: 5, + per_actor: 2, + command_bytes: (24 << 20) + (16 << 10), + in_flight: 1, + maintenance_operations: 2, + maintenance_in_flight: 1, + foreground_burst: 3, + }, + )?; + let (release, entered) = coordinator.pause_for_test().await; + let mut entered = Some(entered); + let mut foreground = Vec::new(); + let mut tickets = Vec::new(); + for (n, actor) in [(90, "owner"), (91, "owner"), (92, "writer")] { + let (prepared, root, budget) = + empty_in_store(&fixture, [n; 16], actor, Arc::clone(&inventory.store)).await?; + let ready = Box::pin(prepared.ready_push( + identity()?, + request(refused()), + root.path(), + budget.clone(), + limits(), + )) + .await?; + tickets.push(coordinator.submit(ready).await?); + if let Some(entered) = entered.take() { + timeout(Duration::from_secs(5), entered).await??; + } + foreground.push((prepared, root, budget)); + } + let before_refs = refs(&fixture.handle).await?; + let mut maintenance = Vec::new(); + for operation in [180, 181] { + let prepared = prepare_compaction(&fixture, &inventory, operation, &[0, 1]).await?; + let compact = Arc::new(prepared.compact); + tickets.push(coordinator.try_reserve(compact.ready_compaction(identity()?).await?)?); + maintenance.push((compact, prepared.root, prepared.budget)); + } + let stats = coordinator.stats().await; + assert_eq!((stats.held, stats.in_flight, stats.queued), (2, 1, 2)); + assert_eq!( + ( + stats.admitted, + stats.foreground, + stats.maintenance, + stats.accounts, + stats.command_bytes + ), + (5, 3, 2, 2, (24 << 20) + (16 << 10)) + ); + let extra = prepare_compaction(&fixture, &inventory, 182, &[0, 1]).await?; + let extra = Arc::new(extra.compact); + let failure = coordinator + .submit(extra.ready_compaction(identity()?).await?) + .await + .err() + .ok_or("maintenance overflow")?; + assert_eq!(failure.reason, PublicationScheduleError::Capacity); + assert!(matches!(failure.ready, ReadyPublication::Compaction(_))); + let duplicate = coordinator + .submit(maintenance[0].0.ready_compaction(identity()?).await?) + .await + .err() + .ok_or("duplicate")?; + assert_eq!(duplicate.reason, PublicationScheduleError::Duplicate); + // The logical ID namespace is shared even when trusted preparation can + // construct a different purpose-bound command under that same lease. + let push_root = tempfile::TempDir::new()?; + let push_budget = DiskBudget::new(256 << 20); + let push_base = Arc::new(maintenance[0].0.preparation_base().select_current().await?); + let push = Arc::new( + CatalogPreparation::new(push_root.path(), push_budget.clone(), push_base, limits()) + .await? + .finish() + .await?, + ); + let ready = Box::pin(push.ready_push( + identity()?, + request(refused()), + push_root.path(), + push_budget.clone(), + limits(), + )) + .await?; + let cross_class = coordinator + .submit(ready) + .await + .err() + .ok_or("cross-class duplicate")?; + assert_eq!(cross_class.reason, PublicationScheduleError::Duplicate); + assert!(matches!(cross_class.ready, ReadyPublication::Push(_))); + for ticket in &tickets { + ticket.activate().await?; + } + release.send(()).map_err(|_| "worker disappeared")?; + for ticket in tickets { + let state = timeout(Duration::from_secs(10), ticket.wait()).await?; + match ticket.class() { + PublicationClass::Foreground => { + finished(state)?; + assert_eq!(ticket.response().await?, refused()); + } + PublicationClass::Maintenance => match state { + PublicationState::Finished(Ok(PublicationOutcome::Compaction(value))) => { + assert!(matches!(value.output, CompactionReply::Published(_))) + } + PublicationState::Finished(Err(error)) => assert!( + matches!(error.as_ref(), PublicationError::Compaction(InvocationError::Rejected(value)) if value.output==CompactionReply::Denied(PreparationDenial::Conflict)) + ), + other => return Err(format!("unexpected {other:?}").into()), + }, + } + } + assert_eq!(refs(&fixture.handle).await?, before_refs); + assert_eq!(outcomes(&fixture.handle).await?, 1); + assert!(coordinator.close_and_drain().await.is_empty()); + assert_eq!(coordinator.reservations_for_test().await, (0, 0, 0)); + drop(failure); + drop(duplicate); + drop(extra); + drop(cross_class); + drop(push); + cleaned(push_root.path(), &push_budget).await?; + for (prepared, root, budget) in foreground { + drop(prepared); + cleaned(root.path(), &budget).await?; + } + for (compact, root, budget) in maintenance { + drop(compact); + cleaned(root.path(), &budget).await?; + } + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn queued_compaction_rechecks_admin_and_canceled_observer_cannot_cancel_publication() -> Result +{ + for revoke in [false, true] { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let inventory = seed(&fixture, 2).await?; + let prepared = prepare_compaction(&fixture, &inventory, 180, &[0, 1]).await?; + let compact = Arc::new(prepared.compact); + let weak = Arc::downgrade(&compact); + let coordinator = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let (release, entered) = coordinator.pause_for_test().await; + let ticket = coordinator + .submit(compact.ready_compaction(identity()?).await?) + .await?; + timeout(Duration::from_secs(5), entered).await??; + drop(ticket); + drop(compact); + assert!(weak.upgrade().is_some()); + if revoke { + edit(&fixture, "UPDATE repository_identity SET owner='other'; INSERT INTO repository_members VALUES('owner','write')").await?; + } + let before = refs(&fixture.handle).await?; + let observer = coordinator.pending([180; 16]).await.ok_or("job lost")?; + release.send(()).map_err(|_| "worker disappeared")?; + let outcome = timeout(Duration::from_secs(10), observer.wait()).await?; + if revoke { + assert!( + matches!(outcome, PublicationState::Finished(Err(ref error)) if matches!(error.as_ref(), PublicationError::Compaction(InvocationError::Rejected(value)) if value.output==CompactionReply::Denied(PreparationDenial::Unauthorized))) + ); + assert_eq!(outcomes(&fixture.handle).await?, 0); + } else { + compacted(outcome)?; + assert_eq!(outcomes(&fixture.handle).await?, 1); + } + assert_eq!(refs(&fixture.handle).await?, before); + assert!(weak.upgrade().is_none()); + assert!(coordinator.close_and_drain().await.is_empty()); + cleaned(prepared.root.path(), &prepared.budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn maintenance_concurrency_cap_keeps_foreground_progressing() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let inventory = seed(&fixture, 2).await?; + let first = prepare_compaction(&fixture, &inventory, 180, &[0, 1]).await?; + let second = prepare_compaction(&fixture, &inventory, 181, &[0, 1]).await?; + let first_compact = Arc::new(first.compact); + let second_compact = Arc::new(second.compact); + let foreground = + empty_in_store(&fixture, [90; 16], "owner", Arc::clone(&inventory.store)).await?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits { + in_flight: 2, + maintenance_in_flight: 1, + ..PublicationLimits::default() + }, + )?; + let (release, entered) = coordinator.pause_for_test().await; + let a = coordinator + .submit(first_compact.ready_compaction(identity()?).await?) + .await?; + timeout(Duration::from_secs(5), entered).await??; + let b = coordinator + .submit(second_compact.ready_compaction(identity()?).await?) + .await?; + let ready = Box::pin(foreground.0.ready_push( + identity()?, + request(refused()), + foreground.1.path(), + foreground.2.clone(), + limits(), + )) + .await?; + let push = coordinator.submit(ready).await?; + finished(timeout(Duration::from_secs(10), push.wait()).await?)?; + assert_eq!(push.response().await?, refused()); + assert!(matches!(a.state(), PublicationState::Running)); + assert!(matches!(b.state(), PublicationState::Queued)); + assert_eq!(coordinator.reservations_for_test().await, (2, 16 << 10, 1)); + release.send(()).map_err(|_| "worker disappeared")?; + compacted(timeout(Duration::from_secs(10), a.wait()).await?)?; + assert!( + matches!(timeout(Duration::from_secs(10), b.wait()).await?, PublicationState::Finished(Err(ref error)) if matches!(error.as_ref(), PublicationError::Compaction(InvocationError::Rejected(value)) if value.output==CompactionReply::Denied(PreparationDenial::Conflict))) + ); + assert_eq!(outcomes(&fixture.handle).await?, 1); + assert!(coordinator.close_and_drain().await.is_empty()); + drop(first_compact); + drop(second_compact); + drop(foreground.0); + cleaned(first.root.path(), &first.budget).await?; + cleaned(second.root.path(), &second.budget).await?; + cleaned(foreground.1.path(), &foreground.2).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/compaction/range.rs b/crates/canopy-server/src/packs/publication/tests/compaction/range.rs new file mode 100644 index 0000000..cc7ad15 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/compaction/range.rs @@ -0,0 +1,694 @@ +use super::*; +use crate::{ + ObjectId, + packs::directory::{ + StoredRun, + index::{IndexError, NodeRef}, + snapshot::{MAX_LEVELS, RunLoader}, + }, +}; + +pub(super) fn job_limits() -> CompactionLimits { + CompactionLimits { + spool: limits(), + output: crate::packs::metadata::MetadataLimits { + max_file_bytes: 16 << 10, + cache_kib: 16, + }, + ..CompactionLimits::default() + } +} +async fn prepare_range( + fixture: &Fixture, + inventory: &Inventory, + operation: u8, + source: CompactionSource, + after: Option, +) -> Result { + prepare_range_with_limits(fixture, inventory, operation, source, after, job_limits()).await +} +async fn prepare_range_with_limits( + fixture: &Fixture, + inventory: &Inventory, + operation: u8, + source: CompactionSource, + after: Option, + limits: CompactionLimits, +) -> Result { + let (base, files, indexes) = + opened(fixture, [operation; 16], Arc::clone(&inventory.store)).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let compact = + PreparedCompaction::prepare_range(root.path(), budget.clone(), base, source, after, limits) + .await? + .ok_or("empty job")?; + Ok(Prepared { + compact, + root, + budget, + files, + indexes, + }) +} +async fn directory( + inventory: &Inventory, + catalog: crate::packs::catalog::StoredCatalog, +) -> Result { + let catalog = CatalogSnapshot::download(&inventory.store, catalog).await?; + Ok(DirectorySnapshot::download(&inventory.store, catalog.directory).await?) +} +async fn runs(indexes: &CatalogIndexes, root: Option) -> Result> { + let mut cursor = indexes.ranges().cursor(root, None)?; + let mut result = Vec::new(); + while let Some(run) = cursor.next().await? { + result.push(run); + } + Ok(result) +} +async fn publish(fixture: &Fixture, compact: &PreparedCompaction) -> Result { + let result = fixture + .client() + .command::( + &fixture.target, + identity()?, + compact.certificate().await?, + ) + .await?; + assert!(matches!(result.output, CompactionReply::Published(_))); + Ok(()) +} +pub(super) async fn entries( + prepared: &Prepared, + catalog: crate::packs::catalog::StoredCatalog, +) -> Result> { + entries_for(&prepared.indexes, &prepared.files, catalog).await +} +pub(super) async fn entries_for( + indexes: &Arc, + files: &Arc, + catalog: crate::packs::catalog::StoredCatalog, +) -> Result> { + let reader = CatalogReader::open(Arc::clone(indexes), catalog).await?; + let directory = reader.directory(); + let mut ids = std::collections::BTreeSet::new(); + for root in directory + .level_zero + .iter() + .copied() + .map(Some) + .chain(directory.levels) + { + for stored in runs(indexes, root).await? { + let run = files.load(stored).await?; + let mut after = None; + loop { + let page = run.entries_after(after)?; + if page.is_empty() { + break; + } + after = page.last().map(|entry| entry.header.object.oid); + ids.extend(page.iter().map(|entry| entry.header.object.oid)); + } + } + } + let mut result = std::collections::BTreeMap::new(); + let ids = ids.into_iter().collect::>(); + for page in ids.chunks(512) { + let entries = reader + .directory() + .lookup_batch(indexes.ranges(), &**files, page) + .await?; + let headers = reader.headers(page, &**files, &**files).await?; + for ((oid, entry), header) in page.iter().zip(entries).zip(headers) { + let entry = entry.ok_or("object")?; + assert_eq!(header, Some(entry.header)); + result.insert(*oid, entry); + } + } + Ok(result) +} + +#[tokio::test] +async fn native_ingress_and_adjacent_level_promotion_preserve_inventory_and_refs() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let inventory = seed(&fixture, 2).await?; + let before = refs(&fixture.handle).await?; + let first = prepare_range( + &fixture, + &inventory, + 180, + CompactionSource::Ingress(0), + None, + ) + .await?; + assert_eq!(first.compact.input_count(), 1); + let old_catalog = first.compact.base().catalog.ok_or("base")?; + let old_directory = directory(&inventory, old_catalog).await?; + let original = runs(&first.indexes, Some(old_directory.level_zero[0])).await?; + let promoted = directory(&inventory, first.compact.catalog()).await?; + assert_eq!(promoted.level_zero.len(), 1); + assert_eq!(runs(&first.indexes, promoted.levels[0]).await?, original); + assert_eq!(first.budget.used(), 0); + assert_eq!(std::fs::read_dir(first.root.path())?.count(), 0); + assert_eq!( + entries(&first, old_catalog).await?, + entries(&first, first.compact.catalog()).await? + ); + publish(&fixture, &first.compact).await?; + + let second = prepare_range( + &fixture, + &inventory, + 181, + CompactionSource::Ingress(0), + None, + ) + .await?; + assert_eq!(second.compact.input_count(), 2); + let original_inventory = + entries(&second, second.compact.base().catalog.ok_or("base")?).await?; + assert_eq!( + entries(&second, second.compact.catalog()).await?, + original_inventory + ); + assert!( + directory(&inventory, second.compact.catalog()) + .await? + .level_zero + .is_empty() + ); + publish(&fixture, &second.compact).await?; + let third = + prepare_range(&fixture, &inventory, 182, CompactionSource::Level(0), None).await?; + let old = directory(&inventory, third.compact.base().catalog.ok_or("base")?).await?; + let new = directory(&inventory, third.compact.catalog()).await?; + assert!(new.levels[0].is_none()); + assert_eq!( + runs(&third.indexes, old.levels[0]).await?, + runs(&third.indexes, new.levels[1]).await? + ); + assert_eq!( + entries(&third, third.compact.catalog()).await?, + original_inventory + ); + publish(&fixture, &third.compact).await?; + assert_eq!(refs(&fixture.handle).await?, before); + assert_eq!(outcomes(&fixture.handle).await?, 3); + let (base, _, _) = opened(&fixture, [183; 16], Arc::clone(&inventory.store)).await?; + assert!( + PreparedCompaction::prepare_range( + third.root.path(), + third.budget.clone(), + base, + CompactionSource::Level(0), + None, + job_limits() + ) + .await? + .is_none() + ); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn native_partial_ranges_reuse_untouched_files_and_reconcile_disjoint_level_updates() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let provider: Arc = Arc::new(InMemory::new()); + let inventory = Inventory { + provider: Arc::clone(&provider), + store: Arc::new(ArtifactStore::new(provider, fixture.repository)), + }; + // Every native artifact and reader uses this exact store capability. + let graph = super::super::reconcile::graph_with_run_limits( + &fixture, + Arc::clone(&inventory.provider), + Arc::clone(&inventory.store), + [20; 16], + 260, + job_limits().output, + ) + .await?; + let proof = graph + .prepared + .ref_proof( + plan(vec![update("refs/heads/native", None, Some(graph.initial))]), + graph.root.path(), + graph.budget.clone(), + limits(), + ) + .await?; + fixture + .client() + .command::(&fixture.target, identity()?, proof) + .await?; + let before = refs(&fixture.handle).await?; + let first = prepare_range( + &fixture, + &inventory, + 180, + CompactionSource::Ingress(0), + None, + ) + .await?; + let original_catalog = first.compact.base().catalog.ok_or("base")?; + let original_directory = directory(&inventory, original_catalog).await?; + let original = runs(&first.indexes, Some(original_directory.level_zero[0])).await?; + assert!(original.len() > 3); + let canonical = entries(&first, original_catalog).await?; + let competing = prepare_range( + &fixture, + &inventory, + 181, + CompactionSource::Ingress(0), + Some(original[0].run.last_oid), + ) + .await?; + publish(&fixture, &first.compact).await?; + assert!(matches!( + competing.compact.reconcile().await, + Err(CatalogPreparationError::Catalog(IndexError::Stale)) + )); + let partial = directory(&inventory, first.compact.catalog()).await?; + assert_eq!( + runs(&first.indexes, Some(partial.level_zero[0])).await?, + original[1..] + ); + assert_eq!( + runs(&first.indexes, partial.levels[0]).await?, + original[..1] + ); + assert_eq!(entries(&first, first.compact.catalog()).await?, canonical); + let second = prepare_range( + &fixture, + &inventory, + 182, + CompactionSource::Ingress(0), + None, + ) + .await?; + publish(&fixture, &second.compact).await?; + let left = + prepare_range(&fixture, &inventory, 183, CompactionSource::Level(0), None).await?; + let right = prepare_range( + &fixture, + &inventory, + 184, + CompactionSource::Level(0), + Some(original[0].run.last_oid), + ) + .await?; + let certificate = right.compact.certificate().await?; + fixture + .client() + .command::(&fixture.target, identity()?, certificate) + .await?; + publish(&fixture, &left.compact).await?; + let rebound = right.compact.reconcile().await?; + assert_eq!(rebound.inventory_digest(), right.compact.inventory_digest()); + assert_eq!(rebound.token(), right.compact.token()); + publish(&fixture, &rebound).await?; + let final_directory = directory(&inventory, rebound.catalog()).await?; + assert!(final_directory.levels[0].is_none()); + assert_eq!( + runs(&right.indexes, final_directory.levels[1]).await?, + original[..2] + ); + assert_eq!( + runs(&right.indexes, Some(final_directory.level_zero[0])).await?, + original[2..] + ); + assert_eq!(entries(&right, rebound.catalog()).await?, canonical); + assert_eq!(entries(&right, original_catalog).await?, canonical); + assert_eq!(refs(&fixture.handle).await?, before); + for prepared in [first, competing, second, left, right] { + let Prepared { root, budget, .. } = prepared; + cleaned(root.path(), &budget).await?; + } + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn range_reconciliation_rejects_new_changed_or_missing_overlapping_target_inputs() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let inventory = seed(&fixture, 2).await?; + let first = prepare_range( + &fixture, + &inventory, + 180, + CompactionSource::Ingress(0), + None, + ) + .await?; + let second = prepare_range( + &fixture, + &inventory, + 181, + CompactionSource::Ingress(1), + None, + ) + .await?; + publish(&fixture, &first.compact).await?; + assert!(matches!( + second.compact.reconcile().await, + Err(CatalogPreparationError::Catalog(IndexError::Stale)) + )); + let overlap = prepare_range( + &fixture, + &inventory, + 182, + CompactionSource::Ingress(0), + None, + ) + .await?; + assert_eq!(overlap.compact.input_count(), 2); + push(&fixture, &inventory, 99, 6).await?; + let changed = prepare_range( + &fixture, + &inventory, + 184, + CompactionSource::Ingress(1), + None, + ) + .await?; + publish(&fixture, &changed.compact).await?; + assert!(matches!( + overlap.compact.reconcile().await, + Err(CatalogPreparationError::Catalog(IndexError::Stale)) + )); + let overlap = prepare_range( + &fixture, + &inventory, + 185, + CompactionSource::Ingress(0), + None, + ) + .await?; + let promote = + prepare_range(&fixture, &inventory, 183, CompactionSource::Level(0), None).await?; + publish(&fixture, &promote.compact).await?; + assert!(matches!( + overlap.compact.reconcile().await, + Err(CatalogPreparationError::Catalog(IndexError::Stale)) + )); + assert_eq!(outcomes(&fixture.handle).await?, 3); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn range_job_rejects_limits_and_invalid_positions_without_scratch_or_outcomes() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let inventory = seed(&fixture, 2).await?; + let first = prepare_range( + &fixture, + &inventory, + 180, + CompactionSource::Ingress(0), + None, + ) + .await?; + publish(&fixture, &first.compact).await?; + let (base, _, _) = opened(&fixture, [181; 16], Arc::clone(&inventory.store)).await?; + let root = tempfile::TempDir::new()?; + let current = base.catalog_parts().0; + let source = first + .indexes + .ranges() + .cursor(Some(current.level_zero[0]), None)? + .next() + .await? + .ok_or("source")?; + let target = first + .indexes + .ranges() + .cursor(current.levels[0], None)? + .next() + .await? + .ok_or("target")?; + for limits in [ + CompactionLimits { + input_bytes: source.run.size + target.run.size - 1, + ..job_limits() + }, + CompactionLimits { + input_runs: 1, + ..job_limits() + }, + CompactionLimits { + input_bytes: 1, + ..job_limits() + }, + ] { + let budget = DiskBudget::new(128 << 20); + let result = PreparedCompaction::prepare_range( + root.path(), + budget.clone(), + Arc::clone(&base), + CompactionSource::Ingress(0), + None, + limits, + ) + .await; + if source.coverage.first_oid < target.coverage.first_oid + && limits.input_bytes >= source.run.size + { + // The budget can still admit the nonoverlapping prefix before the + // first target. Verify that this progress preserves the exact suffix. + let prepared = result?.ok_or("prefix")?; + assert_eq!(prepared.input_count(), 1); + let next = directory(&inventory, prepared.catalog()).await?; + let remainder = runs(&first.indexes, Some(next.level_zero[0])).await?; + assert_eq!(remainder[0].run, source.run); + assert_eq!(remainder[0].artifact, source.artifact); + assert!(remainder[0].coverage.object_count < source.coverage.object_count); + assert!( + runs(&first.indexes, next.levels[0]) + .await? + .contains(&target) + ); + prepared.certificate().await?; + } else { + assert!(result.is_err()); + } + cleaned(root.path(), &budget).await?; + } + for source in [ + CompactionSource::Ingress(usize::MAX), + CompactionSource::Level(usize::MAX), + CompactionSource::Level(MAX_LEVELS - 1), + ] { + assert!( + PreparedCompaction::prepare_range( + root.path(), + DiskBudget::new(1), + Arc::clone(&base), + source, + None, + job_limits() + ) + .await + .is_err() + ); + } + let wrong: ObjectId = vec![1; 32].try_into()?; + assert!( + PreparedCompaction::prepare_range( + root.path(), + DiskBudget::new(1), + Arc::clone(&base), + CompactionSource::Ingress(0), + Some(wrong), + job_limits() + ) + .await + .is_err() + ); + let after = base.catalog_parts().0.level_zero[0].last_key; + assert!( + PreparedCompaction::prepare_range( + root.path(), + DiskBudget::new(1), + base, + CompactionSource::Ingress(0), + Some(after), + job_limits() + ) + .await? + .is_none() + ); + assert_eq!(outcomes(&fixture.handle).await?, 1); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn native_range_jobs_partition_large_inputs_and_merge_multiple_target_files() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let inventory = seed(&fixture, 0).await?; + push(&fixture, &inventory, 20, 260).await?; + let first = prepare_range( + &fixture, + &inventory, + 180, + CompactionSource::Ingress(0), + None, + ) + .await?; + let original = first.compact.base().catalog.ok_or("base")?; + let old = directory(&inventory, original).await?; + let old_runs = runs(&first.indexes, Some(old.level_zero[0])).await?; + assert_eq!(old_runs.len(), 1); + assert!(old_runs[0].run.size > job_limits().output.max_file_bytes); + let new = directory(&inventory, first.compact.catalog()).await?; + let targets = runs(&first.indexes, new.levels[0]).await?; + assert!(targets.len() > 3); + assert!( + targets + .iter() + .all(|run| run.run.size <= job_limits().output.max_file_bytes) + ); + assert_eq!( + entries(&first, original).await?, + entries(&first, first.compact.catalog()).await? + ); + publish(&fixture, &first.compact).await?; + push(&fixture, &inventory, 21, 261).await?; + let second = prepare_range( + &fixture, + &inventory, + 181, + CompactionSource::Ingress(0), + None, + ) + .await?; + assert!(second.compact.input_count() > 2); + let original = second.compact.base().catalog.ok_or("base")?; + assert_eq!( + entries(&second, original).await?, + entries(&second, second.compact.catalog()).await? + ); + let before = refs(&fixture.handle).await?; + publish(&fixture, &second.compact).await?; + assert_eq!(refs(&fixture.handle).await?, before); + assert_eq!( + CatalogSnapshot::download(&inventory.store, original) + .await? + .sources, + CatalogSnapshot::download(&inventory.store, second.compact.catalog()) + .await? + .sources + ); + assert!( + directory(&inventory, second.compact.catalog()) + .await? + .level_zero + .is_empty() + ); + assert_eq!(outcomes(&fixture.handle).await?, 2); + for prepared in [first, second] { + let Prepared { root, budget, .. } = prepared; + cleaned(root.path(), &budget).await?; + } + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn bounded_windows_finish_native_ingress_without_rewriting_suffix_files_or_losing_objects() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let inventory = seed(&fixture, 0).await?; + push(&fixture, &inventory, 20, 260).await?; + let first = prepare_range( + &fixture, + &inventory, + 180, + CompactionSource::Ingress(0), + None, + ) + .await?; + let target = directory(&inventory, first.compact.catalog()).await?; + let targets = runs(&first.indexes, target.levels[0]).await?; + assert!(targets.len() > 3); + publish(&fixture, &first.compact).await?; + push(&fixture, &inventory, 21, 261).await?; + let limits = CompactionLimits { + input_runs: 2, + ..job_limits() + }; + let competing = prepare_range_with_limits( + &fixture, + &inventory, + 181, + CompactionSource::Ingress(0), + None, + limits, + ) + .await?; + let original = competing.compact.base().catalog.ok_or("base")?; + let original_directory = directory(&inventory, original).await?; + let source = runs(&competing.indexes, Some(original_directory.level_zero[0])).await?[0]; + let canonical = entries(&competing, original).await?; + let before = refs(&fixture.handle).await?; + let mut previous = source.coverage.object_count; + let mut jobs = 0; + loop { + jobs += 1; + assert!(jobs <= targets.len() + 2, "no bounded progress"); + let prepared = prepare_range_with_limits( + &fixture, + &inventory, + 182 + jobs as u8, + CompactionSource::Ingress(0), + None, + limits, + ) + .await?; + assert!(prepared.compact.input_count() <= 2); + let next = directory(&inventory, prepared.compact.catalog()).await?; + if let Some(root) = next.level_zero.first() { + let remainder = runs(&prepared.indexes, Some(*root)).await?; + assert_eq!(remainder.len(), 1); + assert_eq!(remainder[0].run, source.run); + assert_eq!(remainder[0].artifact, source.artifact); + assert!(remainder[0].coverage.object_count < previous); + assert_eq!(remainder[0].coverage.last_oid, source.coverage.last_oid); + previous = remainder[0].coverage.object_count; + } + assert_eq!( + entries(&prepared, prepared.compact.catalog()).await?, + canonical + ); + publish(&fixture, &prepared.compact).await?; + assert_eq!(refs(&fixture.handle).await?, before); + if jobs == 1 { + assert!(matches!( + competing.compact.reconcile().await, + Err(CatalogPreparationError::Catalog(IndexError::Stale)) + )); + } + let done = next.level_zero.is_empty(); + let Prepared { root, budget, .. } = prepared; + cleaned(root.path(), &budget).await?; + if done { + break; + } + } + assert!(jobs > 1); + assert_eq!(outcomes(&fixture.handle).await?, jobs as u64 + 1); + assert_eq!(entries(&competing, original).await?, canonical); + fixture.runtime.shutdown().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/compaction/recovery.rs b/crates/canopy-server/src/packs/publication/tests/compaction/recovery.rs new file mode 100644 index 0000000..c086f7f --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/compaction/recovery.rs @@ -0,0 +1,188 @@ +use super::*; + +#[tokio::test] +async fn compaction_final_validation_rejects_changed_floor_base_and_expired_attempts() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let inventory = seed(&fixture, 2).await?; + let prepared = prepare_compaction(&fixture, &inventory, 180, &[0, 1]).await?; + let certificate = prepared.compact.certificate().await?; + let mut data = certificate.data()?; + data.retention_floor = 0; + data.retention_certificate = None; + reject( + &fixture, + CatalogCertificate::seal(&data, &[16; 32])?, + PreparationDenial::Conflict, + ) + .await?; + data = certificate.data()?; + data.base.certificate = Some([99; 32]); + reject( + &fixture, + CatalogCertificate::seal(&data, &[16; 32])?, + PreparationDenial::Conflict, + ) + .await?; + edit(&fixture,"UPDATE catalog_leases SET expires_at_ms=0 WHERE artifact_operation=(SELECT artifact_operation FROM catalog_operations ORDER BY id DESC LIMIT 1); UPDATE catalog_operations SET expires_at_ms=0 WHERE id=(SELECT id FROM catalog_operations ORDER BY id DESC LIMIT 1);").await?; + reject(&fixture, certificate, PreparationDenial::Expired).await?; + assert_eq!(outcomes(&fixture.handle).await?, 0); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn compaction_ack_survives_owner_restore_and_old_fences_cannot_write() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let inventory = seed(&fixture, 3).await?; + let first = prepare_compaction(&fixture, &inventory, 180, &[0, 1]).await?; + let second = prepare_compaction(&fixture, &inventory, 181, &[1, 2]).await?; + let certificate = first.compact.certificate().await?; + let pending = second.compact.certificate().await?; + let mutation = identity()?; + let committed = fixture + .client() + .command::(&fixture.target, mutation, certificate.clone()) + .await?; + let before = state(&fixture.handle).await?; + fixture.handle.drain().await?; + fixture.runtime.shutdown().await?; + let session = SessionId::from_bytes([182; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, session)?; + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await? + .ok_or("idle")?; + let provision = CellCatalog::new(fixture.layout.clone(), fixture.target.tenant()) + .lookup(fixture.target.cell_id()) + .await? + .ok_or("provision")?; + let handle = runtime + .acquire_idle_restored( + provision, + fixture.replica.clone(), + authority, + idle, + fixture.root.path().join("compaction-restored.sqlite"), + Owner { + session, + endpoint: "https://compaction-owner-b.invalid".into(), + }, + ) + .await?; + assert!(handle.owner_fence().epoch > first.compact.token().owner.epoch); + let client = CellClient::local(Arc::clone(&fixture.registry), handle.clone()); + assert_eq!(state(&handle).await?, before); + let replay = client + .command::(&fixture.target, mutation, certificate.clone()) + .await?; + assert_eq!(replay.receipt, committed.receipt); + assert_eq!(replay.output, committed.output); + assert_eq!( + client + .command::(&fixture.target, identity()?, certificate) + .await? + .output, + committed.output + ); + assert_eq!( + client + .query::(&fixture.target, None, fixture.begin([180; 16])) + .await? + .output, + Some(committed.output) + ); + let stale = client + .command::(&fixture.target, identity()?, pending) + .await; + assert!( + matches!(stale,Err(InvocationError::Rejected(ref value)) if value.output==CompactionReply::Denied(PreparationDenial::Stale)), + "{stale:?}" + ); + assert_eq!(state(&handle).await?, before); + assert_eq!(outcomes(&handle).await?, 1); + runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn invalid_selection_and_admission_failure_produce_no_compaction_workspace_or_proof() -> Result +{ + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let inventory = seed(&fixture, 2).await?; + let (base, _, _) = opened(&fixture, [180; 16], Arc::clone(&inventory.store)).await?; + let root = tempfile::TempDir::new()?; + for selected in [&[0][..], &[0, 0][..], &[0, 99][..]] { + let budget = DiskBudget::new(128 << 20); + assert!( + PreparedCompaction::prepare( + root.path(), + budget.clone(), + Arc::clone(&base), + selected, + CompactionLimits::default() + ) + .await + .is_err() + ); + cleaned(root.path(), &budget).await?; + } + for limits in [ + CompactionLimits { + input_runs: 1, + spool: limits(), + ..CompactionLimits::default() + }, + CompactionLimits { + input_bytes: 1, + spool: limits(), + ..CompactionLimits::default() + }, + ] { + let budget = DiskBudget::new(128 << 20); + assert!(matches!( + PreparedCompaction::prepare( + root.path(), + budget.clone(), + Arc::clone(&base), + &[0, 1], + limits + ) + .await, + Err(CatalogPreparationError::Metadata( + crate::packs::metadata::MetadataError::Limit + )) + )); + cleaned(root.path(), &budget).await?; + } + let budget = DiskBudget::new(1); + assert!( + PreparedCompaction::prepare( + root.path(), + budget.clone(), + Arc::clone(&base), + &[0, 1], + CompactionLimits::default() + ) + .await + .is_err() + ); + cleaned(root.path(), &budget).await?; + edit(&fixture,"UPDATE repository_identity SET owner='other'; INSERT INTO repository_members VALUES('owner','write');").await?; + let budget = DiskBudget::new(128 << 20); + assert!( + PreparedCompaction::prepare( + root.path(), + budget.clone(), + base, + &[0, 1], + CompactionLimits::default() + ) + .await + .is_err() + ); + cleaned(root.path(), &budget).await?; + assert_eq!(outcomes(&fixture.handle).await?, 0); + fixture.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/compaction/schedule.rs b/crates/canopy-server/src/packs/publication/tests/compaction/schedule.rs new file mode 100644 index 0000000..9170ffc --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/compaction/schedule.rs @@ -0,0 +1,172 @@ +use super::*; + +#[tokio::test] +async fn geometric_planner_drains_native_ingress_and_level_debt_without_changing_refs() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let inventory = seed(&fixture, 4).await?; + let original_refs = refs(&fixture.handle).await?; + let policy = CompactionPolicy { + base_objects: 2, + level_ratio: 2, + ingress_high_water: 2, + urgent_burst: 2, + }; + let mut planner = CompactionPlanner::new(policy)?; + let coordinator = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let mut expected = None; + let mut first_catalog = None; + let mut jobs = 0; + let mut done = false; + for turn in 0..50 { + let (base, files, indexes) = + opened(&fixture, [180 + turn; 16], Arc::clone(&inventory.store)).await?; + let before = base.generation_fact().catalog.ok_or("catalog")?; + let current = base.catalog_parts().0; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let prepared = planner + .prepare_next(root.path(), budget.clone(), base, range::job_limits()) + .await?; + if let Some(compact) = prepared { + let prepared = Prepared { + compact, + root, + budget, + files, + indexes, + }; + let prior = range::entries(&prepared, before).await?; + if expected.is_none() { + expected = Some(prior.clone()); + first_catalog = Some(before); + } + assert_eq!(Some(&prior), expected.as_ref()); + let after = range::entries(&prepared, prepared.compact.catalog()).await?; + assert_eq!(after, prior); + let compact = Arc::new(prepared.compact); + let ticket = coordinator + .submit(compact.ready_compaction(identity()?).await?) + .await?; + let result = ticket.wait().await; + assert!( + matches!(result, PublicationState::Finished(Ok(PublicationOutcome::Compaction(value))) if matches!(value.output, CompactionReply::Published(_))) + ); + assert_eq!(refs(&fixture.handle).await?, original_refs); + assert_eq!( + Some( + range::entries_for( + &prepared.indexes, + &prepared.files, + first_catalog.ok_or("old")? + ) + .await? + ), + expected + ); + jobs += 1; + drop(compact); + cleaned(prepared.root.path(), &prepared.budget).await?; + } else { + let pressure = policy.pressure(¤t)?; + assert_eq!(pressure.ingress_roots, 0); + assert!( + pressure + .level_objects + .iter() + .zip(pressure.level_targets) + .all(|(objects, target)| *objects <= target) + ); + assert!(current.levels.iter().skip(1).any(Option::is_some)); + cleaned(root.path(), &budget).await?; + done = true; + break; + } + } + assert!(done, "geometric debt must drain in the bounded fixture"); + assert!(jobs > 4, "exercise both ingress and higher-level jobs"); + assert_eq!(outcomes(&fixture.handle).await?, jobs); + assert!(coordinator.close_and_drain().await.is_empty()); + assert_eq!(coordinator.stats().await.admitted, 0); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn planner_admission_failure_retries_the_same_ingress_without_publishing() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let inventory = seed(&fixture, 2).await?; + let (base, _, _) = opened(&fixture, [180; 16], Arc::clone(&inventory.store)).await?; + let original = state(&fixture.handle).await?; + let original_catalog = base.generation_fact().catalog; + let retained = base.catalog_parts().0.level_zero[1]; + let mut planner = CompactionPlanner::new(CompactionPolicy::default())?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + assert!( + planner + .prepare_next( + root.path(), + budget.clone(), + Arc::clone(&base), + CompactionLimits { + input_bytes: 1, + ..range::job_limits() + } + ) + .await + .is_err() + ); + assert_eq!(state(&fixture.handle).await?, original); + cleaned(root.path(), &budget).await?; + let retry = planner + .prepare_next(root.path(), budget.clone(), base, range::job_limits()) + .await? + .ok_or("retry")?; + assert_eq!(retry.base().catalog, original_catalog); + let catalog = CatalogSnapshot::download(&inventory.store, retry.catalog()).await?; + let directory = DirectorySnapshot::download(&inventory.store, catalog.directory).await?; + assert_eq!(directory.level_zero, vec![retained]); + assert_eq!(outcomes(&fixture.handle).await?, 0); + drop(retry); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn empty_planner_rechecks_current_admin_access_before_reporting_no_work() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let inventory = seed(&fixture, 0).await?; + let (base, files, _) = opened(&fixture, [180; 16], Arc::clone(&inventory.store)).await?; + let mut planner = CompactionPlanner::new(CompactionPolicy::default())?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + assert!( + planner + .prepare_next( + root.path(), + budget.clone(), + Arc::clone(&base), + range::job_limits() + ) + .await? + .is_none() + ); + edit(&fixture, "UPDATE repository_identity SET owner='other'; INSERT INTO repository_members VALUES('owner','write');").await?; + let original = state(&fixture.handle).await?; + assert!( + planner + .prepare_next(root.path(), budget.clone(), base, range::job_limits()) + .await + .is_err() + ); + assert_eq!(state(&fixture.handle).await?, original); + assert_eq!(files.stats()?.downloaded_files, 0); + assert_eq!(outcomes(&fixture.handle).await?, 0); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/completion.rs b/crates/canopy-server/src/packs/publication/tests/completion.rs new file mode 100644 index 0000000..7269fff --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/completion.rs @@ -0,0 +1,855 @@ +use super::*; +use super::{ + prepare::cleaned, + publishing::{Graph, assembled, edit, next_graph, plan, state, update}, +}; +use crate::{ + git_http::GitHttpResponse, packs::metadata::tests::limits, push::VerifiedPushCertificate, +}; +use sha2::{Digest as _, Sha256}; + +pub(super) fn packet(out: &mut Vec, body: &[u8]) { + assert!(body.len() + 4 <= 65520); + out.extend_from_slice(format!("{:04x}", body.len() + 4).as_bytes()); + out.extend_from_slice(body); +} +fn response(plan: &crate::PushPlan, sideband: bool, progress: usize) -> GitHttpResponse { + let mut report = Vec::new(); + packet(&mut report, b"unpack ok\n"); + for update in &plan.updates { + packet(&mut report, format!("ok {}\n", update.name).as_bytes()); + } + packet(&mut report, b"ng refs/heads/hook-refused hook declined\n"); + report.extend_from_slice(b"0000"); + let body = if sideband { + let mut body = Vec::new(); + for _ in 0..progress { + packet(&mut body, &[&[2u8][..], &vec![b'x'; 60_000]].concat()); + } + for chunk in report.chunks(37) { + packet(&mut body, &[&[1u8][..], chunk].concat()); + } + body.extend_from_slice(b"0000"); + body + } else { + report + }; + GitHttpResponse { + status: 200, + headers: vec![ + ( + "Content-Type".into(), + "application/x-git-receive-pack-result".into(), + ), + ("X-Native-Trace".into(), "preserved".into()), + ], + body, + } +} +async fn completion( + graph: &Graph, + request: PushCompletionRequest, +) -> Result { + Ok(Box::pin(graph.prepared.push_completion( + request, + graph.root.path(), + graph.budget.clone(), + limits(), + )) + .await?) +} +fn request( + graph: &Graph, + name: &str, + sideband: bool, + progress: usize, + signed: bool, +) -> PushCompletionRequest { + let plan = plan(vec![update(name, None, Some(graph.initial))]); + PushCompletionRequest { + response: response(&plan, sideband, progress), + plan: Some(plan), + options: vec!["canopy.note=reviewed".into()], + certificate: signed.then(|| VerifiedPushCertificate { + target: graph.prepared.base.capability().1.clone(), + request_digest: graph.prepared.token().request_digest, + body: vec![b's'; 600_000], + signer: "owner".into(), + key: "native-verified-key".into(), + }), + } +} +fn completed(reply: CatalogCompletionReply) -> Result { + match reply { + CatalogCompletionReply::Completed(value) => Ok(value), + CatalogCompletionReply::Denied(reason) => Err(format!("denied {reason:?}").into()), + } +} +async fn stored(handle: &CellHandle, id: [u8; 16]) -> Result { + let bytes = handle + .query(0, 2 << 20, move |connection| { + let (status, headers, size, digest): (u16, String, usize, Vec) = connection + .query_row( + "SELECT status,headers,size,digest FROM push_responses WHERE id=?1", + [id.as_slice()], + |row| Ok((row.get(0)?, row.get(1)?, row.get(2)?, row.get(3)?)), + )?; + let mut statement = connection.prepare( + "SELECT part,body FROM push_response_chunks WHERE response_id=?1 ORDER BY part", + )?; + let mut body = Vec::new(); + for (at, row) in statement + .query_map([id.as_slice()], |row| { + Ok((row.get::<_, usize>(0)?, row.get::<_, Vec>(1)?)) + })? + .enumerate() + { + let (part, chunk) = row?; + if part != at { + return Err(Error::Command("stored response chunk gap")); + } + body.extend_from_slice(&chunk); + } + if body.len() != size || blake3::hash(&body).as_bytes().as_slice() != digest { + return Err(Error::Command("stored response digest")); + } + let mut encoder = BoundedEncoder::new(2 << 20)?; + encoder.write_u32(u32::from(status))?; + encoder.write_text(&headers)?; + encoder.write_bytes(&body)?; + Ok(encoder.finish()) + }) + .await?; + let mut decoder = BoundedDecoder::new(&bytes, 2 << 20)?; + let status = u16::try_from(decoder.read_u32()?)?; + let headers = serde_json::from_str(decoder.read_text()?)?; + let body = decoder.read_bytes()?.to_vec(); + decoder.finish()?; + Ok(GitHttpResponse { + status, + headers, + body, + }) +} +pub(super) async fn counts(handle: &CellHandle) -> Result> { + let bytes = handle + .query(0, 1024, |connection| { + let counts = [ + "pushes", + "push_responses", + "push_response_chunks", + "push_certificates", + "push_certificate_chunks", + ] + .into_iter() + .map(|table| { + connection.query_row(&format!("SELECT count(*) FROM {table}"), [], |row| { + row.get::<_, u64>(0) + }) + }) + .collect::>>()?; + serde_json::to_vec(&counts).map_err(|_| Error::Command("outcome counts")) + }) + .await?; + Ok(serde_json::from_slice(&bytes)?) +} +async fn reject( + fixture: &Fixture, + input: CatalogPushCompletion, + reason: PreparationDenial, +) -> Result { + let before = state(&fixture.handle).await?; + let outcomes = counts(&fixture.handle).await?; + let result = Box::pin(fixture.client().command::( + &fixture.target, + identity()?, + input, + )) + .await; + assert!( + matches!(result,Err(InvocationError::Rejected(ref value)) if value.output==CatalogCompletionReply::Denied(reason)), + "{result:?}" + ); + assert_eq!(state(&fixture.handle).await?, before); + assert_eq!(counts(&fixture.handle).await?, outcomes); + Ok(()) +} + +#[tokio::test] +async fn exact_native_response_options_and_signed_bytes_commit_with_catalog_refs() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let graph = assembled(&fixture, [91; 16], 0).await?; + let before = state(&fixture.handle).await?; + let mut native = request(&graph, "refs/heads/開発", true, 10, true); + // Maximum legal note count with JSON escaping exceeds the old 32 KiB + // table constraint. Reuse options with a correctly bounded 64 KiB field. + native.options = vec![format!("canopy.note={}", "\\".repeat(1012)); 16]; + let expected = native.response.clone(); + let options = native.options.clone(); + let input = completion(&graph, native).await?; + assert_eq!(state(&fixture.handle).await?, before); + assert_eq!(counts(&fixture.handle).await?, vec![0; 5]); + let mut encoder = BoundedEncoder::new(4 << 20)?; + input.encode(&mut encoder)?; + let bytes = encoder.finish(); + let mut decoder = BoundedDecoder::new(&bytes, 4 << 20)?; + assert_eq!(CatalogPushCompletion::decode(&mut decoder)?, input); + decoder.finish()?; + let mutation = identity()?; + let first = Box::pin(fixture.client().command::( + &fixture.target, + mutation, + input.clone(), + )) + .await?; + let output = completed(first.output)?; + assert!(!output.rejected); + assert_eq!( + output + .publication + .map(|value| (value.generation, value.ref_generation)), + Some((1, 1)) + ); + assert_eq!(stored(&fixture.handle, output.response_id).await?, expected); + assert_eq!( + graph.prepared.completed_push_response(&first).await?, + expected + ); + assert_eq!(counts(&fixture.handle).await?, vec![1, 1, 2, 1, 2]); + let operation = graph.prepared.token().operation; + fixture.handle.query(0,1<<20,move|connection| { + let (saved_options,digest,size):(String,Vec,usize)=connection.query_row("SELECT p.options,c.digest,c.size FROM pushes p JOIN push_certificates c ON c.push_id=p.id WHERE p.id=?1",[operation.as_slice()],|row|Ok((row.get(0)?,row.get(1)?,row.get(2)?)))?; + assert_eq!(serde_json::from_str::>(&saved_options).unwrap(),options); + let mut statement=connection.prepare("SELECT body FROM push_certificate_chunks WHERE push_id=?1 ORDER BY part")?; + let body=statement.query_map([operation.as_slice()],|row|row.get::<_,Vec>(0))?.collect::>>()?.concat(); + assert_eq!(body,vec![b's';600_000]); + assert_eq!(size,body.len()); + assert_eq!(digest,Sha256::digest(&body).as_slice()); + Ok(Vec::new()) + }).await?; + let after = state(&fixture.handle).await?; + let replay = Box::pin(fixture.client().command::( + &fixture.target, + mutation, + input.clone(), + )) + .await?; + assert_eq!(replay.output, first.output); + assert_eq!(replay.receipt, first.receipt); + let logical = Box::pin(fixture.client().command::( + &fixture.target, + identity()?, + input, + )) + .await?; + assert_eq!(logical.output, first.output); + assert_eq!(state(&fixture.handle).await?, after); + assert_eq!(stored(&fixture.handle, output.response_id).await?, expected); + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn completion_payload_tampering_and_stripping_cannot_publish() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let graph = assembled(&fixture, [92; 16], 0).await?; + let input = completion(&graph, request(&graph, "refs/heads/main", false, 0, true)).await?; + for at in 0..8 { + let mut bad = input.clone(); + match at { + 0 => bad.response_id = [99; 16], + 1 => bad.response.status = 201, + 2 => bad.response.headers.push(("X-Added".into(), "x".into())), + 3 => bad.response.body.push(b'x'), + 4 => bad.options = vec!["canopy.note=altered".into()], + 5 => bad.signed = None, + 6 => bad.signed.as_mut().unwrap().body[0] ^= 1, + _ => bad.signed.as_mut().unwrap().key.push('x'), + } + reject(&fixture, bad, PreparationDenial::Unauthorized).await?; + } + let CompletionCatalogProof::Refs(proof) = input.proof else { + panic!("refs") + }; + // A caller cannot strip the network payload and commit only its refs. + let result = fixture + .client() + .command::(&fixture.target, identity()?, proof) + .await; + assert!( + matches!(result,Err(InvocationError::Rejected(ref value)) if value.output==PublicationReply::Denied(PreparationDenial::Unauthorized)) + ); + assert_eq!(counts(&fixture.handle).await?, vec![0; 5]); + let native = request(&graph, "refs/heads/main", false, 0, false); + for body in [b"0000".to_vec(), b"0014unpack ok\n0000".to_vec()] { + // A bare flush is permitted when report-status was declined; a malformed + // report never becomes an acknowledged publication. + if body == b"0000" { + continue; + } + assert!( + completion( + &graph, + PushCompletionRequest { + response: GitHttpResponse { + body, + ..native.response.clone() + }, + plan: native.plan.clone(), + options: vec![], + certificate: None + } + ) + .await + .is_err() + ); + } + let mut extra = native.response.clone(); + let mut report = Vec::new(); + packet(&mut report, b"unpack ok\n"); + packet(&mut report, b"ok refs/heads/other\n"); + report.extend_from_slice(b"0000"); + extra.body = report; + assert!( + completion( + &graph, + PushCompletionRequest { + response: extra, + plan: native.plan, + options: vec![], + certificate: None + } + ) + .await + .is_err() + ); + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn final_policy_and_permission_refusals_save_exact_rejections_without_publishing() -> Result { + for (revoke, report_status) in [(false, true), (true, true), (false, false), (true, false)] { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let graph = assembled(&fixture, [93; 16], 0).await?; + let mut native = request(&graph, "refs/heads/main", true, 1, false); + if !report_status { + native.response.body.clear(); + } + let input = completion(&graph, native).await?; + let expected = + crate::push::report::rejected_report(&input.response, crate::push::report::REJECTED)?; + if revoke { + edit( + &fixture, + "UPDATE repository_identity SET owner='successor';", + ) + .await?; + } else { + edit( + &fixture, + "INSERT INTO branch_rules(reference,version,enabled,deny_deletions,fast_forward,require_pull_request,required_approvals) VALUES('refs/heads/main',1,1,0,1,1,0);", + ) + .await?; + } + let mutation = identity()?; + let first = fixture + .client() + .command::(&fixture.target, mutation, input.clone()) + .await?; + let output = completed(first.output)?; + assert!(output.rejected); + assert_eq!(output.publication, None); + assert_eq!( + graph.prepared.completed_push_response(&first).await?, + expected + ); + assert_eq!(stored(&fixture.handle, output.response_id).await?, expected); + fixture + .handle + .query(0, 1024, |connection| { + assert_eq!( + connection + .query_row("SELECT generation FROM catalog_state", [], |row| row + .get::<_, u64>(0))?, + 0 + ); + assert_eq!( + connection + .query_row("SELECT count(*) FROM refs", [], |row| row.get::<_, u64>(0))?, + 0 + ); + Ok(Vec::new()) + }) + .await?; + assert_eq!(fixture.counts().await?, (0, 1)); + let logical = fixture + .client() + .command::(&fixture.target, identity()?, input) + .await?; + assert_eq!(logical.output, first.output); + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn late_response_failure_rolls_back_catalog_refs_certificate_and_all_chunks() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let graph = assembled(&fixture, [94; 16], 0).await?; + let input = completion(&graph, request(&graph, "refs/heads/main", true, 10, true)).await?; + edit(&fixture,"CREATE TRIGGER fail_last_response BEFORE INSERT ON push_response_chunks WHEN NEW.part=1 BEGIN SELECT RAISE(ABORT,'late response chunk failure'); END;").await?; + let before = state(&fixture.handle).await?; + let failed = Box::pin(fixture.client().command::( + &fixture.target, + identity()?, + input.clone(), + )) + .await; + assert!(failed.is_err(), "{failed:?}"); + assert_eq!(state(&fixture.handle).await?, before); + assert_eq!(counts(&fixture.handle).await?, vec![0; 5]); + assert_eq!(fixture.counts().await?, (1, 1)); + edit(&fixture, "DROP TRIGGER fail_last_response;").await?; + let committed = Box::pin(fixture.client().command::( + &fixture.target, + identity()?, + input.clone(), + )) + .await?; + let output = completed(committed.output)?; + assert!(output.publication.is_some()); + assert_eq!( + stored(&fixture.handle, output.response_id).await?, + input.response + ); + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn signed_certificate_replay_is_a_durable_refusal_and_outcome_bytes_are_immutable() -> Result +{ + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let graph = assembled(&fixture, [95; 16], 0).await?; + let input = completion(&graph, request(&graph, "refs/heads/main", false, 0, true)).await?; + let first = completed( + fixture + .client() + .command::(&fixture.target, identity()?, input) + .await? + .output, + )?; + let next = next_graph(&fixture, &graph, [96; 16]).await?; + let input = completion(&next, request(&next, "refs/heads/second", false, 0, true)).await?; + let expected = crate::push::report::rejected_report( + &input.response, + "Canopy signed push certificate was already used", + )?; + let refusal = completed( + fixture + .client() + .command::(&fixture.target, identity()?, input) + .await? + .output, + )?; + assert!(refusal.rejected); + assert_eq!(refusal.publication, None); + assert_eq!( + stored(&fixture.handle, refusal.response_id).await?, + expected + ); + assert_eq!(counts(&fixture.handle).await?, vec![2, 2, 2, 1, 2]); + for sql in [ + "UPDATE pushes SET options='[]' WHERE rejected=0;", + "UPDATE pushes SET completion_digest=zeroblob(32);", + "UPDATE pushes SET rejected=1 WHERE rejected=0;", + "UPDATE push_responses SET digest=zeroblob(32);", + "UPDATE push_response_chunks SET body=x'00';", + "UPDATE push_certificate_chunks SET body=x'00';", + "UPDATE push_certificates SET key='changed';", + "INSERT OR REPLACE INTO push_responses SELECT * FROM push_responses;", + "INSERT OR REPLACE INTO push_response_chunks SELECT * FROM push_response_chunks;", + "INSERT OR REPLACE INTO push_certificates SELECT * FROM push_certificates;", + "INSERT OR REPLACE INTO push_certificate_chunks SELECT * FROM push_certificate_chunks;", + ] { + assert!(edit(&fixture, sql).await.is_err(), "{sql}"); + } + assert_eq!( + stored(&fixture.handle, refusal.response_id).await?, + expected + ); + assert!( + !stored(&fixture.handle, first.response_id) + .await? + .body + .is_empty() + ); + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + drop(next.prepared); + cleaned(next.root.path(), &next.budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn empty_commands_and_native_failures_complete_without_a_catalog_generation() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let graph = assembled(&fixture, [97; 16], 0).await?; + let mut failed = Vec::new(); + packet(&mut failed, b"unpack invalid pack\n"); + packet(&mut failed, b"ng refs/heads/main unpacker error\n"); + failed.extend_from_slice(b"0000"); + for (at, response) in [ + GitHttpResponse { + status: 200, + headers: vec![], + body: b"0000".to_vec(), + }, + GitHttpResponse { + status: 200, + headers: vec![], + body: failed, + }, + GitHttpResponse { + status: 413, + headers: vec![("Content-Type".into(), "text/plain".into())], + body: b"native input too large\n".to_vec(), + }, + ] + .into_iter() + .enumerate() + { + let owner = if at == 0 { + None + } else { + Some(next_graph(&fixture, &graph, [at as u8 + 100; 16]).await?) + }; + let graph = owner.as_ref().unwrap_or(&graph); + let input = completion( + graph, + PushCompletionRequest { + plan: None, + response: response.clone(), + options: vec![], + certificate: None, + }, + ) + .await?; + let result = completed( + fixture + .client() + .command::(&fixture.target, identity()?, input) + .await? + .output, + )?; + assert_eq!(result.publication, None); + assert!(!result.rejected); + assert_eq!(stored(&fixture.handle, result.response_id).await?, response); + if let Some(graph) = owner { + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + } + } + fixture + .handle + .query(0, 1024, |connection| { + assert_eq!( + connection.query_row("SELECT generation FROM catalog_state", [], |row| row + .get::<_, u64>(0))?, + 0 + ); + Ok(Vec::new()) + }) + .await?; + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn moving_catalog_remains_retryable_and_owner_restore_preserves_exact_completion() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let graph = assembled(&fixture, [104; 16], 0).await?; + let pending = assembled(&fixture, [105; 16], 0).await?; + let input = completion(&graph, request(&graph, "refs/heads/main", true, 1, false)).await?; + let stale = completion( + &pending, + request(&pending, "refs/heads/pending", false, 0, false), + ) + .await?; + let mutation = identity()?; + let first = fixture + .client() + .command::(&fixture.target, mutation, input.clone()) + .await?; + let output = completed(first.output)?; + assert!(matches!( + pending.prepared.completed_push_response(&first).await, + Err(CatalogPushResponseError::Invalid) + )); + reject(&fixture, stale.clone(), PreparationDenial::Conflict).await?; + let before = state(&fixture.handle).await?; + let outcomes = counts(&fixture.handle).await?; + let response = stored(&fixture.handle, output.response_id).await?; + fixture.handle.drain().await?; + fixture.runtime.shutdown().await?; + let session = SessionId::from_bytes([106; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, session)?; + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await? + .ok_or("idle")?; + let provision = CellCatalog::new(fixture.layout.clone(), fixture.target.tenant()) + .lookup(fixture.target.cell_id()) + .await? + .ok_or("provision")?; + let handle = runtime + .acquire_idle_restored( + provision, + fixture.replica.clone(), + authority, + idle, + fixture.root.path().join("completion-b.sqlite"), + Owner { + session, + endpoint: "https://completion-b.invalid".into(), + }, + ) + .await?; + let client = CellClient::local(Arc::clone(&fixture.registry), handle.clone()); + let replay = client + .command::(&fixture.target, mutation, input.clone()) + .await?; + assert_eq!(replay.output, first.output); + assert_eq!(replay.receipt, first.receipt); + let logical = client + .command::(&fixture.target, identity()?, input) + .await?; + assert_eq!(logical.output, first.output); + assert_eq!(stored(&handle, output.response_id).await?, response); + assert_eq!( + replay_push_response( + &client, + &fixture.target, + fixture.begin(graph.prepared.token().operation), + None + ) + .await?, + Some(response) + ); + let failed = client + .command::(&fixture.target, identity()?, stale) + .await; + assert!( + matches!(failed,Err(InvocationError::Rejected(ref value)) if value.output==CatalogCompletionReply::Denied(PreparationDenial::Stale)) + ); + assert_eq!(state(&handle).await?, before); + assert_eq!(counts(&handle).await?, outcomes); + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + drop(pending.prepared); + cleaned(pending.root.path(), &pending.budget).await?; + runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn service_completion_checks_native_witness_scope_and_supports_checked_noop_refs() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let graph = assembled(&fixture, [107; 16], 0).await?; + let before = state(&fixture.handle).await?; + for at in 0..3 { + let mut native = request(&graph, "refs/heads/main", false, 0, true); + let certificate = native.certificate.as_mut().unwrap(); + match at { + 0 => certificate.request_digest[0] ^= 1, + 1 => { + certificate.target = crate::repository_target( + fixture.target.tenant(), + ApplicationId::from_bytes([99; 16]), + fixture.repository, + )? + } + _ => certificate.signer = "another-account".into(), + } + assert!(completion(&graph, native).await.is_err()); + assert_eq!(state(&fixture.handle).await?, before); + } + let first = Box::pin(graph.prepared.complete_push( + identity()?, + request(&graph, "refs/heads/main", false, 0, true), + graph.root.path(), + graph.budget.clone(), + limits(), + )) + .await?; + assert!(!completed(first.output)?.rejected); + let next = next_graph(&fixture, &graph, [108; 16]).await?; + let same = plan(vec![update( + "refs/heads/main", + Some((graph.initial, 1)), + Some(graph.initial), + )]); + let native = PushCompletionRequest { + response: response(&same, false, 0), + plan: Some(same), + options: vec![], + certificate: None, + }; + let second = Box::pin(next.prepared.complete_push( + identity()?, + native, + next.root.path(), + next.budget.clone(), + limits(), + )) + .await?; + assert_eq!( + completed(second.output)? + .publication + .map(|value| (value.generation, value.ref_generation)), + Some((2, 2)) + ); + assert!( + next.prepared + .completed_push_response(&second) + .await? + .body + .windows(18) + .any(|bytes| bytes == b"ok refs/heads/main") + ); + // A reused certificate accompanying an already failed native unpack records + // its refusal without attempting to turn the failed report into success. + let failure = next_graph(&fixture, &graph, [109; 16]).await?; + let mut native = request(&failure, "refs/heads/unused", false, 0, true); + native.plan = None; + let mut body = Vec::new(); + packet(&mut body, b"unpack invalid pack\n"); + packet(&mut body, b"ng refs/heads/unused unpacker error\n"); + body.extend_from_slice(b"0000"); + native.response.body = body.clone(); + let failed = Box::pin(failure.prepared.complete_push( + identity()?, + native, + failure.root.path(), + failure.budget.clone(), + limits(), + )) + .await?; + assert!(completed(failed.output)?.rejected); + assert_eq!( + failure + .prepared + .completed_push_response(&failed) + .await? + .body, + body + ); + for graph in [graph, next, failure] { + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + } + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn completed_preflight_prevents_repreparation_and_preserves_the_logical_identity() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let operation = [110; 16]; + let begin = fixture.begin(operation); + assert_eq!( + fixture + .client() + .query::(&fixture.target, None, begin.clone()) + .await? + .output, + None + ); + let graph = assembled(&fixture, operation, 0).await?; + assert_eq!( + fixture + .client() + .query::(&fixture.target, None, begin.clone()) + .await? + .output, + None + ); + let first = Box::pin(graph.prepared.complete_push( + identity()?, + request(&graph, "refs/heads/main", false, 0, false), + graph.root.path(), + graph.budget.clone(), + limits(), + )) + .await?; + let before = state(&fixture.handle).await?; + let found = fixture + .client() + .query::(&fixture.target, Some(first.receipt), begin.clone()) + .await?; + assert_eq!(found.output, Some(first.output)); + let response = graph.prepared.completed_push_response(&first).await?; + assert_eq!( + replay_push_response( + &fixture.client(), + &fixture.target, + begin.clone(), + Some(first.receipt) + ) + .await?, + Some(response) + ); + let result = fixture + .client() + .command::(&fixture.target, identity()?, begin.clone()) + .await; + assert!( + matches!(result,Err(InvocationError::Rejected(ref value)) if value.output==PreparationReply::Denied(PreparationDenial::Conflict)) + ); + assert_eq!(state(&fixture.handle).await?, before); + assert_eq!(fixture.counts().await?, (0, 1)); + let mut wrong = begin.clone(); + wrong.request_digest[0] ^= 1; + assert_eq!( + fixture + .client() + .query::(&fixture.target, None, wrong) + .await? + .output, + Some(CatalogCompletionReply::Denied(PreparationDenial::Conflict)) + ); + let mut outsider = begin; + outsider.actor = "outsider".into(); + assert_eq!( + fixture + .client() + .query::(&fixture.target, None, outsider) + .await? + .output, + Some(CatalogCompletionReply::Denied( + PreparationDenial::Unauthorized + )) + ); + assert_eq!(state(&fixture.handle).await?, before); + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +mod outcome; diff --git a/crates/canopy-server/src/packs/publication/tests/completion/outcome.rs b/crates/canopy-server/src/packs/publication/tests/completion/outcome.rs new file mode 100644 index 0000000..127afbf --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/completion/outcome.rs @@ -0,0 +1,361 @@ +use super::*; +use tokio::time::{Duration, timeout}; + +async fn opened(fixture: &Fixture, operation: [u8; 16]) -> Result> { + let started = fixture + .client() + .command::(&fixture.target, identity()?, fixture.begin(operation)) + .await?; + let token = lease(started.output)?.token; + Ok(Arc::new( + PreparationSession::open( + fixture.client(), + fixture.target.clone(), + check(token), + Some(started.receipt), + ) + .await?, + )) +} +fn native(status: u16, signed: bool, session: &PreparationSession) -> PushCompletionRequest { + let mut body = Vec::new(); + packet(&mut body, b"unpack ok\n"); + packet(&mut body, b"ng refs/heads/main hook declined\n"); + body.extend_from_slice(b"0000"); + PushCompletionRequest { + plan: None, + response: GitHttpResponse { + status, + headers: vec![("X-Native-Trace".into(), "exact".into())], + body, + }, + options: vec!["canopy.note=refused".into()], + certificate: signed.then(|| VerifiedPushCertificate { + target: session.target.clone(), + request_digest: session.check.token.request_digest, + signer: session.check.actor.clone(), + key: "native-verified".into(), + body: b"verified native signature bytes".to_vec(), + }), + } +} + +#[tokio::test] +async fn refusals_and_noop_outcomes_need_no_catalog_artifacts_or_generation_increment() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + for (i, status) in [200, 503, 204].into_iter().enumerate() { + let operation = [130 + i as u8; 16]; + let session = opened(&fixture, operation).await?; + let mut request = native(status, i == 0, &session); + if status == 204 { + request.response.body.clear(); + request.options.clear(); + } + let expected = request.response.clone(); + let before = state(&fixture.handle).await?; + let input = session.push_outcome(request).await?; + assert_eq!(state(&fixture.handle).await?, before); + let mut encoder = BoundedEncoder::new(4 << 20)?; + input.encode(&mut encoder)?; + let wire = encoder.finish(); + let mut decoder = BoundedDecoder::new(&wire, 4 << 20)?; + let input = CatalogPushCompletion::decode(&mut decoder)?; + decoder.finish()?; + let CompletionCatalogProof::OutcomeOnly(ref proof) = input.proof else { + return Err("wrong purpose".into()); + }; + let mut bounded = BoundedEncoder::new(CERTIFICATE_BYTES)?; + proof.encode(&mut bounded)?; + let mutation = identity()?; + let first = fixture + .client() + .command::(&fixture.target, mutation, input.clone()) + .await?; + let result = completed(first.output)?; + assert_eq!(result.publication, None); + assert!(!result.rejected); + assert_eq!(session.completed_push_response(&first).await?, expected); + assert_eq!(state(&fixture.handle).await?, before); + let exact = fixture + .client() + .command::(&fixture.target, mutation, input.clone()) + .await?; + assert_eq!(exact.output, first.output); + assert_eq!(exact.receipt, first.receipt); + let logical = fixture + .client() + .command::(&fixture.target, identity()?, input) + .await?; + assert_eq!(logical.output, first.output); + assert_eq!( + replay_push_response( + &fixture.client(), + &fixture.target, + fixture.begin(operation), + None + ) + .await?, + Some(expected) + ); + } + assert_eq!(fixture.counts().await?, (0, 3)); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn outcome_on_unavailable_old_catalog_survives_moving_frontier_without_loading_it() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + // The trusted fixture publishes a descriptor whose temporary artifact + // store is already gone. The session API has no loader/provider input. + fixture.install_empty_root(1).await?; + let session = opened(&fixture, [134; 16]).await?; + assert_eq!(session.lease.base.generation, 1); + let request = native(200, false, &session); + let expected = request.response.clone(); + let input = session.push_outcome(request).await?; + fixture.install_empty_root(2).await?; + let before = state(&fixture.handle).await?; + let committed = fixture + .client() + .command::(&fixture.target, identity()?, input) + .await?; + assert_eq!(completed(committed.output)?.publication, None); + assert_eq!(state(&fixture.handle).await?, before); + assert_eq!(session.completed_push_response(&committed).await?, expected); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn outcome_cannot_claim_refs_cross_purpose_or_edit_authenticated_native_bytes() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let session = opened(&fixture, [135; 16]).await?; + let input = session.push_outcome(native(200, true, &session)).await?; + for variant in 0..8 { + let mut bad = input.clone(); + match variant { + 0 => bad.response_id[0] ^= 1, + 1 => bad.response.status = 503, + 2 => bad + .response + .headers + .push(("X-Changed".into(), "yes".into())), + 3 => bad.response.body.push(b'x'), + 4 => bad.options.push("canopy.note=changed".into()), + 5 => bad.signed.as_mut().unwrap().key.push('x'), + 6 => bad.signed.as_mut().unwrap().body[0] ^= 1, + _ => bad.signed = None, + } + reject(&fixture, bad, PreparationDenial::Unauthorized).await?; + } + let mut request = native(200, false, &session); + request.plan = Some(plan(Vec::new())); + assert!(session.push_outcome(request).await.is_err()); + let mut request = native(200, false, &session); + let mut success = Vec::new(); + packet(&mut success, b"unpack ok\n"); + packet(&mut success, b"ok refs/heads/main\n"); + success.extend_from_slice(b"0000"); + request.response.body = success; + assert!(session.push_outcome(request).await.is_err()); + let CompletionCatalogProof::OutcomeOnly(outcome) = input.proof else { + return Err("purpose".into()); + }; + let graph = assembled(&fixture, [136; 16], 0).await?; + let real = graph.prepared.certificate().await?; + let before = state(&fixture.handle).await?; + // Reusing the same envelope does not make the two purposes interchangeable. + let fake = CatalogCertificate(outcome.0.clone()); + assert!( + fake.encode(&mut BoundedEncoder::new(CERTIFICATE_BYTES)?) + .is_err() + ); + let fake = OutcomeCertificate(real.0); + assert!( + fake.encode(&mut BoundedEncoder::new(CERTIFICATE_BYTES)?) + .is_err() + ); + assert_eq!(state(&fixture.handle).await?, before); + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn outcome_checks_current_write_authority_expiry_and_reconnect_read_authority() -> Result { + for expired in [false, true] { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let session = opened(&fixture, [137; 16]).await?; + let input = session.push_outcome(native(200, false, &session)).await?; + if expired { + edit(&fixture, "UPDATE catalog_operations SET expires_at_ms=0; UPDATE catalog_leases SET expires_at_ms=0;").await?; + reject(&fixture, input, PreparationDenial::Expired).await?; + } else { + edit( + &fixture, + "UPDATE repository_identity SET owner='successor';", + ) + .await?; + reject(&fixture, input.clone(), PreparationDenial::Unauthorized).await?; + assert!( + session + .push_outcome(native(200, false, &session)) + .await + .is_err() + ); + edit( + &fixture, + "INSERT INTO repository_members VALUES('owner','write');", + ) + .await?; + let committed = fixture + .client() + .command::(&fixture.target, identity()?, input) + .await?; + assert_eq!(completed(committed.output)?.publication, None); + edit( + &fixture, + "DELETE FROM repository_members WHERE account='owner';", + ) + .await?; + assert!(matches!( + replay_push_response( + &fixture.client(), + &fixture.target, + fixture.begin([137; 16]), + None + ) + .await, + Err(CatalogPushResponseError::Denied( + PreparationDenial::Unauthorized + )) + )); + } + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn outcome_dispatch_retains_session_and_exact_command_after_observer_cancel_or_lost_reply() +-> Result { + for fault in [0, 1, 2, 3] { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let session = opened(&fixture, [140 + fault; 16]).await?; + let weak = Arc::downgrade(&session); + let expected = native(200, false, &session).response; + let ready = session + .ready_outcome(identity()?, native(200, false, &session)) + .await?; + let coordinator = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let (release, entered) = coordinator.pause_for_test().await; + coordinator.fault_for_test(fault); + let ticket = coordinator.submit(ready).await?; + timeout(Duration::from_secs(5), entered).await??; + let lookup = fixture.begin(session.check.token.operation); + drop(session); + assert!(weak.upgrade().is_some()); + drop(ticket); + release.send(()).map_err(|_| "dispatch disappeared")?; + let pending = timeout(Duration::from_secs(10), coordinator.close_and_drain()).await?; + if fault == 0 { + assert!(pending.is_empty()); + } else { + assert_eq!(pending.len(), 1); + assert!(weak.upgrade().is_some()); + coordinator.recover(&pending[0]).await?; + assert!(matches!( + timeout(Duration::from_secs(10), pending[0].wait()).await?, + PublicationState::Finished(Ok(PublicationOutcome::Push(_))) + )); + } + assert!(weak.upgrade().is_none()); + assert_eq!(coordinator.reservations_for_test().await, (0, 0, 0)); + assert_eq!( + replay_push_response(&fixture.client(), &fixture.target, lookup, None).await?, + Some(expected) + ); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn restored_owner_rejects_unfinished_outcome_and_replays_original_completed_receipt() -> Result +{ + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let completed_session = opened(&fixture, [145; 16]).await?; + let pending_session = opened(&fixture, [146; 16]).await?; + let expected = native(200, true, &completed_session).response; + let input = completed_session + .push_outcome(native(200, true, &completed_session)) + .await?; + let pending = pending_session + .push_outcome(native(503, false, &pending_session)) + .await?; + let mutation = identity()?; + let first = fixture + .client() + .command::(&fixture.target, mutation, input.clone()) + .await?; + let before = state(&fixture.handle).await?; + let outcomes = counts(&fixture.handle).await?; + fixture.handle.drain().await?; + fixture.runtime.shutdown().await?; + let session = SessionId::from_bytes([147; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, session)?; + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await? + .ok_or("idle")?; + let provision = CellCatalog::new(fixture.layout.clone(), fixture.target.tenant()) + .lookup(fixture.target.cell_id()) + .await? + .ok_or("provision")?; + let handle = runtime + .acquire_idle_restored( + provision, + fixture.replica.clone(), + authority, + idle, + fixture.root.path().join("outcome-b.sqlite"), + Owner { + session, + endpoint: "https://outcome-b.invalid".into(), + }, + ) + .await?; + let client = CellClient::local(Arc::clone(&fixture.registry), handle.clone()); + let replay = client + .command::(&fixture.target, mutation, input.clone()) + .await?; + assert_eq!(replay.output, first.output); + assert_eq!(replay.receipt, first.receipt); + let logical = client + .command::(&fixture.target, identity()?, input) + .await?; + assert_eq!(logical.output, first.output); + assert_eq!( + replay_push_response(&client, &fixture.target, fixture.begin([145; 16]), None).await?, + Some(expected) + ); + let failed = client + .command::(&fixture.target, identity()?, pending) + .await; + assert!( + matches!(failed,Err(InvocationError::Rejected(ref value)) if value.output==CatalogCompletionReply::Denied(PreparationDenial::Stale)) + ); + assert_eq!(state(&handle).await?, before); + assert_eq!(counts(&handle).await?, outcomes); + runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/coordinator.rs b/crates/canopy-server/src/packs/publication/tests/coordinator.rs new file mode 100644 index 0000000..404ecfb --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/coordinator.rs @@ -0,0 +1,633 @@ +mod held; +mod preparation; +use super::*; +use super::{ + prepare::cleaned, + publishing::{assembled, edit, plan, update}, +}; +use crate::{ + git_http::GitHttpResponse, + packs::{ + catalog::{CatalogFileLimits, CatalogFiles, CatalogIndexes}, + metadata::tests::limits, + }, +}; +use canopy_object_storage::artifact::ArtifactStore; +use cellule_ltx::DiskBudget; +use tokio::time::{Duration, timeout}; + +pub(super) fn refused() -> GitHttpResponse { + GitHttpResponse { + status: 400, + headers: vec![("X-Native-Trace".into(), "exact-refusal".into())], + body: b"native input refused\n".to_vec(), + } +} +pub(super) fn request(response: GitHttpResponse) -> PushCompletionRequest { + PushCompletionRequest { + plan: None, + response, + options: vec!["canopy.note=queue".into()], + certificate: None, + } +} +fn accepted(tip: crate::ObjectId, name: &str) -> PushCompletionRequest { + let mut report = Vec::new(); + for body in [b"unpack ok\n".to_vec(), format!("ok {name}\n").into_bytes()] { + report.extend_from_slice(format!("{:04x}", body.len() + 4).as_bytes()); + report.extend_from_slice(&body); + } + report.extend_from_slice(b"0000"); + PushCompletionRequest { + plan: Some(plan(vec![update(name, None, Some(tip))])), + response: GitHttpResponse { + status: 200, + headers: vec![("X-Native-Trace".into(), "exact-acceptance".into())], + body: report, + }, + options: vec![], + certificate: None, + } +} +pub(super) fn finished( + state: PublicationState, +) -> Result> { + match state { + PublicationState::Finished(Ok(PublicationOutcome::Push(value))) => Ok(value), + other => Err(format!("unexpected {other:?}").into()), + } +} +async fn empty( + fixture: &Fixture, + operation: [u8; 16], + actor: &str, +) -> Result<(Arc, tempfile::TempDir, DiskBudget)> { + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + empty_in_store(fixture, operation, actor, store).await +} +pub(super) async fn empty_in_store( + fixture: &Fixture, + operation: [u8; 16], + actor: &str, + store: Arc, +) -> Result<(Arc, tempfile::TempDir, DiskBudget)> { + let mut input = fixture.begin(operation); + input.actor = actor.into(); + let started = fixture + .client() + .command::(&fixture.target, identity()?, input) + .await?; + let token = lease(started.output)?.token; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(256 << 20); + let indexes = Arc::new(CatalogIndexes::new(Arc::clone(&store), fixture.format)); + let files = Arc::new(CatalogFiles::new( + root.path(), + budget.clone(), + store, + fixture.format, + CatalogFileLimits::default(), + )?); + let base = Arc::new( + PreparationBaseResolver::open( + fixture.client(), + fixture.target.clone(), + LeaseCheck { + token, + actor: actor.into(), + }, + indexes, + files, + Some(started.receipt), + ) + .await?, + ); + let prepared = Arc::new( + CatalogPreparation::new(root.path(), budget.clone(), base, limits()) + .await? + .finish() + .await?, + ); + Ok((prepared, root, budget)) +} + +#[tokio::test] +async fn canceled_observer_does_not_cancel_admitted_native_publication_or_release_scratch() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let graph = assembled(&fixture, [60; 16], 0).await?; + let prepared = Arc::new(graph.prepared); + let weak = Arc::downgrade(&prepared); + let ready = Box::pin(prepared.ready_push( + identity()?, + accepted(graph.initial, "refs/heads/queued"), + graph.root.path(), + graph.budget.clone(), + limits(), + )) + .await?; + let coordinator = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let (release, entered) = coordinator.pause_for_test().await; + let ticket = coordinator.submit(ready).await?; + timeout(Duration::from_secs(5), entered).await??; + assert!(matches!(ticket.state(), PublicationState::Running)); + assert_eq!(coordinator.reservations_for_test().await, (1, 8 << 20, 1)); + drop(ticket); + drop(prepared); + assert!(weak.upgrade().is_some()); + assert_eq!( + graph.budget.used(), + crate::packs::metadata::growth::INITIAL_BYTES * 3 + ); + release.send(()).map_err(|_| "worker disappeared")?; + assert!( + timeout(Duration::from_secs(10), coordinator.close_and_drain()) + .await? + .is_empty() + ); + assert_eq!(coordinator.reservations_for_test().await, (0, 0, 0)); + assert!(weak.upgrade().is_none()); + assert_eq!( + replay_push_response( + &fixture.client(), + &fixture.target, + fixture.begin([60; 16]), + None + ) + .await?, + Some(accepted(graph.initial, "refs/heads/queued").response) + ); + cleaned(graph.root.path(), &graph.budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn admission_accounts_for_running_and_queued_work_without_losing_rejected_ready_input() +-> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + edit( + &fixture, + "INSERT INTO repository_members VALUES('writer','write')", + ) + .await?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits { + operations: 5, + per_actor: 2, + command_bytes: (24 << 20) + (8 << 10), + in_flight: 1, + maintenance_operations: 1, + maintenance_in_flight: 1, + foreground_burst: 3, + }, + )?; + let (release, entered) = coordinator.pause_for_test().await; + let mut attempts = Vec::new(); + for (n, actor) in [ + (61, "owner"), + (62, "owner"), + (63, "owner"), + (64, "writer"), + (65, "writer"), + ] { + let (prepared, root, budget) = empty(&fixture, [n; 16], actor).await?; + let ready = Box::pin(prepared.ready_push( + identity()?, + request(refused()), + root.path(), + budget.clone(), + limits(), + )) + .await?; + attempts.push((prepared, root, budget, Some(ReadyPublication::from(ready)))); + } + let first = coordinator + .submit(attempts[0].3.take().ok_or("ready")?) + .await?; + timeout(Duration::from_secs(5), entered).await??; + let duplicate = Box::pin(attempts[0].0.ready_push( + identity()?, + request(refused()), + attempts[0].1.path(), + attempts[0].2.clone(), + limits(), + )) + .await?; + assert_eq!( + coordinator + .submit(duplicate) + .await + .err() + .ok_or("duplicate admitted")? + .reason, + PublicationScheduleError::Duplicate + ); + let second = coordinator + .submit(attempts[1].3.take().ok_or("ready")?) + .await?; + let failure = coordinator + .submit(attempts[2].3.take().ok_or("ready")?) + .await + .err() + .ok_or("account quota")?; + assert_eq!(failure.reason, PublicationScheduleError::Capacity); + attempts[2].3 = Some(failure.ready); + let third = coordinator + .submit(attempts[3].3.take().ok_or("ready")?) + .await?; + let failure = coordinator + .submit(attempts[4].3.take().ok_or("ready")?) + .await + .err() + .ok_or("byte quota")?; + assert_eq!(failure.reason, PublicationScheduleError::Capacity); + attempts[4].3 = Some(failure.ready); + assert_eq!(coordinator.reservations_for_test().await, (3, 24 << 20, 2)); + let foreign = PublicationCoordinator::new( + crate::repository_target( + fixture.target.tenant(), + fixture.target.application(), + uuid::Uuid::new_v4().into_bytes(), + )?, + PublicationLimits::default(), + )?; + let failure = foreign + .submit(attempts[2].3.take().ok_or("ready")?) + .await + .err() + .ok_or("foreign admitted")?; + assert_eq!(failure.reason, PublicationScheduleError::Foreign); + attempts[2].3 = Some(failure.ready); + release.send(()).map_err(|_| "worker disappeared")?; + for ticket in [first, second, third] { + finished(timeout(Duration::from_secs(10), ticket.wait()).await?)?; + assert_eq!(ticket.response().await?, refused()); + } + assert_eq!(coordinator.reservations_for_test().await, (0, 0, 0)); + let retry = coordinator + .submit(attempts[2].3.take().ok_or("retained ready")?) + .await?; + finished(timeout(Duration::from_secs(10), retry.wait()).await?)?; + assert!(coordinator.close_and_drain().await.is_empty()); + let failure = coordinator + .submit(attempts[4].3.take().ok_or("ready")?) + .await + .err() + .ok_or("closed admitted")?; + assert_eq!(failure.reason, PublicationScheduleError::Closed); + drop(failure); + for (prepared, root, budget, ready) in attempts { + drop(ready); + drop(prepared); + cleaned(root.path(), &budget).await?; + } + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn unknown_absent_lost_ack_and_worker_panic_recover_exact_native_command() -> Result { + for fault in [1, 2, 3] { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let graph = assembled(&fixture, [70 + fault; 16], 0).await?; + let prepared = Arc::new(graph.prepared); + let weak = Arc::downgrade(&prepared); + let native = accepted(graph.initial, "refs/heads/recover"); + let expected = native.response.clone(); + let ready = Box::pin(prepared.ready_push( + identity()?, + native, + graph.root.path(), + graph.budget.clone(), + limits(), + )) + .await?; + let coordinator = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + coordinator.fault_for_test(fault); + let ticket = coordinator.submit(ready).await?; + drop(prepared); + let uncertain = timeout(Duration::from_secs(10), ticket.wait()).await?; + assert!(matches!(uncertain, PublicationState::Uncertain(_))); + let PublicationState::Uncertain(error) = &uncertain else { + return Err("no retained evidence".into()); + }; + let PublicationError::Push(InvocationError::Pending(evidence)) = error.as_ref() else { + return Err("no exact mutation evidence".into()); + }; + let original_sequence = match fixture.client().resolve(evidence).await? { + cellule_runtime::Resolution::Committed(value) => Some(value.commit_sequence()), + cellule_runtime::Resolution::Absent => None, + other => return Err(format!("unexpected resolution {other:?}").into()), + }; + assert_eq!(original_sequence.is_some(), fault != 1); + // Advance unrelated SQL after a committed result. Recovery must retain + // the original result's receipt, not substitute this later observation. + edit( + &fixture, + "UPDATE ref_generation SET visibility='public' WHERE singleton=1", + ) + .await?; + let stats = coordinator.stats().await; + assert_eq!( + ( + stats.admitted, + stats.queued, + stats.in_flight, + stats.uncertain + ), + (1, 0, 0, 1) + ); + assert_eq!(coordinator.reservations_for_test().await, (1, 8 << 20, 1)); + assert_eq!( + graph.budget.used(), + crate::packs::metadata::growth::INITIAL_BYTES * 3 + ); + assert!(weak.upgrade().is_some()); + drop(ticket); + let retained = coordinator + .pending([70 + fault; 16]) + .await + .ok_or("uncertain input lost")?; + let drained = coordinator.close_and_drain().await; + assert_eq!(drained.len(), 1); + let before = replay_push_response( + &fixture.client(), + &fixture.target, + fixture.begin([70 + fault; 16]), + None, + ) + .await?; + assert_eq!(before.is_some(), fault != 1); + coordinator.recover(&retained).await?; + let completed = finished(timeout(Duration::from_secs(10), retained.wait()).await?)?; + if let Some(sequence) = original_sequence { + assert_eq!(completed.receipt.commit_sequence, sequence); + } + assert_eq!(retained.response().await?, expected); + assert!(matches!( + completed.output, + CatalogCompletionReply::Completed(CompletedCatalogPush { + rejected: false, + publication: Some(PublishedRefs { + generation: 1, + ref_generation: 1, + .. + }), + .. + }) + )); + assert_eq!(coordinator.reservations_for_test().await, (0, 0, 0)); + assert!(weak.upgrade().is_none()); + assert_eq!( + coordinator.recover(&retained).await, + Err(PublicationScheduleError::NotUncertain) + ); + // Resolved tickets carry only small receipt/read context, not scratch. + assert!(coordinator.close_and_drain().await.is_empty()); + cleaned(graph.root.path(), &graph.budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn stale_ready_command_has_durable_conflict_then_reconciliation_can_reenter_queue() -> Result +{ + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new( + Arc::clone(&provider), + fixture.repository, + )); + let first = super::reconcile::graph( + &fixture, + Arc::clone(&provider), + Arc::clone(&store), + [80; 16], + 4, + ) + .await?; + let second = super::reconcile::graph(&fixture, provider, store, [81; 16], 5).await?; + let a = Arc::new(first.prepared); + let b = Arc::new(second.prepared); + let a_ready = Box::pin(a.ready_push( + identity()?, + accepted(first.initial, "refs/heads/a"), + first.root.path(), + first.budget.clone(), + limits(), + )) + .await?; + let b_ready = Box::pin(b.ready_push( + identity()?, + accepted(second.initial, "refs/heads/b"), + second.root.path(), + second.budget.clone(), + limits(), + )) + .await?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits { + in_flight: 1, + maintenance_in_flight: 1, + ..PublicationLimits::default() + }, + )?; + let (release, entered) = coordinator.pause_for_test().await; + let a_ticket = coordinator.submit(a_ready).await?; + entered.await?; + let old_ticket = coordinator.submit(b_ready).await?; + release.send(()).map_err(|_| "worker disappeared")?; + finished(a_ticket.wait().await)?; + let old = old_ticket.wait().await; + assert!( + matches!(old, PublicationState::Finished(Err(ref error)) if matches!(error.as_ref(), PublicationError::Push(InvocationError::Rejected(value)) if value.output==CatalogCompletionReply::Denied(PreparationDenial::Conflict))) + ); + assert_eq!( + replay_push_response( + &fixture.client(), + &fixture.target, + fixture.begin([81; 16]), + None + ) + .await?, + None + ); + let reconciled = Arc::new(b.reconcile().await?); + assert_eq!(reconciled.base().generation, 1); + assert_eq!(reconciled.token(), b.token()); + let ready = Box::pin(reconciled.ready_push( + identity()?, + accepted(second.initial, "refs/heads/b"), + second.root.path(), + second.budget.clone(), + limits(), + )) + .await?; + let published = coordinator.submit(ready).await?; + let value = finished(published.wait().await)?; + assert!(matches!( + value.output, + CatalogCompletionReply::Completed(CompletedCatalogPush { + publication: Some(PublishedRefs { generation: 2, .. }), + .. + }) + )); + assert_eq!( + published.response().await?, + accepted(second.initial, "refs/heads/b").response + ); + assert!(coordinator.close_and_drain().await.is_empty()); + drop(a); + drop(b); + drop(reconciled); + cleaned(first.root.path(), &first.budget).await?; + cleaned(second.root.path(), &second.budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn queued_publication_still_evaluates_current_authorization() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let graph = assembled(&fixture, [90; 16], 0).await?; + let prepared = Arc::new(graph.prepared); + let mut native = accepted(graph.initial, "refs/heads/revoked"); + // report-status was declined; revocation must become an explicit HTTP + // failure rather than an empty success response. + native.response.body.clear(); + let ready = Box::pin(prepared.ready_push( + identity()?, + native, + graph.root.path(), + graph.budget.clone(), + limits(), + )) + .await?; + let coordinator = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let (release, entered) = coordinator.pause_for_test().await; + let ticket = coordinator.submit(ready).await?; + entered.await?; + edit(&fixture, "UPDATE repository_identity SET owner='other'; INSERT INTO repository_members VALUES('owner','read')").await?; + release.send(()).map_err(|_| "worker disappeared")?; + let completed = finished(ticket.wait().await)?; + assert!(matches!( + completed.output, + CatalogCompletionReply::Completed(CompletedCatalogPush { + rejected: true, + publication: None, + .. + }) + )); + assert_eq!(ticket.response().await?.status, 409); + assert!(coordinator.close_and_drain().await.is_empty()); + drop(prepared); + cleaned(graph.root.path(), &graph.budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn bounded_dispatch_allows_another_command_to_progress_before_first_outcome() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let first = empty(&fixture, [91; 16], "owner").await?; + let second = empty(&fixture, [92; 16], "owner").await?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits { + in_flight: 2, + maintenance_in_flight: 1, + ..PublicationLimits::default() + }, + )?; + let (release, entered) = coordinator.pause_for_test().await; + let ready = Box::pin(first.0.ready_push( + identity()?, + request(refused()), + first.1.path(), + first.2.clone(), + limits(), + )) + .await?; + let a = coordinator.submit(ready).await?; + entered.await?; + let ready = Box::pin(second.0.ready_push( + identity()?, + request(refused()), + second.1.path(), + second.2.clone(), + limits(), + )) + .await?; + let b = coordinator.submit(ready).await?; + finished(timeout(Duration::from_secs(10), b.wait()).await?)?; + assert!(matches!(a.state(), PublicationState::Running)); + assert_eq!(coordinator.reservations_for_test().await, (1, 8 << 20, 1)); + release.send(()).map_err(|_| "worker disappeared")?; + finished(a.wait().await)?; + assert!(coordinator.close_and_drain().await.is_empty()); + for (prepared, root, budget) in [first, second] { + drop(prepared); + cleaned(root.path(), &budget).await?; + } + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn oversized_inline_completion_fails_before_dispatch_while_inventory_stays_reusable() -> Result +{ + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let (prepared, root, budget) = empty(&fixture, [93; 16], "owner").await?; + let mut native = request(refused()); + native.response.body.resize(5 << 20, b'x'); + let error = + Box::pin(prepared.ready_push(identity()?, native, root.path(), budget.clone(), limits())) + .await + .err() + .ok_or("oversize prepared")?; + assert!(matches!(error, PushCompletionProofError::Codec(_))); + assert_eq!( + replay_push_response( + &fixture.client(), + &fixture.target, + fixture.begin([93; 16]), + None + ) + .await?, + None + ); + prepared.ensure_live()?; + assert_eq!( + budget.used(), + crate::packs::metadata::growth::INITIAL_BYTES * 3 + ); + let ready = Box::pin(prepared.ready_push( + identity()?, + request(refused()), + root.path(), + budget.clone(), + limits(), + )) + .await?; + let coordinator = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let ticket = coordinator.submit(ready).await?; + finished(ticket.wait().await)?; + assert!(coordinator.close_and_drain().await.is_empty()); + drop(prepared); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/coordinator/held.rs b/crates/canopy-server/src/packs/publication/tests/coordinator/held.rs new file mode 100644 index 0000000..8b116b7 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/coordinator/held.rs @@ -0,0 +1,411 @@ +use super::*; + +async fn outcomes(handle: &CellHandle) -> Result { + Ok(super::super::completion::counts(handle).await?[0]) +} + +#[tokio::test] +async fn held_native_proof_survives_canceled_observation_and_closed_activation() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let graph = assembled(&fixture, [60; 16], 0).await?; + let prepared = Arc::new(graph.prepared); + let weak = Arc::downgrade(&prepared); + let response = accepted(graph.initial, "refs/heads/held"); + let ready = Box::pin(prepared.ready_push( + identity()?, + accepted(graph.initial, "refs/heads/held"), + graph.root.path(), + graph.budget.clone(), + limits(), + )) + .await?; + let coordinator = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let ticket = coordinator.try_reserve(ready)?; + drop(prepared); + let observer = ticket.clone(); + let waiter = tokio::spawn(async move { observer.wait().await }); + tokio::task::yield_now().await; + waiter.abort(); + assert!(waiter.await.unwrap_err().is_cancelled()); + drop(ticket); + let ticket = coordinator + .pending([60; 16]) + .await + .ok_or("held admission lost")?; + assert!(matches!(ticket.state(), PublicationState::Held)); + let stats = coordinator.stats().await; + assert_eq!( + ( + stats.held, + stats.queued, + stats.in_flight, + stats.command_bytes + ), + (1, 0, 0, 8 << 20) + ); + assert!(weak.upgrade().is_some()); + assert_eq!(outcomes(&fixture.handle).await?, 0); + assert_eq!( + graph.budget.used(), + crate::packs::metadata::growth::INITIAL_BYTES * 3 + ); + assert!(ticket.response().await.is_err()); + assert_eq!( + ticket.recover().await, + Err(PublicationScheduleError::NotUncertain) + ); + let closed = coordinator.close_and_drain().await; + assert_eq!(closed.len(), 1); + assert!(matches!(closed[0].state(), PublicationState::Held)); + assert_eq!(coordinator.reservations_for_test().await, (1, 8 << 20, 1)); + let (release, entered) = coordinator.pause_for_test().await; + let (a, b) = tokio::join!(ticket.activate(), ticket.activate()); + a?; + b?; + timeout(Duration::from_secs(5), entered).await??; + assert_eq!( + ticket.discard_held().await, + Err(PublicationScheduleError::NotHeld) + ); + release.send(()).map_err(|_| "dispatch disappeared")?; + let committed = finished(timeout(Duration::from_secs(10), ticket.wait()).await?)?; + assert!(matches!( + committed.output, + CatalogCompletionReply::Completed(_) + )); + ticket.activate().await?; + assert_eq!(outcomes(&fixture.handle).await?, 1); + assert_eq!(ticket.response().await?, response.response); + assert!(weak.upgrade().is_none()); + assert_eq!(coordinator.reservations_for_test().await, (0, 0, 0)); + cleaned(graph.root.path(), &graph.budget).await?; + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn held_discard_drops_native_proof_before_credit_and_never_executes() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let graph = assembled(&fixture, [60; 16], 0).await?; + let prepared = Arc::new(graph.prepared); + let weak = Arc::downgrade(&prepared); + let ready = Box::pin(prepared.ready_push( + identity()?, + accepted(graph.initial, "refs/heads/discarded"), + graph.root.path(), + graph.budget.clone(), + limits(), + )) + .await?; + let coordinator = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let ticket = coordinator.try_reserve(ready)?; + drop(prepared); + assert_eq!(coordinator.close_and_drain().await.len(), 1); + ticket.discard_held().await?; + assert!(matches!(ticket.wait().await, PublicationState::Discarded)); + assert!(weak.upgrade().is_none()); + cleaned(graph.root.path(), &graph.budget).await?; + assert_eq!(coordinator.reservations_for_test().await, (0, 0, 0)); + assert!(coordinator.pending([60; 16]).await.is_none()); + assert_eq!( + ticket.activate().await, + Err(PublicationScheduleError::NotHeld) + ); + assert_eq!( + ticket.discard_held().await, + Err(PublicationScheduleError::NotHeld) + ); + assert_eq!( + ticket.recover().await, + Err(PublicationScheduleError::NotUncertain) + ); + assert!(ticket.response().await.is_err()); + assert_eq!(outcomes(&fixture.handle).await?, 0); + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn held_admission_uses_existing_account_bytes_and_returns_refused_ready() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + edit( + &fixture, + "INSERT INTO repository_members VALUES('writer','write')", + ) + .await?; + let coordinator = PublicationCoordinator::new( + fixture.target.clone(), + PublicationLimits { + operations: 5, + per_actor: 2, + command_bytes: (24 << 20) + (8 << 10), + in_flight: 1, + maintenance_operations: 1, + maintenance_in_flight: 1, + foreground_burst: 3, + }, + )?; + let mut attempts = Vec::new(); + for (n, actor) in [ + (61, "owner"), + (62, "owner"), + (63, "owner"), + (64, "writer"), + (65, "writer"), + ] { + let (prepared, root, budget) = empty(&fixture, [n; 16], actor).await?; + let ready = Box::pin(prepared.ready_push( + identity()?, + request(refused()), + root.path(), + budget.clone(), + limits(), + )) + .await?; + attempts.push((prepared, root, budget, Some(ReadyPublication::from(ready)))); + } + let foreign = PublicationCoordinator::new( + crate::repository_target( + TenantId::from_bytes([99; 16]), + fixture.target.application(), + fixture.repository, + )?, + PublicationLimits::default(), + )?; + let failure = foreign + .try_reserve(attempts[0].3.take().unwrap()) + .err() + .ok_or("refused admission unexpectedly accepted")?; + assert_eq!(failure.reason, PublicationScheduleError::Foreign); + let first = coordinator.try_reserve(failure.ready)?; + let failure = coordinator + .with_admission_for_test(|| coordinator.try_reserve(attempts[1].3.take().unwrap())) + .await + .err() + .ok_or("contended synchronous admission unexpectedly accepted")?; + assert_eq!(failure.reason, PublicationScheduleError::Capacity); + attempts[1].3 = Some(failure.ready); + // The same logical operation cannot have a second held/executing slot. + let duplicate = Box::pin(attempts[0].0.ready_push( + identity()?, + request(refused()), + attempts[0].1.path(), + attempts[0].2.clone(), + limits(), + )) + .await?; + let failure = coordinator + .try_reserve(duplicate) + .err() + .ok_or("refused admission unexpectedly accepted")?; + assert_eq!(failure.reason, PublicationScheduleError::Duplicate); + drop(failure); + let second = coordinator.try_reserve(attempts[1].3.take().unwrap())?; + let failure = coordinator + .try_reserve(attempts[2].3.take().unwrap()) + .err() + .ok_or("refused admission unexpectedly accepted")?; + assert_eq!(failure.reason, PublicationScheduleError::Capacity); + attempts[2].3 = Some(failure.ready); + let third = coordinator.try_reserve(attempts[3].3.take().unwrap())?; + let failure = coordinator + .try_reserve(attempts[4].3.take().unwrap()) + .err() + .ok_or("refused admission unexpectedly accepted")?; + assert_eq!(failure.reason, PublicationScheduleError::Capacity); + attempts[4].3 = Some(failure.ready); + assert_eq!(coordinator.reservations_for_test().await, (3, 24 << 20, 2)); + assert_eq!(coordinator.stats().await.held, 3); + first.discard_held().await?; + let retry = coordinator.try_reserve(attempts[2].3.take().unwrap())?; + assert_eq!(coordinator.reservations_for_test().await, (3, 24 << 20, 2)); + let (release, entered) = coordinator.pause_for_test().await; + second.activate().await?; + timeout(Duration::from_secs(5), entered).await??; + third.activate().await?; + retry.activate().await?; + release.send(()).map_err(|_| "dispatch disappeared")?; + for ticket in [second, third, retry] { + finished(timeout(Duration::from_secs(10), ticket.wait()).await?)?; + } + assert_eq!(outcomes(&fixture.handle).await?, 3); + assert_eq!(coordinator.reservations_for_test().await, (0, 0, 0)); + assert!(coordinator.close_and_drain().await.is_empty()); + let failure = coordinator + .try_reserve(attempts[4].3.take().unwrap()) + .err() + .ok_or("refused admission unexpectedly accepted")?; + assert_eq!(failure.reason, PublicationScheduleError::Closed); + let successor = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let ticket = successor.try_reserve(failure.ready)?; + ticket.activate().await?; + finished(timeout(Duration::from_secs(10), ticket.wait()).await?)?; + assert_eq!(outcomes(&fixture.handle).await?, 4); + assert!(successor.close_and_drain().await.is_empty()); + drop(attempts); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn activated_held_command_recovers_exact_receipt_or_checks_absent_authority_after_close() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in [1, 2, 3] { + let fixture = Fixture::new(format).await?; + let (prepared, root, budget) = empty(&fixture, [66; 16], "owner").await?; + let ready = Box::pin(prepared.ready_push( + identity()?, + request(refused()), + root.path(), + budget.clone(), + limits(), + )) + .await?; + let coordinator = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let ticket = coordinator.try_reserve(ready)?; + assert!(matches!(ticket.state(), PublicationState::Held)); + coordinator.fault_for_test(fault); + ticket.activate().await?; + let PublicationState::Uncertain(error) = + timeout(Duration::from_secs(10), ticket.wait()).await? + else { + return Err("exact uncertainty lost".into()); + }; + let PublicationError::Push(InvocationError::Pending(evidence)) = error.as_ref() else { + return Err("wrong exact evidence".into()); + }; + let original = match fixture.client().resolve(evidence).await? { + cellule_runtime::Resolution::Committed(value) => Some(value.commit_sequence()), + cellule_runtime::Resolution::Absent => None, + other => return Err(format!("unexpected resolution {other:?}").into()), + }; + assert_eq!(original.is_some(), fault != 1); + let stats = coordinator.stats().await; + assert_eq!( + (stats.held, stats.uncertain, stats.command_bytes), + (0, 1, 8 << 20) + ); + // Activation never retries an ambiguous command, and discard cannot + // erase it. Only explicit recovery joins the exact fair queue. + ticket.activate().await?; + assert!(matches!(ticket.state(), PublicationState::Uncertain(_))); + assert_eq!( + ticket.discard_held().await, + Err(PublicationScheduleError::NotHeld) + ); + edit( + &fixture, + "UPDATE repository_identity SET owner='replacement'; UPDATE ref_generation SET visibility='private'", + ) + .await?; + drop(ticket); + let closed = coordinator.close_and_drain().await; + assert_eq!(closed.len(), 1); + let retained = coordinator + .pending([66; 16]) + .await + .ok_or("exact held command lost")?; + retained.recover().await?; + match timeout(Duration::from_secs(10), retained.wait()).await? { + PublicationState::Finished(Ok(PublicationOutcome::Push(value))) if fault != 1 => { + assert_eq!(Some(value.receipt.commit_sequence), original); + assert!(matches!(value.output, CatalogCompletionReply::Completed(_))); + assert_eq!(outcomes(&fixture.handle).await?, 1); + assert_eq!(retained.response().await?, refused()); + assert!(matches!( + replay_push_response( + &fixture.client(), + &fixture.target, + fixture.begin([66; 16]), + None + ) + .await, + Err(CatalogPushResponseError::Denied( + PreparationDenial::Unauthorized + )) + )); + edit(&fixture, "UPDATE repository_identity SET owner='owner'").await?; + assert_eq!(retained.response().await?, refused()); + } + PublicationState::Finished(Err(error)) if fault == 1 => { + assert!( + matches!(error.as_ref(), PublicationError::Push(InvocationError::Rejected(value)) + if value.output == CatalogCompletionReply::Denied(PreparationDenial::Unauthorized)) + ); + assert_eq!(outcomes(&fixture.handle).await?, 0); + assert!(retained.response().await.is_err()); + } + other => return Err(format!("unexpected recovery {other:?}").into()), + } + assert_eq!(coordinator.reservations_for_test().await, (0, 0, 0)); + assert!(coordinator.close_and_drain().await.is_empty()); + drop(prepared); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn held_activation_and_discard_race_selects_one_exact_disposition() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let coordinator = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let mut published = 0; + for n in 100..108 { + let (prepared, root, budget) = empty(&fixture, [n; 16], "owner").await?; + let ready = Box::pin(prepared.ready_push( + identity()?, + request(refused()), + root.path(), + budget.clone(), + limits(), + )) + .await?; + let ticket = coordinator.try_reserve(ready)?; + let activator = ticket.clone(); + let discard = ticket.clone(); + let barrier = Arc::new(tokio::sync::Barrier::new(3)); + let a_barrier = Arc::clone(&barrier); + let d_barrier = Arc::clone(&barrier); + let a = tokio::spawn(async move { + a_barrier.wait().await; + activator.activate().await + }); + let d = tokio::spawn(async move { + d_barrier.wait().await; + discard.discard_held().await + }); + barrier.wait().await; + let (a, d) = (a.await?, d.await?); + match (a, d) { + (Ok(()), Err(PublicationScheduleError::NotHeld)) => { + finished(timeout(Duration::from_secs(10), ticket.wait()).await?)?; + published += 1; + } + (Err(PublicationScheduleError::NotHeld), Ok(())) => { + assert!(matches!(ticket.wait().await, PublicationState::Discarded)) + } + other => return Err(format!("contradictory disposition {other:?}").into()), + } + assert_eq!(outcomes(&fixture.handle).await?, published); + assert_eq!(coordinator.reservations_for_test().await, (0, 0, 0)); + drop(prepared); + cleaned(root.path(), &budget).await?; + } + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/coordinator/preparation.rs b/crates/canopy-server/src/packs/publication/tests/coordinator/preparation.rs new file mode 100644 index 0000000..8f402a0 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/coordinator/preparation.rs @@ -0,0 +1,423 @@ +use super::*; + +async fn session(f: &Fixture, operation: [u8; 16]) -> Result> { + let started = f + .client() + .command::(&f.target, identity()?, f.begin(operation)) + .await?; + let lease = lease(started.output)?; + Ok(Arc::new( + PreparationSession::open( + f.client(), + f.target.clone(), + check(lease.token), + Some(started.receipt), + ) + .await?, + )) +} +fn request_for(s: &PreparationSession) -> LeaseRequest { + LeaseRequest { + check: s.check.clone(), + lease_ms: DEFAULT_LEASE_MS, + } +} +async fn ready( + f: &Fixture, + s: &Arc, + kind: PreparationCommandKind, + mutation: MutationIdentity, +) -> Result { + Ok(match kind { + PreparationCommandKind::Claim => { + ReadyPreparation::claim(f.client(), f.target.clone(), request_for(s), mutation).await? + } + PreparationCommandKind::Renew => s.ready_renew(mutation, DEFAULT_LEASE_MS).await?, + }) +} +fn changed(state: PublicationState) -> Result { + match state { + PublicationState::Finished(Ok(PublicationOutcome::Preparation(value))) => Ok(value), + other => Err(format!("preparation outcome: {other:?}").into()), + } +} +async fn replay( + f: &Fixture, + s: &PreparationSession, + kind: PreparationCommandKind, + mutation: MutationIdentity, +) -> Result> { + Ok(match kind { + PreparationCommandKind::Claim => { + f.client() + .command::(&f.target, mutation, request_for(s)) + .await? + } + PreparationCommandKind::Renew => { + f.client() + .command::(&f.target, mutation, request_for(s)) + .await? + } + }) +} +#[tokio::test] +async fn bound_lease_commands_keep_exact_identity_and_original_floor_through_closed_uncertain_recovery() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for kind in [PreparationCommandKind::Claim, PreparationCommandKind::Renew] { + for fault in [1, 2, 3] { + let f = Fixture::new(format).await?; + let s = session(&f, [196; 16]).await?; + f.install_empty_root(1).await?; + let coordinator = + PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let mutation = identity()?; + coordinator.fault_for_test(fault); + let ticket = coordinator + .submit(ready(&f, &s, kind, mutation).await?) + .await?; + let PublicationState::Uncertain(error) = + timeout(Duration::from_secs(10), ticket.wait()).await? + else { + return Err("lease uncertainty".into()); + }; + let PublicationError::Preparation(cellule_runtime::InvocationError::Pending( + evidence, + )) = &*error + else { + return Err("exact lease evidence".into()); + }; + let evidence = (**evidence).clone(); + let original = match f.client().resolve(&evidence).await? { + cellule_runtime::Resolution::Absent => None, + cellule_runtime::Resolution::Committed(value) => Some(value.commit_sequence()), + other => return Err(format!("unexpected {other:?}").into()), + }; + assert_eq!(original.is_some(), fault != 1); + assert_eq!(coordinator.reservations_for_test().await, (1, 8192, 1)); + drop(ticket); + let retained = coordinator + .pending([196; 16]) + .await + .ok_or("lease retained")?; + assert_eq!(coordinator.close_and_drain().await.len(), 1); + coordinator.recover(&retained).await?; + let outcome = changed(timeout(Duration::from_secs(10), retained.wait()).await?)?; + assert_eq!(outcome.kind, kind); + assert_eq!(outcome.committed, replay(&f, &s, kind, mutation).await?); + if let Some(sequence) = original { + assert_eq!(outcome.committed.receipt.commit_sequence, sequence); + } + let fresh = outcome.session.map_err(|e| e.to_string())?; + let (lease, _) = fresh.live_lease()?; + assert_eq!(lease.format, format); + match kind { + PreparationCommandKind::Claim => { + assert_ne!(lease.token, s.lease.token); + assert_eq!(lease.base.generation, 1); + } + PreparationCommandKind::Renew => { + assert_eq!(lease.token, s.lease.token); + assert_eq!(lease.base.generation, 0); + assert!(Arc::ptr_eq(&fresh, &s)); + } + } + let old = s.lease.token; + let old_floor = f.handle.query(0, 32, move |conn| { + Ok(conn.query_row("SELECT generation FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2", rusqlite::params![old.owner.incarnation.as_bytes().as_slice(), old.attempt as i64], |row| row.get::<_, i64>(0))?.to_be_bytes().to_vec()) + }).await?; + assert_eq!(old_floor.as_slice(), 0i64.to_be_bytes()); + assert!(retained.response().await.is_err()); + assert_eq!(coordinator.reservations_for_test().await, (0, 0, 0)); + assert!(coordinator.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + } + } + Ok(()) +} +#[tokio::test] +async fn bound_lease_ready_admission_preserves_command_and_canceled_observer_session_ownership() +-> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let s = session(&f, [197; 16]).await?; + let weak = Arc::downgrade(&s); + let mutation = identity()?; + let prepared = s.ready_renew(mutation, DEFAULT_LEASE_MS).await?; + let foreign = PublicationCoordinator::new( + crate::repository_target( + f.target.tenant(), + f.target.application(), + uuid::Uuid::new_v4().into_bytes(), + )?, + PublicationLimits::default(), + )?; + let rejected = foreign + .submit(prepared) + .await + .err() + .ok_or("foreign admitted")?; + assert_eq!(rejected.reason, PublicationScheduleError::Foreign); + let coordinator = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let (release, entered) = coordinator.pause_for_test().await; + let ticket = coordinator.submit(rejected.ready).await?; + timeout(Duration::from_secs(5), entered).await??; + let duplicate = coordinator + .submit(s.ready_renew(identity()?, DEFAULT_LEASE_MS).await?) + .await + .err() + .ok_or("duplicate admitted")?; + assert_eq!(duplicate.reason, PublicationScheduleError::Duplicate); + drop(duplicate); + drop(s); + drop(ticket); + assert!(weak.upgrade().is_some()); + let pending = coordinator + .pending([197; 16]) + .await + .ok_or("running lease retained")?; + release.send(()).map_err(|_| "worker gone")?; + let outcome = changed(timeout(Duration::from_secs(10), pending.wait()).await?)?; + let fresh = outcome.session.map_err(|e| e.to_string())?; + assert_eq!( + outcome.committed, + replay(&f, &fresh, PreparationCommandKind::Renew, mutation).await? + ); + assert!(coordinator.close_and_drain().await.is_empty()); + let prepared = fresh.ready_renew(identity()?, DEFAULT_LEASE_MS).await?; + let refused = coordinator + .submit(prepared) + .await + .err() + .ok_or("closed admitted")?; + assert_eq!(refused.reason, PublicationScheduleError::Closed); + let other = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let retried = other.submit(refused.ready).await?; + changed(timeout(Duration::from_secs(10), retried.wait()).await?)?; + assert!(other.close_and_drain().await.is_empty()); + assert!(foreign.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + Ok(()) +} +#[tokio::test] +async fn bound_lease_committed_recovery_keeps_receipt_when_fresh_custody_is_revoked_expired_or_superseded() +-> Result { + for kind in [PreparationCommandKind::Claim, PreparationCommandKind::Renew] { + for mode in [0, 1, 2] { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let s = session(&f, [198; 16]).await?; + let coordinator = + PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let mutation = identity()?; + coordinator.fault_for_test(2); + let ticket = coordinator + .submit(ready(&f, &s, kind, mutation).await?) + .await?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait()).await?, + PublicationState::Uncertain(_) + )); + let original = replay(&f, &s, kind, mutation).await?; + let current = lease(original.output.clone())?; + match mode { + 0 => edit(&f, "UPDATE repository_identity SET owner='other' WHERE singleton=1").await?, + 1 => edit(&f, "UPDATE catalog_operations SET expires_at_ms=0; UPDATE catalog_leases SET expires_at_ms=0").await?, + _ => { f.client().command::(&f.target, identity()?, LeaseRequest { check: check(current.token), lease_ms: DEFAULT_LEASE_MS }).await?; } + } + coordinator.recover(&ticket).await?; + let outcome = changed(timeout(Duration::from_secs(10), ticket.wait()).await?)?; + assert_eq!(outcome.committed, original); + assert!(outcome.session.is_err()); + if kind == PreparationCommandKind::Renew { + assert!(s.live_lease().is_err()); + } + assert_eq!(outcome.committed, replay(&f, &s, kind, mutation).await?); + assert!(coordinator.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + } + Ok(()) +} +#[tokio::test] +async fn bound_lease_absent_recovery_rechecks_authority_and_claim_can_recover_expired_source() +-> Result { + for kind in [PreparationCommandKind::Claim, PreparationCommandKind::Renew] { + for mode in [0, 1, 2] { + let f = Fixture::new(ObjectFormat::Sha1).await?; + let s = session(&f, [199; 16]).await?; + let coordinator = + PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + coordinator.fault_for_test(1); + let ticket = coordinator + .submit(ready(&f, &s, kind, identity()?).await?) + .await?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait()).await?, + PublicationState::Uncertain(_) + )); + let denial = match mode { + 0 => { + edit( + &f, + "UPDATE repository_identity SET owner='other' WHERE singleton=1", + ) + .await?; + PreparationDenial::Unauthorized + } + 1 => { + edit(&f, "UPDATE catalog_operations SET expires_at_ms=0; UPDATE catalog_leases SET expires_at_ms=0").await?; + PreparationDenial::Expired + } + _ => { + f.client() + .command::(&f.target, identity()?, request_for(&s)) + .await?; + PreparationDenial::Stale + } + }; + coordinator.recover(&ticket).await?; + let state = timeout(Duration::from_secs(10), ticket.wait()).await?; + if kind == PreparationCommandKind::Claim && mode == 1 { + let result = changed(state)?; + let fresh = result.session.map_err(|e| e.to_string())?; + assert_ne!(fresh.live_lease()?.0.token, s.lease.token); + } else { + let PublicationState::Finished(Err(error)) = state else { + return Err("absent lease accepted".into()); + }; + assert!( + matches!(&*error, PublicationError::Preparation(cellule_runtime::InvocationError::Rejected(value)) if value.output == PreparationReply::Denied(denial)) + ); + if kind == PreparationCommandKind::Renew { + assert!(s.live_lease().is_err()); + } + } + assert!(coordinator.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + } + Ok(()) +} +#[tokio::test] +async fn bound_lease_ready_rejects_invalid_context_size_duration_and_never_revives_fenced_session() +-> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let s = session(&f, [200; 16]).await?; + for duration in [0, MAX_LEASE_MS + 1] { + assert!(s.ready_renew(identity()?, duration).await.is_err()); + } + let mut wrong = request_for(&s); + wrong.check.token.repository = uuid::Uuid::new_v4().into_bytes(); + assert!( + ReadyPreparation::claim(f.client(), f.target.clone(), wrong, identity()?) + .await + .is_err() + ); + let mut huge = request_for(&s); + huge.check.actor = "x".repeat(8192); + assert!( + ReadyPreparation::claim(f.client(), f.target.clone(), huge, identity()?) + .await + .is_err() + ); + let coordinator = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + coordinator.fault_for_test(2); + let ticket = coordinator + .submit(s.ready_renew(identity()?, DEFAULT_LEASE_MS).await?) + .await?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait()).await?, + PublicationState::Uncertain(_) + )); + s.fence(); + coordinator.recover(&ticket).await?; + let result = changed(timeout(Duration::from_secs(10), ticket.wait()).await?)?; + assert!(result.session.is_err()); + assert!(s.ready_renew(identity()?, DEFAULT_LEASE_MS).await.is_err()); + assert!(s.live_lease().is_err()); + assert!(coordinator.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn bound_lease_claim_after_actual_owner_restore_uses_new_fence_and_preserves_original_pin() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let original = session(&f, [201; 16]).await?; + let old = original.lease.token; + f.handle.drain().await?; + f.runtime.shutdown().await?; + let owner_session = SessionId::from_bytes([202; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, owner_session)?; + let authority = CellAuthority::new(f.layout.clone()); + let idle = authority.load(f.target.cell_id()).await?.ok_or("idle")?; + let provision = CellCatalog::new(f.layout.clone(), f.target.tenant()) + .lookup(f.target.cell_id()) + .await? + .ok_or("provision")?; + let handle = runtime + .acquire_idle_restored( + provision, + f.replica.clone(), + authority, + idle, + f.root.path().join("bound-restored.sqlite"), + Owner { + session: owner_session, + endpoint: "https://bound-restored.invalid".into(), + }, + ) + .await?; + let client = CellClient::local(Arc::clone(&f.registry), handle.clone()); + let coordinator = + PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let mutation = identity()?; + coordinator.fault_for_test(2); + let ticket = coordinator + .submit( + ReadyPreparation::claim( + client.clone(), + f.target.clone(), + request_for(&original), + mutation, + ) + .await?, + ) + .await?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait()).await?, + PublicationState::Uncertain(_) + )); + drop(ticket); + let retained = coordinator + .pending(old.operation) + .await + .ok_or("restored Claim retained")?; + coordinator.recover(&retained).await?; + let outcome = changed(timeout(Duration::from_secs(10), retained.wait()).await?)?; + let replay = client + .command::(&f.target, mutation, request_for(&original)) + .await?; + assert_eq!(outcome.committed, replay); + let current = outcome.session.map_err(|e| e.to_string())?; + let next = current.live_lease()?.0; + assert_ne!(next.token.owner, old.owner); + assert_ne!(next.token.artifact_operation, old.artifact_operation); + assert_eq!(next.base, original.lease.base); + let old_pin = handle.query(0, 32, move |conn| { + Ok(conn.query_row("SELECT generation FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2", rusqlite::params![old.owner.incarnation.as_bytes().as_slice(), old.attempt as i64], |row| row.get::<_, i64>(0))?.to_be_bytes().to_vec()) + }).await?; + assert_eq!(old_pin.as_slice(), 0i64.to_be_bytes()); + let renewed = coordinator + .submit(current.ready_renew(identity()?, DEFAULT_LEASE_MS).await?) + .await?; + let renewed = changed(timeout(Duration::from_secs(10), renewed.wait()).await?)?; + assert_eq!(renewed.kind, PreparationCommandKind::Renew); + assert!(renewed.session.is_ok()); + assert!(coordinator.close_and_drain().await.is_empty()); + runtime.shutdown().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/frontier.rs b/crates/canopy-server/src/packs/publication/tests/frontier.rs new file mode 100644 index 0000000..4188e12 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/frontier.rs @@ -0,0 +1,453 @@ +use super::*; + +async fn refresh( + fixture: &Fixture, + token: PreparationToken, +) -> Result> { + Ok(fixture + .client() + .query::(&fixture.target, None, check(token)) + .await? + .output) +} +pub(super) async fn reap(fixture: &Fixture) -> Result { + Ok(fixture + .client() + .command::( + &fixture.target, + identity()?, + MaintenanceRequest { + repository: fixture.repository, + actor: "owner".into(), + owner: fixture.handle.owner_fence(), + }, + ) + .await? + .output) +} +pub(super) async fn facts(fixture: &Fixture) -> Result> { + Ok(fixture + .handle + .query(0, 4096, |connection| { + let mut query = connection + .prepare("SELECT generation FROM catalog_generations ORDER BY generation")?; + let values = query + .query_map([], |row| row.get::<_, u64>(0))? + .collect::>>()?; + Ok(values.into_iter().flat_map(u64::to_be_bytes).collect()) + }) + .await?) +} +pub(super) fn expected(values: &[u64]) -> Vec { + values + .iter() + .flat_map(|value| value.to_be_bytes()) + .collect() +} + +#[tokio::test] +async fn frontier_reads_latest_root_without_claim_namespace_or_lease_extension() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let started = fixture + .client() + .command::(&fixture.target, identity()?, fixture.begin([70; 16])) + .await?; + let original = lease(started.output)?; + assert_eq!( + refresh(&fixture, original.token) + .await? + .ok_or("frontier")? + .current, + original.base + ); + let first = fixture.install_empty_root(1).await?; + let one = refresh(&fixture, original.token) + .await? + .ok_or("first frontier")?; + assert_eq!(one.current.catalog, Some(first)); + let second = fixture.install_empty_root(2).await?; + let before = fixture + .client() + .query::(&fixture.target, None, check(original.token)) + .await?; + let two = before.output.ok_or("second frontier")?; + assert_eq!(two.current.generation, 2); + assert_eq!(two.current.catalog, Some(second)); + assert_eq!(two.lease.base, original.base); + assert_eq!(two.lease.token, original.token); + assert_eq!(two.lease.expires_at_ms, original.expires_at_ms); + assert!(two.lease.observed_at_ms >= one.lease.observed_at_ms); + let after = fixture + .client() + .query::( + &fixture.target, + Some(before.receipt), + check(original.token), + ) + .await?; + assert_eq!(after.receipt, before.receipt); + assert_eq!(fixture.counts().await?, (1, 1)); + assert_eq!(reap(&fixture).await?, 0); + assert_eq!(facts(&fixture).await?, expected(&[0, 1, 2])); + let mut encoder = BoundedEncoder::new(1024)?; + two.encode(&mut encoder)?; + let bytes = encoder.finish(); + let mut decoder = BoundedDecoder::new(&bytes, 1024)?; + assert_eq!(PreparationFrontier::decode(&mut decoder)?, two); + decoder.finish()?; + // A genuinely new operation gets the next namespace, not one consumed + // by any of the frontier queries above. + let next = lease( + fixture + .client() + .command::(&fixture.target, identity()?, fixture.begin([71; 16])) + .await? + .output, + )?; + assert_eq!(next.token.artifact_operation, artifact_number(2)); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn detached_attempt_floor_retains_intervening_roots_until_reaped() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + for n in 1..=3 { + fixture.install_empty_root(n).await?; + } + let first = lease( + fixture + .client() + .command::(&fixture.target, identity()?, fixture.begin([72; 16])) + .await? + .output, + )?; + fixture.install_empty_root(4).await?; + fixture.install_empty_root(5).await?; + let next = lease( + fixture + .client() + .command::(&fixture.target, identity()?, request(first.token)) + .await? + .output, + )?; + fixture.install_empty_root(6).await?; + assert!(refresh(&fixture, first.token).await?.is_none()); + assert_eq!( + refresh(&fixture, next.token) + .await? + .ok_or("claimed frontier")? + .current + .generation, + 6 + ); + // The superseded pin protects 3,4,5,6; the new pin alone would protect 5,6. + assert_eq!(reap(&fixture).await?, 2); + assert_eq!(facts(&fixture).await?, expected(&[0, 3, 4, 5, 6])); + fixture + .client() + .command::(&fixture.target, identity()?, check(next.token)) + .await?; + assert!(refresh(&fixture, next.token).await?.is_none()); + assert_eq!(reap(&fixture).await?, 0); + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([72; 32]), + sql::now(0)?, + 1, + 0, + |tx| { + // Expiration alone does not authorize fact deletion. Removing the + // expired independent pins is the reaper's preceding transaction step. + tx.execute("UPDATE catalog_leases SET expires_at_ms=0", [])?; + assert!( + tx.execute("DELETE FROM catalog_generations WHERE generation=4", []) + .is_err() + ); + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + assert_eq!(reap(&fixture).await?, 5); // two pins and three obsolete facts + assert_eq!(facts(&fixture).await?, expected(&[0, 6])); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn frontier_rechecks_actor_identity_expiry_and_active_attempt() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([73; 32]), + sql::now(0)?, + 1, + 0, + |tx| { + tx.execute( + "INSERT INTO repository_members VALUES('writer','write')", + [], + )?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + let mut input = fixture.begin([73; 16]); + input.actor = "writer".into(); + let original = lease( + fixture + .client() + .command::(&fixture.target, identity()?, input) + .await? + .output, + )?; + let correct = LeaseCheck { + token: original.token, + actor: "writer".into(), + }; + for mut wrong in [ + correct.clone(), + correct.clone(), + correct.clone(), + correct.clone(), + ] + .into_iter() + .enumerate() + { + match wrong.0 { + 0 => wrong.1.actor = "owner".into(), + 1 => wrong.1.token.repository = *uuid::Uuid::new_v4().as_bytes(), + 2 => wrong.1.token.request_digest[0] ^= 1, + _ => wrong.1.token.artifact_operation = artifact_number(2), + } + assert!( + fixture + .client() + .query::(&fixture.target, None, wrong.1) + .await? + .output + .is_none() + ); + } + assert!( + fixture + .client() + .query::(&fixture.target, None, correct.clone()) + .await? + .output + .is_some() + ); + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([74; 32]), + sql::now(0)?, + 1, + 0, + |tx| { + tx.execute( + "UPDATE repository_members SET role='read' WHERE account='writer'", + [], + )?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + assert!( + fixture + .client() + .query::(&fixture.target, None, correct.clone()) + .await? + .output + .is_none() + ); + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([75; 32]), + sql::now(0)?, + 1, + 0, + |tx| { + tx.execute( + "UPDATE repository_members SET role='write' WHERE account='writer'", + [], + )?; + tx.execute("UPDATE catalog_operations SET expires_at_ms=0", [])?; + tx.execute("UPDATE catalog_leases SET expires_at_ms=0", [])?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + assert!( + fixture + .client() + .query::(&fixture.target, None, correct) + .await? + .output + .is_none() + ); + assert_eq!(fixture.counts().await?, (1, 1)); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[test] +fn floor_reaping_is_indexed_bounded_and_cannot_remove_intervening_roots() -> Result { + let mut connection = rusqlite::Connection::open_in_memory()?; + connection.execute_batch("PRAGMA foreign_keys=ON")?; + connection.execute_batch(SCHEMA)?; + let tx = connection.transaction()?; + for generation in 1..=1200 { + tx.execute( + "INSERT INTO catalog_generations(generation,catalog,certificate) VALUES(?1,x'01',zeroblob(32))", + [generation], + )?; + } + tx.execute("UPDATE catalog_state SET generation=1200", [])?; + tx.execute("INSERT INTO catalog_leases(incarnation,admission_sequence,operation,owner_epoch,artifact_operation,generation,expires_at_ms) VALUES(zeroblob(16),1,zeroblob(16),x'0000000000000001',?1,900,0)", [artifact_number(1).as_slice()])?; + tx.commit()?; + for generation in [0, 900, 901, 1199] { + assert!( + connection + .execute( + "DELETE FROM catalog_generations WHERE generation=?1", + [generation] + ) + .is_err() + ); + } + let mut explain = connection.prepare(&format!( + "EXPLAIN QUERY PLAN {}", + commands::REAP_GENERATIONS + ))?; + let plan = explain + .query_map([REAP_ROWS], |row| row.get::<_, String>(3))? + .collect::>>()?; + assert!( + plan.iter() + .any(|line| line.contains("SEARCH g USING PRIMARY KEY")), + "{plan:?}" + ); + assert!( + plan.iter() + .any(|line| line.contains("USING COVERING INDEX catalog_leases_by_generation")), + "{plan:?}" + ); + assert!(!plan.iter().any(|line| line.contains("SCAN g")), "{plan:?}"); + drop(explain); + assert_eq!( + connection.execute(commands::REAP_GENERATIONS, [REAP_ROWS])?, + REAP_ROWS as usize + ); + assert_eq!( + connection.execute(commands::REAP_GENERATIONS, [REAP_ROWS])?, + 899 - REAP_ROWS as usize + ); + assert_eq!( + connection.execute(commands::REAP_GENERATIONS, [REAP_ROWS])?, + 0 + ); + let retained: u64 = connection.query_row( + "SELECT count(*) FROM catalog_generations WHERE generation BETWEEN 900 AND 1200", + [], + |row| row.get(0), + )?; + assert_eq!(retained, 301); + connection.execute("DELETE FROM catalog_leases", [])?; + assert_eq!( + connection.execute(commands::REAP_GENERATIONS, [REAP_ROWS])?, + 300 + ); + assert_eq!( + connection.query_row("SELECT count(*) FROM catalog_generations", [], |row| row + .get::<_, u64>(0))?, + 2 + ); + Ok(()) +} + +#[test] +fn frontier_codec_rejects_rollback_foreign_catalog_and_changed_same_generation() -> Result { + let token = PreparationToken { + repository: *uuid::Uuid::new_v4().as_bytes(), + operation: [76; 16], + artifact_operation: artifact_number(1), + request_digest: [77; 32], + owner: OwnerFence { + incarnation: IncarnationId::from_bytes([78; 16]), + epoch: 1, + }, + attempt: 1, + }; + let empty = GenerationFact { + generation: 0, + catalog: None, + refs: None, + certificate: None, + }; + let mut value = PreparationFrontier { + lease: PreparationLease { + token, + base: empty, + format: ObjectFormat::Sha1, + observed_at_ms: 10, + expires_at_ms: 11, + }, + current: empty, + }; + value.validate()?; + value.lease.observed_at_ms = 11; + assert!(value.validate().is_err()); + value.lease.observed_at_ms = 10; + // Reuse a valid descriptor from the directory/catalog codec fixture shape. + let artifact = canopy_object_storage::artifact::ArtifactDescriptor { + size: 1, + digest: [0; 32], + manifest_digest: [0; 32], + }; + let catalog = StoredCatalog { + repository: token.repository, + operation: artifact_number(1), + format: ObjectFormat::Sha1, + artifact, + }; + value.lease.base = GenerationFact { + generation: 1, + catalog: Some(catalog), + refs: None, + certificate: Some([1; 32]), + }; + assert!(value.validate().is_err()); // current zero is below the floor + value.current = value.lease.base; + value.validate()?; + value.current.certificate = Some([2; 32]); + assert!(value.validate().is_err()); + value.current = GenerationFact { + generation: 2, + catalog: Some(StoredCatalog { + repository: *uuid::Uuid::new_v4().as_bytes(), + ..catalog + }), + refs: None, + certificate: Some([3; 32]), + }; + assert!(value.validate().is_err()); + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/initialization.rs b/crates/canopy-server/src/packs/publication/tests/initialization.rs new file mode 100644 index 0000000..8fff876 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/initialization.rs @@ -0,0 +1,443 @@ +use super::*; +use super::{ + prepare::{cleaned, opened}, + publishing::{edit, plan, state, update}, + reconcile::graph, +}; +use crate::packs::{ + catalog::CatalogSnapshot, metadata::tests::limits, ref_state::RefStateSnapshotRoot, +}; +use canopy_object_storage::artifact::ArtifactStore; +use cellule_ltx::DiskBudget; + +async fn empty( + fixture: &Fixture, + operation: [u8; 16], + store: Arc, +) -> Result<(PreparedCatalog, tempfile::TempDir, DiskBudget)> { + let (base, _, _) = opened(fixture, operation, store).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let prepared = CatalogPreparation::new(root.path(), budget.clone(), base, limits()) + .await? + .finish() + .await?; + Ok((prepared, root, budget)) +} +fn initialized(reply: InitializationReply) -> Result { + match reply { + InitializationReply::Initialized(fact) => Ok(*fact), + InitializationReply::Denied(why) => Err(format!("initialization denied {why:?}").into()), + } +} +async fn reject(fixture: &Fixture, input: InitialRefProof, reason: PreparationDenial) -> Result { + let before = state(&fixture.handle).await?; + let result = fixture + .client() + .command::(&fixture.target, identity()?, input) + .await; + assert!( + matches!(result,Err(InvocationError::Rejected(ref value)) if value.output==InitializationReply::Denied(reason)), + "{result:?}" + ); + assert_eq!(state(&fixture.handle).await?, before); + Ok(()) +} + +#[tokio::test] +async fn fresh_initialization_commits_joint_empty_roots_and_enables_first_ref_preparation() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new(provider.clone(), fixture.repository)); + let (prepared, root, budget) = empty(&fixture, [220; 16], store.clone()).await?; + fn send(value: T) -> T { + value + } + let proof = send(prepared.empty_ref_initialization()).await?; + let retry = prepared.empty_ref_initialization().await?; + assert_eq!(retry, proof); + let mut e = BoundedEncoder::new(INITIALIZATION_BYTES)?; + proof.encode(&mut e)?; + let bytes = e.finish(); + assert!(bytes.len() <= INITIALIZATION_BYTES as usize); + let mut d = BoundedDecoder::new(&bytes, INITIALIZATION_BYTES)?; + assert_eq!(InitialRefProof::decode(&mut d)?, proof); + d.finish()?; + for cut in 0..bytes.len() { + let mut d = BoundedDecoder::new(&bytes[..cut], INITIALIZATION_BYTES)?; + assert!(InitialRefProof::decode(&mut d).is_err()); + } + assert!( + fixture + .client() + .query::(&fixture.target, None, fixture.begin([220; 16])) + .await? + .output + .is_none() + ); + let mutation = identity()?; + let committed = fixture + .client() + .command::(&fixture.target, mutation, proof.clone()) + .await?; + let fact = initialized(committed.output.clone())?; + let mut e = BoundedEncoder::new(512)?; + committed.output.encode(&mut e)?; + let reply_bytes = e.finish(); + let mut d = BoundedDecoder::new(&reply_bytes, 512)?; + assert_eq!(InitializationReply::decode(&mut d)?, committed.output); + d.finish()?; + for invalid in [ + GenerationFact { + generation: 2, + ..fact + }, + GenerationFact { refs: None, ..fact }, + ] { + assert!( + InitializationReply::Initialized(Box::new(invalid)) + .encode(&mut BoundedEncoder::new(512)?) + .is_err() + ); + let mut e = BoundedEncoder::new(512)?; + e.write_u8(0)?; + invalid.encode(&mut e)?; + let bytes = e.finish(); + assert!(InitializationReply::decode(&mut BoundedDecoder::new(&bytes, 512)?).is_err()); + } + assert_eq!( + (fact.generation, fact.catalog, fact.refs), + (1, Some(prepared.catalog()), Some(proof.refs)) + ); + assert_eq!( + fact.certificate, + Some(*blake3::hash(&proof.certificate.bytes()?).as_bytes()) + ); + let snapshot = proof.refs.read(&store).await?; + assert_eq!( + (snapshot.generation, snapshot.root, snapshot.default_branch), + (0, None, "refs/heads/main".into()) + ); + let catalog = CatalogSnapshot::download(&store, prepared.catalog()).await?; + assert!(catalog.sources.is_none()); + assert_eq!( + fixture + .client() + .command::(&fixture.target, mutation, proof.clone()) + .await? + .receipt, + committed.receipt + ); + assert_eq!( + fixture + .client() + .command::(&fixture.target, identity()?, proof.clone()) + .await? + .output, + committed.output + ); + assert_eq!( + fixture + .client() + .query::(&fixture.target, None, fixture.begin([220; 16])) + .await? + .output, + Some(fact) + ); + for mode in 0..4 { + let mut request = fixture.begin([220; 16]); + if mode == 0 { + request.actor = "other".into(); + } else if mode == 1 { + request.request_digest = [91; 32]; + } else if mode == 2 { + request.operation = [221; 16]; + } else { + request.repository[15] ^= 1; + } + assert!( + fixture + .client() + .query::(&fixture.target, None, request) + .await? + .output + .is_none() + ); + } + rejected( + fixture + .client() + .command::(&fixture.target, identity()?, fixture.begin([220; 16])) + .await, + PreparationDenial::Conflict, + ); + let next = graph(&fixture, provider, store.clone(), [222; 16], 4).await?; + assert_eq!(next.prepared.base().refs, Some(proof.refs)); + let planned = plan(vec![update("refs/heads/main", None, Some(next.initial))]); + let refs = next.prepared.prepare_ref_snapshot(&planned).await?; + assert_eq!(refs.snapshot().read(&store).await?.generation, 1); + assert!(matches!( + next.prepared.empty_ref_initialization().await, + Err(InitializationPreparationError::Ineligible) + )); + for sql in [ + "UPDATE catalog_initialization SET actor='other'", + "DELETE FROM catalog_initialization", + "INSERT OR REPLACE INTO catalog_initialization SELECT * FROM catalog_initialization", + ] { + let before = state(&fixture.handle).await?; + assert!(edit(&fixture, sql).await.is_err()); + assert_eq!(state(&fixture.handle).await?, before); + } + drop(next.prepared); + cleaned(next.root.path(), &next.budget).await?; + drop(prepared); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn initialization_refuses_history_head_changes_revocation_expiry_and_forged_authority() +-> Result { + for (sql, reason) in [ + ( + "INSERT INTO refs VALUES('refs/heads/deleted',NULL,1)", + PreparationDenial::Conflict, + ), + ( + "UPDATE ref_generation SET generation=1", + PreparationDenial::Conflict, + ), + ( + "UPDATE ref_generation SET default_branch='refs/heads/other'", + PreparationDenial::Conflict, + ), + ( + "UPDATE repository_identity SET owner='replacement'", + PreparationDenial::Unauthorized, + ), + ( + "UPDATE catalog_leases SET expires_at_ms=0; UPDATE catalog_operations SET expires_at_ms=0", + PreparationDenial::Expired, + ), + ] { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let (prepared, root, budget) = empty(&fixture, [223; 16], store).await?; + let proof = prepared.empty_ref_initialization().await?; + edit(&fixture, sql).await?; + reject(&fixture, proof, reason).await?; + drop(prepared); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + } + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let (prepared, root, budget) = empty(&fixture, [224; 16], store).await?; + let proof = prepared.empty_ref_initialization().await?; + let mut stale = proof.clone(); + let mut data = stale.certificate.data()?; + data.token.owner.epoch += 1; + stale.certificate = CatalogCertificate::seal(&data, &[16; 32])?; + reject(&fixture, stale, PreparationDenial::Stale).await?; + let mut forged = proof.clone(); + forged.certificate = CatalogCertificate::seal(&proof.certificate.data()?, &[17; 32])?; + reject(&fixture, forged, PreparationDenial::Unauthorized).await?; + let mut wrong = proof.clone(); + let mut data = wrong.certificate.data()?; + data.tenant = [96; 16]; + wrong.certificate = CatalogCertificate::seal(&data, &[16; 32])?; + reject(&fixture, wrong, PreparationDenial::Unauthorized).await?; + let mut e = BoundedEncoder::new(128)?; + proof.refs.encode(&mut e)?; + let mut bytes = e.finish(); + let end = bytes.len() - 1; + bytes[end] ^= 1; + let mut d = BoundedDecoder::new(&bytes, 128)?; + let refs = RefStateSnapshotRoot::decode(&mut d)?; + d.finish()?; + let raw = InitialRefProof { + refs, + certificate: proof.certificate, + }; + assert!( + raw.encode(&mut BoundedEncoder::new(INITIALIZATION_BYTES)?) + .is_err() + ); + drop(prepared); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn initialization_late_failure_rolls_back_roots_checkpoint_and_outcome_and_races_commit_once() +-> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let (first, root_a, budget_a) = empty(&fixture, [225; 16], store.clone()).await?; + let (second, root_b, budget_b) = empty(&fixture, [226; 16], store).await?; + let a = first.empty_ref_initialization().await?; + let b = second.empty_ref_initialization().await?; + edit(&fixture,"CREATE TRIGGER fail_initialization BEFORE INSERT ON catalog_initialization BEGIN SELECT RAISE(ABORT,'late initialization fault'); END;").await?; + let before = state(&fixture.handle).await?; + assert!( + fixture + .client() + .command::(&fixture.target, identity()?, a.clone()) + .await + .is_err() + ); + assert_eq!(state(&fixture.handle).await?, before); + assert!( + fixture + .client() + .query::(&fixture.target, None, fixture.begin([225; 16])) + .await? + .output + .is_none() + ); + edit(&fixture, "DROP TRIGGER fail_initialization").await?; + let client = fixture.client(); + let (result_a, result_b) = tokio::join!( + client.command::(&fixture.target, identity()?, a.clone()), + client.command::(&fixture.target, identity()?, b.clone()), + ); + let (committed, losing, winner, loser) = match (result_a, result_b) { + (Ok(committed), Err(InvocationError::Rejected(rejected))) => { + assert_eq!( + rejected.output, + InitializationReply::Denied(PreparationDenial::Conflict) + ); + (committed, b, [225; 16], [226; 16]) + } + (Err(InvocationError::Rejected(rejected)), Ok(committed)) => { + assert_eq!( + rejected.output, + InitializationReply::Denied(PreparationDenial::Conflict) + ); + (committed, a, [226; 16], [225; 16]) + } + other => { + return Err(format!("initialization must have exactly one winner: {other:?}").into()); + } + }; + let fact = initialized(committed.output)?; + assert_eq!(fact.generation, 1); + assert_eq!( + client + .query::(&fixture.target, None, fixture.begin(winner)) + .await? + .output, + Some(fact) + ); + assert!( + client + .query::(&fixture.target, None, fixture.begin(loser)) + .await? + .output + .is_none() + ); + reject(&fixture, losing, PreparationDenial::Conflict).await?; + drop(first); + drop(second); + cleaned(root_a.path(), &budget_a).await?; + cleaned(root_b.path(), &budget_b).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn initialization_exact_outcome_survives_owner_restore_and_pending_old_attempts_cannot_write() +-> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let (first, root_a, budget_a) = empty(&fixture, [227; 16], store.clone()).await?; + let (second, root_b, budget_b) = empty(&fixture, [228; 16], store).await?; + let a = first.empty_ref_initialization().await?; + let b = second.empty_ref_initialization().await?; + let mutation = identity()?; + let committed = fixture + .client() + .command::(&fixture.target, mutation, a.clone()) + .await?; + fixture.handle.drain().await?; + fixture.runtime.shutdown().await?; + let session = SessionId::from_bytes([229; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, session)?; + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await? + .ok_or("idle")?; + let provision = CellCatalog::new(fixture.layout.clone(), fixture.target.tenant()) + .lookup(fixture.target.cell_id()) + .await? + .ok_or("provision")?; + let handle = runtime + .acquire_idle_restored( + provision, + fixture.replica.clone(), + authority, + idle, + fixture.root.path().join("initialization-restored.sqlite"), + Owner { + session, + endpoint: "https://initialization-restored.invalid".into(), + }, + ) + .await?; + assert!(handle.owner_fence().epoch > first.token().owner.epoch); + let client = CellClient::local(Arc::clone(&fixture.registry), handle.clone()); + let exact = client + .command::(&fixture.target, mutation, a.clone()) + .await?; + assert_eq!( + (exact.output, exact.receipt), + (committed.output.clone(), committed.receipt) + ); + assert_eq!( + client + .command::(&fixture.target, identity()?, a) + .await? + .output, + committed.output + ); + let before = state(&handle).await?; + let stale = client + .command::(&fixture.target, identity()?, b) + .await; + assert!( + matches!(stale,Err(InvocationError::Rejected(ref value)) if value.output==InitializationReply::Denied(PreparationDenial::Stale)) + ); + assert_eq!(state(&handle).await?, before); + assert_eq!( + client + .query::(&fixture.target, None, fixture.begin([227; 16])) + .await? + .output, + Some(initialized(committed.output)?) + ); + drop(first); + drop(second); + cleaned(root_a.path(), &budget_a).await?; + cleaned(root_b.path(), &budget_b).await?; + runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/inputs.rs b/crates/canopy-server/src/packs/publication/tests/inputs.rs new file mode 100644 index 0000000..c35dad5 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/inputs.rs @@ -0,0 +1,1163 @@ +use super::*; +mod bound; +mod custody; +mod requests; +use crate::packs::sources::{NativeInputIndex, NativePackDescriptor}; +use canopy_object_storage::artifact::{ArtifactDescriptor, ArtifactStore}; +use tokio::time::{Duration, timeout}; + +async fn capture_real( + context: StagingContext, + store: Arc, +) -> std::result::Result> { + let source = crate::packs::metadata::tests::fixture(context.format(), 4) + .await + .map_err(|e| std::io::Error::other(e.to_string()))?; + let root = tempfile::TempDir::new()?; + let disk = cellule_ltx::DiskBudget::new(16 << 20); + let backend = crate::git_http::GitHttpBackend::initialize( + root.path().into(), + disk, + "refs/heads/main", + context.format(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + for entry in std::fs::read_dir(source.root.path().join("objects/pack"))? { + let entry = entry?; + std::fs::copy( + entry.path(), + backend + .git_dir() + .join("objects/pack") + .join(entry.file_name()), + )?; + } + let inputs = backend + .stage_native_packs( + &context, + &store, + crate::packs::verification::physical::tests::physical_limits(), + ) + .await?; + Ok(context.seal_native_inputs(store, inputs).await?) +} + +pub(super) async fn active( + fixture: &Fixture, + operation: [u8; 16], +) -> Result<(StagingCoordinator, StagingTicket)> { + let coordinator = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let ready = ReadyStaging::new( + fixture.client(), + fixture.target.clone(), + fixture.begin(operation), + identity()?, + ) + .await?; + let ticket = coordinator.submit(ready).map_err(|(error, _)| error)?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait()).await?, + StagingState::Active(_) + )); + Ok((coordinator, ticket)) +} +fn records( + repository: [u8; 16], + operation: [u8; 16], + format: ObjectFormat, + count: u32, +) -> impl Iterator + Send { + (1..=count).map(move |n| { + let digest = *blake3::hash(&n.to_be_bytes()).as_bytes(); + let git_checksum = crate::ObjectId::try_from(&digest[..format.bytes()]).unwrap(); + let artifact = |size| ArtifactDescriptor { + size, + digest, + manifest_digest: [18; 32], + }; + NativePackDescriptor { + repository, + operation, + format, + git_checksum, + object_count: 1, + pack: artifact(100), + index: artifact(8 + 1024 + (format.bytes() as u64 + 8) + 2 * format.bytes() as u64), + } + }) +} +pub(super) async fn seal( + fixture: &Fixture, + ticket: &StagingTicket, + store: Arc, + count: u32, +) -> Result { + let repository = fixture.repository; + let format = fixture.format; + let task = ticket.spawn(move |context| async move { + let token = context.token()?; + context + .seal_native_inputs( + store, + records(repository, token.artifact_operation, format, count), + ) + .await + .map_err(|e| StagingError::Input(Box::new(e))) + })?; + task.wait() + .await + .map_err(|e| format!("input factory: {e:?}").into()) +} +async fn check( + client: &CellClient, + target: &CellTarget, + token: PreparationToken, +) -> Result> { + Ok(client + .query::( + target, + None, + LeaseCheck { + token, + actor: "owner".into(), + }, + ) + .await? + .output) +} +fn denied( + result: std::result::Result< + cellule_runtime::Committed, + InvocationError, + >, + reason: PreparationDenial, +) { + assert!( + matches!(result, Err(InvocationError::Rejected(ref value)) if value.output == StagingReply::Denied(reason)) + ); +} +async fn mutate(handle: &CellHandle, statement: String) -> Result { + let digest = Digest::from_bytes(*blake3::hash(statement.as_bytes()).as_bytes()); + handle + .execute( + identity()?, + digest, + sql::now(0)?, + statement.len(), + 0, + move |tx| { + tx.execute_batch(&statement)?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + Ok(()) +} + +#[tokio::test] +async fn staged_inputs_reuse_bounded_index_and_checkpoint_is_immutable() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let (coordinator, ticket) = active(&fixture, [170; 16]).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let proof = seal(&fixture, &ticket, store.clone(), 300).await?; + let root = proof.root()?.ok_or("root")?; + assert_eq!((root.record_count, root.object_count), (300, 300)); + assert!(root.height > 0 && root.artifact.size <= 64 << 10); + let mut e = BoundedEncoder::new(CERTIFICATE_BYTES)?; + proof.encode(&mut e)?; + assert!(e.finish().len() < 1024); + let before = fixture.counts().await?; + let mutation = identity()?; + let first = fixture + .client() + .command::(&fixture.target, mutation, proof.clone()) + .await?; + let replay = fixture + .client() + .command::(&fixture.target, mutation, proof.clone()) + .await?; + assert_eq!(first.receipt, replay.receipt); + assert_eq!( + check(&fixture.client(), &fixture.target, proof.token()?).await?, + Some(proof.clone()) + ); + assert_eq!(fixture.counts().await?, before); + let index = NativeInputIndex::new(store, format); + let mut cursor = index.cursor(Some(root), None)?; + let mut count = 0; + while let Some(native) = cursor.next().await? { + assert_eq!(native.operation, proof.token()?.artifact_operation); + count += 1; + } + assert_eq!(count, 300); + let different = seal( + &fixture, + &ticket, + Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )), + 1, + ) + .await?; + denied( + fixture + .client() + .command::(&fixture.target, identity()?, different) + .await, + PreparationDenial::Conflict, + ); + let sql = cellule_runtime::primitives::sql::SqlCell::::new( + fixture.client(), + fixture.target.clone(), + )?; + let result=sql.query(Some(first.receipt),sql::statement("SELECT input_checkpoint,input_checkpoint_digest,generation FROM catalog_leases",vec![])).await?; + assert!(matches!(sql::rows(&result.output)?[0][2], SqlValue::Null)); + ticket.stop(); + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn input_checkpoint_rejects_tampering_scope_revocation_and_stale_attempt() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let (coordinator, ticket) = active(&fixture, [171; 16]).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let proof = seal(&fixture, &ticket, store, 1).await?; + let mut e = BoundedEncoder::new(CERTIFICATE_BYTES)?; + proof.encode(&mut e)?; + let mut bytes = e.finish(); + let last = bytes.len() - 1; + bytes[last] ^= 1; + let mut d = BoundedDecoder::new(&bytes, CERTIFICATE_BYTES)?; + let tampered = NativeInputCertificate::decode(&mut d)?; + d.finish()?; + denied( + fixture + .client() + .command::(&fixture.target, identity()?, tampered) + .await, + PreparationDenial::Conflict, + ); + assert!( + check(&fixture.client(), &fixture.target, proof.token()?) + .await? + .is_none() + ); + let foreign = Fixture::new(ObjectFormat::Sha256).await?; + denied( + foreign + .client() + .command::(&foreign.target, identity()?, proof.clone()) + .await, + PreparationDenial::Unauthorized, + ); + foreign.runtime.shutdown().await?; + mutate( + &fixture.handle, + "UPDATE repository_identity SET owner='another'".into(), + ) + .await?; + denied( + fixture + .client() + .command::(&fixture.target, identity()?, proof.clone()) + .await, + PreparationDenial::Unauthorized, + ); + assert!( + check(&fixture.client(), &fixture.target, proof.token()?) + .await? + .is_none() + ); + mutate( + &fixture.handle, + "UPDATE repository_identity SET owner='owner'".into(), + ) + .await?; + // Claim gets a new admitted identity even on this same owner. Its previous + // independent pin survives, but an old proof cannot populate the new pin. + let claimed = fixture + .client() + .command::( + &fixture.target, + identity()?, + LeaseRequest { + check: LeaseCheck { + token: proof.token()?, + actor: "owner".into(), + }, + lease_ms: DEFAULT_LEASE_MS, + }, + ) + .await?; + assert!(matches!(claimed.output, StagingReply::Granted(_))); + denied( + fixture + .client() + .command::(&fixture.target, identity()?, proof.clone()) + .await, + PreparationDenial::Stale, + ); + ticket.stop(); + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[test] +fn input_checkpoint_sql_pairs_and_immutability_reject_partial_or_replaced_facts() -> Result { + let connection = rusqlite::Connection::open_in_memory()?; + connection.execute_batch(SCHEMA)?; + connection.execute("INSERT INTO catalog_leases(incarnation,admission_sequence,operation,owner_epoch,artifact_operation,generation,expires_at_ms) VALUES(zeroblob(16),1,zeroblob(16),x'0000000000000001',x'43414e4f505930310000000000000001',NULL,100)", [])?; + for sql in [ + "UPDATE catalog_leases SET input_checkpoint=x'01'", + "UPDATE catalog_leases SET input_checkpoint_digest=zeroblob(32)", + ] { + assert!(connection.execute(sql, []).is_err()); + } + connection.execute( + "UPDATE catalog_leases SET input_checkpoint=x'01',input_checkpoint_digest=zeroblob(32)", + [], + )?; + for sql in [ + "UPDATE catalog_leases SET input_checkpoint=NULL,input_checkpoint_digest=NULL", + "UPDATE catalog_leases SET input_checkpoint=x'02'", + "UPDATE catalog_leases SET input_checkpoint_digest=randomblob(32)", + ] { + assert!(connection.execute(sql, []).is_err()); + } + connection.execute( + "UPDATE catalog_leases SET input_checkpoint=x'01',input_checkpoint_digest=zeroblob(32)", + [], + )?; + Ok(()) +} + +#[test] +fn input_checkpoint_sql_append_requires_exact_predecessor_unbound_phase_and_revision_capacity() +-> Result { + let connection = rusqlite::Connection::open_in_memory()?; + connection.execute_batch(SCHEMA)?; + connection.execute("INSERT INTO catalog_leases(incarnation,admission_sequence,operation,owner_epoch,artifact_operation,generation,expires_at_ms) VALUES(zeroblob(16),1,zeroblob(16),x'0000000000000001',x'43414e4f505930310000000000000001',NULL,100)", [])?; + connection.execute( + "UPDATE catalog_leases SET input_checkpoint=x'01',input_checkpoint_digest=zeroblob(32)", + [], + )?; + for invalid in [ + "UPDATE catalog_leases SET input_checkpoint=x'02',input_checkpoint_digest=randomblob(32),input_checkpoint_revision=1,input_checkpoint_previous_digest=randomblob(32)", + "UPDATE catalog_leases SET input_checkpoint=x'02',input_checkpoint_digest=randomblob(32),input_checkpoint_revision=2,input_checkpoint_previous_digest=input_checkpoint_digest", + "UPDATE catalog_leases SET input_checkpoint=x'02',input_checkpoint_revision=1,input_checkpoint_previous_digest=input_checkpoint_digest", + "UPDATE catalog_leases SET generation=0,input_checkpoint=x'02',input_checkpoint_digest=randomblob(32),input_checkpoint_revision=1,input_checkpoint_previous_digest=input_checkpoint_digest", + ] { + assert!(connection.execute(invalid, []).is_err(), "{invalid}"); + } + connection.execute_batch("SAVEPOINT bound; UPDATE catalog_leases SET generation=0;")?; + assert!(connection.execute("UPDATE catalog_leases SET input_checkpoint=x'02',input_checkpoint_digest=randomblob(32),input_checkpoint_revision=1,input_checkpoint_previous_digest=input_checkpoint_digest", []).is_err()); + connection.execute_batch("ROLLBACK TO bound; RELEASE bound;")?; + for revision in 1i64..=256 { + let prior: Vec = connection.query_row( + "SELECT input_checkpoint_digest FROM catalog_leases", + [], + |row| row.get(0), + )?; + let bytes = revision.to_be_bytes(); + let digest = *blake3::hash(&bytes).as_bytes(); + assert_eq!(connection.execute("UPDATE catalog_leases SET input_checkpoint=?1,input_checkpoint_digest=?2,input_checkpoint_previous_digest=?3,input_checkpoint_revision=?4 WHERE input_checkpoint_digest=?3",rusqlite::params![bytes.as_slice(),digest.as_slice(),prior,revision])?, 1); + assert_eq!(connection.execute("UPDATE catalog_leases SET input_checkpoint=input_checkpoint,input_checkpoint_digest=input_checkpoint_digest,input_checkpoint_previous_digest=input_checkpoint_previous_digest,input_checkpoint_revision=input_checkpoint_revision", [])?, 1); + } + assert!(connection.execute("UPDATE catalog_leases SET input_checkpoint=x'03',input_checkpoint_digest=randomblob(32),input_checkpoint_revision=257,input_checkpoint_previous_digest=input_checkpoint_digest", []).is_err()); + connection.execute("UPDATE catalog_leases SET generation=0", [])?; + assert!(connection.execute("UPDATE catalog_leases SET input_checkpoint=x'04',input_checkpoint_digest=randomblob(32),input_checkpoint_revision=257,input_checkpoint_previous_digest=input_checkpoint_digest", []).is_err()); + assert_eq!( + connection.query_row( + "SELECT input_checkpoint_revision FROM catalog_leases", + [], + |row| row.get::<_, i64>(0) + )?, + 256 + ); + Ok(()) +} + +#[tokio::test] +async fn source_pin_expiry_after_reconstruction_refuses_final_adoption() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let (old_coordinator, old_ticket) = active(&fixture, [174; 16]).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let old = seal(&fixture, &old_ticket, store.clone(), 1).await?; + fixture + .client() + .command::(&fixture.target, identity()?, old.clone()) + .await?; + old_ticket.stop(); + assert!(old_coordinator.close_and_drain().await.is_empty()); + let coordinator = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let ready = ReadyStaging::claim( + fixture.client(), + fixture.target.clone(), + LeaseRequest { + check: LeaseCheck { + token: old.token()?, + actor: "owner".into(), + }, + lease_ms: DEFAULT_LEASE_MS, + }, + identity()?, + ) + .await?; + let ticket = coordinator.submit(ready).map_err(|(e, _)| e)?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait()).await?, + StagingState::Active(_) + )); + let previous = old.clone(); + let task = ticket.spawn(move |context| async move { + context + .adopt_native_inputs(store, &previous) + .await + .map_err(|e| StagingError::Input(Box::new(e))) + })?; + let adopted = task + .wait() + .await + .map_err(|e| format!("reconstruct: {e:?}"))?; + mutate( + &fixture.handle, + format!( + "UPDATE catalog_leases SET expires_at_ms=0 WHERE admission_sequence={}", + old.token()?.attempt + ), + ) + .await?; + assert!( + check(&fixture.client(), &fixture.target, old.token()?) + .await? + .is_none() + ); + denied( + fixture + .client() + .command::(&fixture.target, identity()?, adopted.clone()) + .await, + PreparationDenial::Expired, + ); + assert!( + check(&fixture.client(), &fixture.target, adopted.token()?) + .await? + .is_none() + ); + ticket.stop(); + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn bound_preparation_claim_adopts_exact_input_root_without_copying_nodes() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let (coordinator, ticket) = active(&fixture, [175; 16]).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let prior = seal(&fixture, &ticket, store.clone(), 300).await?; + fixture + .client() + .command::(&fixture.target, identity()?, prior.clone()) + .await?; + ticket.seal()?; + let StagingState::Bound(bound) = + timeout(Duration::from_secs(10), ticket.wait_terminal()).await? + else { + return Err("bind".into()); + }; + assert!(coordinator.close_and_drain().await.is_empty()); + let claimed = fixture + .client() + .command::( + &fixture.target, + identity()?, + LeaseRequest { + check: LeaseCheck { + token: bound.lease.token, + actor: "owner".into(), + }, + lease_ms: DEFAULT_LEASE_MS, + }, + ) + .await?; + let next = lease(claimed.output)?; + let session = Arc::new( + PreparationSession::open( + fixture.client(), + fixture.target.clone(), + LeaseCheck { + token: next.token, + actor: "owner".into(), + }, + Some(claimed.receipt), + ) + .await?, + ); + let adopted = session.adopt_native_inputs(store, &prior).await?; + assert_eq!(adopted.root()?, prior.root()?); + assert_eq!(adopted.token()?, next.token); + assert_ne!( + adopted.token()?.artifact_operation, + prior.token()?.artifact_operation + ); + let publisher = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let ready = session.ready_inputs(identity()?, adopted.clone()).await?; + let registered = publisher.submit(ready).await?; + let PublicationState::Finished(Ok(PublicationOutcome::Inputs(result))) = + timeout(Duration::from_secs(10), registered.wait()).await? + else { + return Err("bound registration".into()); + }; + assert!(result.custody.is_ok()); + assert!(publisher.close_and_drain().await.is_empty()); + assert_eq!( + check(&fixture.client(), &fixture.target, next.token).await?, + Some(adopted) + ); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn claimed_staging_retains_exact_dispatch_after_absence_lost_ack_and_panic() -> Result { + for fault in [1, 2, 3] { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let (old_coordinator, old_ticket) = active(&fixture, [176; 16]).await?; + let StagingState::Active(old) = old_ticket.state() else { + return Err("old active".into()); + }; + old_ticket.stop(); + assert!(old_coordinator.close_and_drain().await.is_empty()); + let coordinator = + StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + coordinator.fault_for_test(fault); + let ready = ReadyStaging::claim( + fixture.client(), + fixture.target.clone(), + LeaseRequest { + check: LeaseCheck { + token: old.token, + actor: "owner".into(), + }, + lease_ms: DEFAULT_LEASE_MS, + }, + identity()?, + ) + .await?; + let ticket = coordinator.submit(ready).map_err(|(e, _)| e)?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait()).await?, + StagingState::Uncertain(_) + )); + let evidence = match ticket.state() { + StagingState::Uncertain(error) => match &*error { + StagingError::Claim(error) => match &**error { + InvocationError::Pending(pending) => (**pending).clone(), + _ => return Err("claim evidence".into()), + }, + _ => return Err("claim type".into()), + }, + _ => return Err("uncertain".into()), + }; + drop(ticket); + let retained = coordinator + .pending(old.token.operation) + .ok_or("retained claim")?; + assert_eq!(coordinator.stats().command_bytes, 8192); + coordinator.recover(&retained)?; + let StagingState::Active(next) = timeout(Duration::from_secs(10), retained.wait()).await? + else { + return Err("resolved claim".into()); + }; + assert_ne!(next.token, old.token); + assert_eq!(fixture.counts().await?, (1, 2)); + assert!(matches!( + fixture.client().resolve(&evidence).await?, + cellule_runtime::Resolution::Committed(_) + )); + retained.stop(); + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn restored_owner_claims_and_adopts_only_a_retained_exact_input_checkpoint() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let (coordinator, ticket) = active(&fixture, [172; 16]).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let provider = store.clone(); + let task = ticket.spawn(move |context| async move { + capture_real(context, provider) + .await + .map_err(StagingError::Input) + })?; + let proof = task + .wait() + .await + .map_err(|e| format!("native capture: {e:?}"))?; + let mutation = identity()?; + let first = fixture + .client() + .command::(&fixture.target, mutation, proof.clone()) + .await?; + ticket.stop(); + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.handle.drain().await?; + fixture.runtime.shutdown().await?; + let session = SessionId::from_bytes([173; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, session)?; + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await? + .ok_or("idle")?; + let provision = CellCatalog::new(fixture.layout.clone(), fixture.target.tenant()) + .lookup(fixture.target.cell_id()) + .await? + .ok_or("provision")?; + let handle = runtime + .acquire_idle_restored( + provision, + fixture.replica.clone(), + authority, + idle, + fixture.root.path().join("inputs-b.sqlite"), + Owner { + session, + endpoint: "https://inputs-b.invalid".into(), + }, + ) + .await?; + let client = CellClient::local(Arc::clone(&fixture.registry), handle.clone()); + let replay = client + .command::(&fixture.target, mutation, proof.clone()) + .await?; + assert_eq!(replay.receipt, first.receipt); + denied( + client + .command::(&fixture.target, identity()?, proof.clone()) + .await, + PreparationDenial::Stale, + ); + assert_eq!( + check(&client, &fixture.target, proof.token()?).await?, + Some(proof.clone()) + ); + let coordinator = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let ready = ReadyStaging::claim( + client.clone(), + fixture.target.clone(), + LeaseRequest { + check: LeaseCheck { + token: proof.token()?, + actor: "owner".into(), + }, + lease_ms: DEFAULT_LEASE_MS, + }, + identity()?, + ) + .await?; + let ticket = coordinator.submit(ready).map_err(|(e, _)| e)?; + let StagingState::Active(new) = timeout(Duration::from_secs(10), ticket.wait()).await? else { + return Err("claim".into()); + }; + assert_ne!(new.token.owner, proof.token()?.owner); + assert_ne!( + new.token.artifact_operation, + proof.token()?.artifact_operation + ); + let old = proof.clone(); + let provider = store.clone(); + let task = ticket.spawn(move |context| async move { + context + .adopt_native_inputs(provider, &old) + .await + .map_err(|e| StagingError::Input(Box::new(e))) + })?; + let adopted = task.wait().await.map_err(|e| format!("adopt: {e:?}"))?; + assert_eq!(adopted.token()?, new.token); + assert_eq!(adopted.root()?.ok_or("new root")?.record_count, 1); + assert_eq!(adopted.root()?, proof.root()?); + let registration_identity = identity()?; + let registered = client + .command::(&fixture.target, registration_identity, adopted.clone()) + .await?; + // The committed destination root retains its native incarnations after the + // source pin expires. Re-registration must not require the parent again. + mutate( + &handle, + format!( + "UPDATE catalog_leases SET expires_at_ms=0 WHERE admission_sequence={}", + proof.token()?.attempt + ), + ) + .await?; + assert!( + check(&client, &fixture.target, proof.token()?) + .await? + .is_none() + ); + client + .command::(&fixture.target, identity()?, adopted.clone()) + .await?; + let index = NativeInputIndex::new(store.clone(), fixture.format); + let mut old = index.cursor(proof.root()?, None)?; + let mut new_cursor = index.cursor(adopted.root()?, None)?; + assert_eq!( + check(&client, &fixture.target, new.token).await?, + Some(adopted.clone()) + ); + while let Some(native) = old.next().await? { + assert_eq!(new_cursor.next().await?, Some(native)); + let scratch = tempfile::TempDir::new()?; + let budget = cellule_ltx::DiskBudget::new(64 << 20); + let resources = crate::native_resources::NativeResources::default(); + let mut verifier = crate::packs::verification::PhysicalVerifier::download( + scratch.path(), + budget, + &store, + native, + crate::packs::verification::physical::tests::physical_limits(), + resources.scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + let segment = verifier.inspect_next_shard(native.object_count).await?; + let witness = verifier.finish().await?; + let tip = segment + .headers_after(None)? + .into_iter() + .find(|h| h.object.kind == crate::ObjectKind::Commit) + .ok_or("retained commit")? + .object + .oid; + ticket.seal()?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait_terminal()).await?, + StagingState::Bound(_) + )); + let indexes = Arc::new(crate::packs::catalog::CatalogIndexes::new( + store.clone(), + fixture.format, + )); + let files = Arc::new(crate::packs::catalog::CatalogFiles::new( + scratch.path(), + cellule_ltx::DiskBudget::new(64 << 20), + store.clone(), + fixture.format, + crate::packs::catalog::CatalogFileLimits::default(), + )?); + let base = Arc::new(ticket.open_base(indexes, files).await?); + let mut builder = CatalogPreparation::new( + scratch.path(), + cellule_ltx::DiskBudget::new(64 << 20), + base, + crate::packs::metadata::tests::limits(), + ) + .await?; + builder.begin_retained_pack(witness).await?; + builder.add_segment(segment).await?; + builder.finish_pack().await?; + let prepared = builder.finish().await?; + let publication = prepared + .ref_proof( + super::publishing::plan(vec![super::publishing::update( + "refs/heads/main", + None, + Some(tip), + )]), + scratch.path(), + cellule_ltx::DiskBudget::new(64 << 20), + crate::packs::metadata::tests::limits(), + ) + .await?; + let committed = client + .command::(&fixture.target, identity()?, publication) + .await?; + assert!(matches!(committed.output, PublicationReply::Published(_))); + } + assert!(new_cursor.next().await?.is_none()); + // Completion retires the active operation, while its independent pin and + // the published source roots keep their immutable custody facts. + assert!(check(&client, &fixture.target, new.token).await?.is_none()); + let token = new.token; + let pinned = handle.query(0,32,move |connection| { + Ok(connection.query_row("SELECT input_checkpoint_digest FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2",rusqlite::params![token.owner.incarnation.as_bytes().as_slice(),token.attempt as i64],|row|row.get::<_,Vec>(0))?) + }).await?; + let mut e = BoundedEncoder::new(CERTIFICATE_BYTES)?; + adopted.encode(&mut e)?; + assert_eq!(pinned, blake3::hash(&e.finish()).as_bytes()); + let replay = client + .command::(&fixture.target, registration_identity, adopted) + .await?; + assert_eq!(replay.receipt, registered.receipt); + ticket.stop(); + assert!(coordinator.close_and_drain().await.is_empty()); + runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn service_checkpoint_retains_exact_identity_after_cancellation_absence_lost_reply_and_panic() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in [1, 2, 3] { + let fixture = Fixture::new(format).await?; + let (coordinator, ticket) = active(&fixture, [177; 16]).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let proof = seal(&fixture, &ticket, store, 1).await?; + let mutation = identity()?; + coordinator.fault_for_test(fault); + let observer = ticket + .register_inputs(proof.clone(), mutation) + .map_err(|(e, _)| e)?; + drop(observer); + let StagingState::Uncertain(error) = + timeout(Duration::from_secs(10), ticket.wait_terminal()).await? + else { + return Err("checkpoint uncertainty".into()); + }; + let StagingError::Checkpoint(error) = &*error else { + return Err("checkpoint command".into()); + }; + let InvocationError::Pending(evidence) = &**error else { + return Err("checkpoint evidence".into()); + }; + let evidence = (**evidence).clone(); + let retained = ticket.pending_inputs().ok_or("checkpoint observer")?; + assert!( + matches!(retained.wait().await, Err(e) if matches!(&*e, StagingError::Checkpoint(_))) + ); + assert_eq!(coordinator.stats().command_bytes, 3 * 4096); + // Closing retains unknown registrations and their charged command. + let pending = timeout(Duration::from_secs(10), coordinator.close_and_drain()).await?; + assert_eq!(pending.len(), 1); + coordinator.recover(&pending[0])?; + let receipt = timeout(Duration::from_secs(10), retained.wait()) + .await? + .map_err(|e| format!("registration recovery: {e:?}"))?; + assert!(matches!( + timeout(Duration::from_secs(10), pending[0].wait_terminal()).await?, + StagingState::Stopped + )); + let replay = fixture + .client() + .command::(&fixture.target, mutation, proof.clone()) + .await?; + assert_eq!(receipt, replay.receipt); + let cellule_runtime::Resolution::Committed(outcome) = + fixture.client().resolve(&evidence).await? + else { + return Err("original checkpoint outcome".into()); + }; + assert_eq!(receipt.commit_sequence, outcome.commit_sequence()); + assert_eq!( + check(&fixture.client(), &fixture.target, proof.token()?).await?, + Some(proof) + ); + assert_eq!(fixture.counts().await?, (1, 1)); + assert_eq!(coordinator.stats().admitted, 0); + assert_eq!(coordinator.stats().command_bytes, 0); + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn service_checkpoint_seal_orders_registration_before_binding_and_keeps_original_receipt() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let (coordinator, ticket) = active(&fixture, [178; 16]).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let proof = seal(&fixture, &ticket, store, 1).await?; + let mutation = identity()?; + let observer = ticket + .register_inputs(proof.clone(), mutation) + .map_err(|(e, _)| e)?; + ticket.seal()?; + let StagingState::Bound(bound) = + timeout(Duration::from_secs(10), ticket.wait_terminal()).await? + else { + return Err("checkpoint then bind".into()); + }; + let receipt = observer + .wait() + .await + .map_err(|e| format!("checkpoint: {e:?}"))?; + assert!(receipt.commit_sequence < bound.receipt.commit_sequence); + assert_eq!( + receipt, + ticket + .pending_inputs() + .ok_or("retained receipt")? + .wait() + .await + .map_err(|e| e.to_string())? + ); + assert_eq!( + check(&fixture.client(), &fixture.target, proof.token()?).await?, + Some(proof.clone()) + ); + let replay = fixture + .client() + .command::(&fixture.target, mutation, proof.clone()) + .await?; + assert_eq!(receipt, replay.receipt); + denied( + fixture + .client() + .command::(&fixture.target, identity()?, proof) + .await, + PreparationDenial::Conflict, + ); + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn service_checkpoint_rejects_foreign_attempt_duplicate_and_closed_admission_without_execution() +-> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let (coordinator, ticket) = active(&fixture, [179; 16]).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let proof = seal(&fixture, &ticket, store.clone(), 1).await?; + let ready = ReadyStaging::new( + fixture.client(), + fixture.target.clone(), + fixture.begin([180; 16]), + identity()?, + ) + .await?; + let other = coordinator.submit(ready).map_err(|(e, _)| e)?; + assert!(matches!( + timeout(Duration::from_secs(10), other.wait()).await?, + StagingState::Active(_) + )); + let foreign_attempt = other.register_inputs(proof.clone(), identity()?); + assert!(matches!(foreign_attempt, Err((StagingError::Context, _)))); + assert!(other.pending_inputs().is_none()); + let another = Fixture::new(ObjectFormat::Sha256).await?; + let (another_coordinator, another_ticket) = active(&another, [181; 16]).await?; + let foreign = seal( + &another, + &another_ticket, + Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + another.repository, + )), + 1, + ) + .await?; + assert!(matches!( + ticket.register_inputs(foreign, identity()?), + Err((StagingError::Context, _)) + )); + assert!(ticket.pending_inputs().is_none()); + let mutation = identity()?; + let registration = ticket + .register_inputs(proof.clone(), mutation) + .map_err(|(e, _)| e)?; + let receipt = timeout(Duration::from_secs(10), registration.wait()) + .await? + .map_err(|e| e.to_string())?; + // Wait for the fresh live probe before checking duplicate admission. + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait()).await?, + StagingState::Active(_) + )); + let refused_identity = identity()?; + assert!(matches!( + ticket.register_inputs(proof.clone(), refused_identity), + Err((StagingError::Duplicate, _)) + )); + let unused = fixture + .client() + .prepare_command::(&fixture.target, refused_identity, proof.clone()) + .await?; + assert!(matches!( + fixture.client().resolve(unused.evidence()).await?, + cellule_runtime::Resolution::Absent + )); + other.stop(); + ticket.stop(); + assert!(matches!( + ticket.register_inputs(proof, identity()?), + Err((StagingError::Inactive, _)) + )); + assert_eq!( + registration.wait().await.map_err(|e| e.to_string())?, + receipt + ); + assert!(coordinator.close_and_drain().await.is_empty()); + another_ticket.stop(); + assert!(another_coordinator.close_and_drain().await.is_empty()); + another.runtime.shutdown().await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn service_checkpoint_recovery_preserves_commit_but_refuses_fresh_authority_after_revocation() +-> Result { + for fault in [1, 2] { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let (coordinator, ticket) = active(&fixture, [182; 16]).await?; + let proof = seal( + &fixture, + &ticket, + Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )), + 1, + ) + .await?; + let mutation = identity()?; + coordinator.fault_for_test(fault); + let registration = ticket + .register_inputs(proof.clone(), mutation) + .map_err(|(e, _)| e)?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait_terminal()).await?, + StagingState::Uncertain(_) + )); + mutate( + &fixture.handle, + "UPDATE repository_identity SET owner='other' WHERE singleton=1".into(), + ) + .await?; + coordinator.recover(&ticket)?; + let StagingState::Fenced(_) = + timeout(Duration::from_secs(10), ticket.wait_terminal()).await? + else { + return Err("revoked checkpoint fence".into()); + }; + let result = registration.wait().await; + if fault == 2 { + let receipt = result.map_err(|e| e.to_string())?; + let replay = fixture + .client() + .command::(&fixture.target, mutation, proof.clone()) + .await?; + assert_eq!(receipt, replay.receipt); + } else { + assert!( + matches!(result, Err(e) if matches!(&*e, StagingError::Checkpoint(error) if matches!(&**error, InvocationError::Rejected(value) if value.output == StagingReply::Denied(PreparationDenial::Unauthorized)))) + ); + } + assert!( + check(&fixture.client(), &fixture.target, proof.token()?) + .await? + .is_none() + ); + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn service_checkpoint_absent_recovery_rejects_authoritative_expiry_without_attaching_inventory() +-> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let (coordinator, ticket) = active(&fixture, [183; 16]).await?; + let proof = seal( + &fixture, + &ticket, + Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )), + 1, + ) + .await?; + coordinator.fault_for_test(1); + let registration = ticket + .register_inputs(proof.clone(), identity()?) + .map_err(|(e, _)| e)?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait_terminal()).await?, + StagingState::Uncertain(_) + )); + mutate( + &fixture.handle, + "UPDATE catalog_operations SET expires_at_ms=0; UPDATE catalog_leases SET expires_at_ms=0" + .into(), + ) + .await?; + coordinator.recover(&ticket)?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait_terminal()).await?, + StagingState::Fenced(_) + )); + assert!( + matches!(registration.wait().await, Err(e) if matches!(&*e, StagingError::Checkpoint(error) if matches!(&**error, InvocationError::Rejected(value) if value.output == StagingReply::Denied(PreparationDenial::Expired)))) + ); + let sql = cellule_runtime::primitives::sql::SqlCell::::new( + fixture.client(), + fixture.target.clone(), + )?; + let result = sql + .query( + None, + sql::statement( + "SELECT input_checkpoint,input_checkpoint_digest FROM catalog_leases", + vec![], + ), + ) + .await?; + assert!(matches!( + sql::rows(&result.output)?[0].as_slice(), + [SqlValue::Null, SqlValue::Null] + )); + assert!( + check(&fixture.client(), &fixture.target, proof.token()?) + .await? + .is_none() + ); + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/inputs/bound.rs b/crates/canopy-server/src/packs/publication/tests/inputs/bound.rs new file mode 100644 index 0000000..88a4f45 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/inputs/bound.rs @@ -0,0 +1,561 @@ +use super::*; +use crate::packs::catalog::{CatalogFileLimits, CatalogFiles, CatalogIndexes}; +use crate::packs::metadata::tests::limits; +use crate::packs::verification::{PhysicalVerifier, physical::tests::physical_limits}; +use cellule_ltx::DiskBudget; + +struct Bound { + fixture: Fixture, + store: Arc, + prior: NativeInputCertificate, + proof: NativeInputCertificate, + session: Arc, + coordinator: PublicationCoordinator, +} +impl Bound { + async fn new(format: ObjectFormat, operation: [u8; 16], real: bool) -> Result { + let fixture = Fixture::new(format).await?; + let (staging, ticket) = active(&fixture, operation).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let prior = if real { + let provider = store.clone(); + let work = ticket.spawn(move |context| async move { + capture_real(context, provider) + .await + .map_err(StagingError::Input) + })?; + work.wait().await.map_err(|e| e.to_string())? + } else { + seal(&fixture, &ticket, store.clone(), 300).await? + }; + ticket + .register_inputs(prior.clone(), identity()?) + .map_err(|(e, _)| e)? + .wait() + .await + .map_err(|e| e.to_string())?; + ticket.seal()?; + let StagingState::Bound(bound) = + timeout(Duration::from_secs(10), ticket.wait_terminal()).await? + else { + return Err("bound source".into()); + }; + assert!(staging.close_and_drain().await.is_empty()); + let coordinator = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let claimed = coordinator + .submit( + ReadyPreparation::claim( + fixture.client(), + fixture.target.clone(), + LeaseRequest { + check: LeaseCheck { + token: bound.lease.token, + actor: "owner".into(), + }, + lease_ms: DEFAULT_LEASE_MS, + }, + identity()?, + ) + .await?, + ) + .await?; + let PublicationState::Finished(Ok(PublicationOutcome::Preparation(outcome))) = + timeout(Duration::from_secs(10), claimed.wait()).await? + else { + return Err("supervised bound Claim".into()); + }; + assert_eq!(outcome.kind, PreparationCommandKind::Claim); + let session = outcome.session.map_err(|e| e.to_string())?; + let proof = session.adopt_native_inputs(store.clone(), &prior).await?; + assert_eq!(proof.root()?, prior.root()?); + Ok(Self { + fixture, + store, + prior, + proof, + session, + coordinator, + }) + } + async fn submit(&self, mutation: MutationIdentity) -> Result { + Ok(self + .coordinator + .submit( + self.session + .ready_inputs(mutation, self.proof.clone()) + .await?, + ) + .await?) + } +} +fn registered(state: PublicationState) -> Result { + match state { + PublicationState::Finished(Ok(PublicationOutcome::Inputs(value))) => Ok(value), + state => Err(format!("bound checkpoint outcome: {state:?}").into()), + } +} +#[tokio::test] +async fn bound_checkpoint_keeps_exact_command_through_absence_lost_reply_panic_and_closed_recovery() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in [1, 2, 3] { + let bound = Bound::new(format, [185; 16], false).await?; + let mutation = identity()?; + bound.coordinator.fault_for_test(fault); + let ticket = bound.submit(mutation).await?; + let PublicationState::Uncertain(error) = + timeout(Duration::from_secs(10), ticket.wait()).await? + else { + return Err("bound registration uncertainty".into()); + }; + let PublicationError::Inputs(InvocationError::Pending(evidence)) = &*error else { + return Err("bound exact evidence".into()); + }; + let evidence = (**evidence).clone(); + drop(ticket); + let pending = bound + .coordinator + .pending([185; 16]) + .await + .ok_or("retained bound command")?; + let stats = bound.coordinator.stats().await; + assert_eq!( + ( + stats.admitted, + stats.command_bytes, + stats.foreground, + stats.maintenance + ), + (1, 8192, 1, 0) + ); + let closed = bound.coordinator.close_and_drain().await; + assert_eq!(closed.len(), 1); + bound.coordinator.recover(&pending).await?; + let result = registered(timeout(Duration::from_secs(10), pending.wait()).await?)?; + assert!(result.custody.is_ok()); + assert!(matches!( + result.registration.output, + StagingReply::Granted(_) + )); + let replay = bound + .fixture + .client() + .command::( + &bound.fixture.target, + mutation, + bound.proof.clone(), + ) + .await?; + assert_eq!(result.registration, replay); + let cellule_runtime::Resolution::Committed(outcome) = + bound.fixture.client().resolve(&evidence).await? + else { + return Err("bound original resolution".into()); + }; + assert_eq!(replay.receipt.commit_sequence, outcome.commit_sequence()); + assert_eq!( + check( + &bound.fixture.client(), + &bound.fixture.target, + bound.proof.token()? + ) + .await?, + Some(bound.proof.clone()) + ); + assert!(pending.response().await.is_err()); + assert!(bound.session.live_lease().is_ok()); + assert_eq!(bound.coordinator.stats().await.command_bytes, 0); + assert!(bound.coordinator.close_and_drain().await.is_empty()); + bound.fixture.runtime.shutdown().await?; + } + } + Ok(()) +} +#[tokio::test] +async fn bound_checkpoint_canceled_observer_and_foreign_duplicate_closed_admission_keep_original_ready() +-> Result { + let Bound { + fixture, + store, + prior, + proof, + session, + coordinator, + } = Bound::new(ObjectFormat::Sha256, [186; 16], false).await?; + assert!(matches!( + session.ready_inputs(identity()?, prior).await, + Err(NativeInputReadyError::Codec(_)) + )); + let mutation = identity()?; + let ready = session.ready_inputs(mutation, proof.clone()).await?; + let other = Fixture::new(ObjectFormat::Sha256).await?; + let foreign = PublicationCoordinator::new(other.target.clone(), PublicationLimits::default())?; + let refused = foreign + .submit(ready) + .await + .err() + .ok_or("expected refused admission")?; + assert_eq!(refused.reason, PublicationScheduleError::Foreign); + assert!(matches!(refused.ready, ReadyPublication::Inputs(_))); + assert!( + check(&fixture.client(), &fixture.target, proof.token()?) + .await? + .is_none() + ); + let (release, entered) = coordinator.pause_for_test().await; + let ticket = coordinator.submit(refused.ready).await?; + timeout(Duration::from_secs(5), entered).await??; + let extra = session.ready_inputs(identity()?, proof.clone()).await?; + let refused = coordinator + .submit(extra) + .await + .err() + .ok_or("expected refused admission")?; + assert_eq!(refused.reason, PublicationScheduleError::Duplicate); + let weak = Arc::downgrade(&session); + drop(refused); + drop(session); + drop(ticket); + assert!(weak.upgrade().is_some()); + let pending = coordinator + .pending([186; 16]) + .await + .ok_or("canceled bound observer")?; + assert_eq!(coordinator.reservations_for_test().await, (1, 8192, 1)); + release.send(()).map_err(|_| "bound worker stopped")?; + let result = registered(timeout(Duration::from_secs(10), pending.wait()).await?)?; + assert!(result.custody.is_ok()); + assert!(weak.upgrade().is_none()); + assert_eq!( + result.registration, + fixture + .client() + .command::(&fixture.target, mutation, proof.clone()) + .await? + ); + assert!(coordinator.close_and_drain().await.is_empty()); + let fresh = Arc::new( + PreparationSession::open( + fixture.client(), + fixture.target.clone(), + LeaseCheck { + token: proof.token()?, + actor: "owner".into(), + }, + Some(result.registration.receipt), + ) + .await?, + ); + let ready = fresh.ready_inputs(identity()?, proof).await?; + assert_eq!( + coordinator + .submit(ready) + .await + .err() + .ok_or("expected refused admission")? + .reason, + PublicationScheduleError::Closed + ); + assert!(foreign.close_and_drain().await.is_empty()); + drop(store); + fixture.runtime.shutdown().await?; + other.runtime.shutdown().await?; + Ok(()) +} +#[tokio::test] +async fn bound_checkpoint_committed_recovery_preserves_original_receipt_and_fences_revoked_expired_or_claimed_session() +-> Result { + for mode in [0, 1, 2] { + let bound = Bound::new(ObjectFormat::Sha256, [187; 16], false).await?; + let mutation = identity()?; + bound.coordinator.fault_for_test(2); + let ticket = bound.submit(mutation).await?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait()).await?, + PublicationState::Uncertain(_) + )); + match mode { + 0 => mutate(&bound.fixture.handle, "UPDATE repository_identity SET owner='other' WHERE singleton=1".into()).await?, + 1 => mutate(&bound.fixture.handle, "UPDATE catalog_operations SET expires_at_ms=0; UPDATE catalog_leases SET expires_at_ms=0".into()).await?, + _ => { bound.fixture.client().command::(&bound.fixture.target, identity()?, LeaseRequest { check: bound.session.check.clone(), lease_ms: DEFAULT_LEASE_MS }).await?; } + } + bound.coordinator.recover(&ticket).await?; + let result = registered(timeout(Duration::from_secs(10), ticket.wait()).await?)?; + assert!(result.custody.is_err()); + assert!(bound.session.live_lease().is_err()); + let replay = bound + .fixture + .client() + .command::(&bound.fixture.target, mutation, bound.proof.clone()) + .await?; + assert_eq!(result.registration, replay); + assert!(matches!( + session_ready(&bound).await, + Err(NativeInputReadyError::Base(_)) + )); + assert!(bound.coordinator.close_and_drain().await.is_empty()); + bound.fixture.runtime.shutdown().await?; + } + Ok(()) +} +async fn session_ready( + bound: &Bound, +) -> std::result::Result { + bound + .session + .ready_inputs( + crate::server::mutation_identity().unwrap(), + bound.proof.clone(), + ) + .await +} +#[tokio::test] +async fn bound_checkpoint_absent_recovery_denies_revocation_expiry_and_claim_without_attaching_inventory() +-> Result { + for mode in [0, 1, 2] { + let bound = Bound::new(ObjectFormat::Sha1, [188; 16], false).await?; + bound.coordinator.fault_for_test(1); + let ticket = bound.submit(identity()?).await?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait()).await?, + PublicationState::Uncertain(_) + )); + let reason = match mode { + 0 => { + mutate( + &bound.fixture.handle, + "UPDATE repository_identity SET owner='other' WHERE singleton=1".into(), + ) + .await?; + PreparationDenial::Unauthorized + } + 1 => { + mutate(&bound.fixture.handle, "UPDATE catalog_operations SET expires_at_ms=0; UPDATE catalog_leases SET expires_at_ms=0".into()).await?; + PreparationDenial::Expired + } + _ => { + bound + .fixture + .client() + .command::( + &bound.fixture.target, + identity()?, + LeaseRequest { + check: bound.session.check.clone(), + lease_ms: DEFAULT_LEASE_MS, + }, + ) + .await?; + PreparationDenial::Stale + } + }; + bound.coordinator.recover(&ticket).await?; + let PublicationState::Finished(Err(error)) = + timeout(Duration::from_secs(10), ticket.wait()).await? + else { + return Err("absent bound checkpoint accepted".into()); + }; + assert!( + matches!(&*error, PublicationError::Inputs(InvocationError::Rejected(value)) if value.output == StagingReply::Denied(reason)) + ); + assert!(bound.session.live_lease().is_err()); + let token = bound.proof.token()?; + let count = bound.fixture.handle.query(0,32,move |conn| { Ok(conn.query_row("SELECT count(*) FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2 AND input_checkpoint IS NOT NULL",rusqlite::params![token.owner.incarnation.as_bytes().as_slice(),token.attempt as i64],|row|row.get::<_,i64>(0))?.to_be_bytes().to_vec()) }).await?; + assert_eq!(count.as_slice(), 0i64.to_be_bytes()); + assert!(bound.coordinator.close_and_drain().await.is_empty()); + bound.fixture.runtime.shutdown().await?; + } + Ok(()) +} +#[tokio::test] +async fn bound_checkpoint_real_retained_pair_publishes_after_source_pin_expiry_in_both_formats() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let bound = Bound::new(format, [189; 16], true).await?; + let ticket = bound.submit(identity()?).await?; + let registered = registered(timeout(Duration::from_secs(10), ticket.wait()).await?)?; + assert!(registered.custody.is_ok()); + assert!(bound.coordinator.close_and_drain().await.is_empty()); + let old = bound.prior.token()?; + mutate(&bound.fixture.handle, format!("UPDATE catalog_leases SET expires_at_ms=0 WHERE incarnation=x'{}' AND admission_sequence={}",hex::encode(old.owner.incarnation.as_bytes()),old.attempt)).await?; + let input_index = NativeInputIndex::new(bound.store.clone(), format); + let mut cursor = input_index.cursor(bound.proof.root()?, None)?; + let native = cursor.next().await?.ok_or("retained native input")?; + assert!(cursor.next().await?.is_none()); + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let resources = crate::native_resources::NativeResources::default(); + let mut physical = PhysicalVerifier::download( + root.path(), + budget.clone(), + &bound.store, + native, + physical_limits(), + resources.scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + let segment = physical.inspect_next_shard(native.object_count).await?; + let tip = segment + .headers_after(None)? + .into_iter() + .find(|h| h.object.kind == crate::ObjectKind::Commit) + .ok_or("retained tip")? + .object + .oid; + let witness = physical.finish().await?; + let indexes = Arc::new(CatalogIndexes::new(bound.store.clone(), format)); + let files = Arc::new(CatalogFiles::new( + root.path(), + budget.clone(), + bound.store, + format, + CatalogFileLimits::default(), + )?); + let base = Arc::new( + PreparationBaseResolver::open( + bound.fixture.client(), + bound.fixture.target.clone(), + bound.session.check.clone(), + indexes, + files, + Some(registered.registration.receipt), + ) + .await?, + ); + let mut preparation = + CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + preparation.begin_retained_pack(witness).await?; + preparation.add_segment(segment).await?; + preparation.finish_pack().await?; + let prepared = preparation.finish().await?; + let publication = prepared + .ref_proof( + super::super::publishing::plan(vec![super::super::publishing::update( + "refs/heads/main", + None, + Some(tip), + )]), + root.path(), + budget.clone(), + limits(), + ) + .await?; + let result = bound + .fixture + .client() + .command::(&bound.fixture.target, identity()?, publication) + .await?; + assert!(matches!(result.output, PublicationReply::Published(_))); + bound.fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn bound_checkpoint_shares_push_actor_quotas_and_exact_mixed_byte_admission() -> Result { + use super::super::coordinator::{empty_in_store, finished, refused, request}; + use super::super::prepare::cleaned; + let mut bound = Bound::new(ObjectFormat::Sha256, [190; 16], false).await?; + assert!(bound.coordinator.close_and_drain().await.is_empty()); + bound.coordinator = PublicationCoordinator::new( + bound.fixture.target.clone(), + PublicationLimits { + operations: 6, + per_actor: 2, + command_bytes: (24 << 20) + (16 << 10), + in_flight: 1, + maintenance_operations: 1, + maintenance_in_flight: 1, + foreground_burst: 3, + }, + )?; + mutate(&bound.fixture.handle, "INSERT INTO repository_members VALUES('writer','write'); INSERT INTO repository_members VALUES('third','write')".into()).await?; + let (release, entered) = bound.coordinator.pause_for_test().await; + let input = bound.submit(identity()?).await?; + timeout(Duration::from_secs(5), entered).await??; + let mut pushes = Vec::new(); + for (operation, actor) in [ + (191, "owner"), + (192, "owner"), + (193, "writer"), + (194, "writer"), + (195, "third"), + ] { + let (prepared, root, budget) = + empty_in_store(&bound.fixture, [operation; 16], actor, bound.store.clone()).await?; + let ready = prepared + .ready_push( + identity()?, + request(refused()), + root.path(), + budget.clone(), + limits(), + ) + .await?; + pushes.push((prepared, root, budget, Some(ReadyPublication::from(ready)))); + } + let owner = bound + .coordinator + .submit(pushes[0].3.take().ok_or("owner ready")?) + .await?; + let refused_owner = bound + .coordinator + .submit(pushes[1].3.take().ok_or("quota ready")?) + .await + .err() + .ok_or("owner quota bypassed")?; + assert_eq!(refused_owner.reason, PublicationScheduleError::Capacity); + pushes[1].3 = Some(refused_owner.ready); + let writer_a = bound + .coordinator + .submit(pushes[2].3.take().ok_or("writer ready")?) + .await?; + let writer_b = bound + .coordinator + .submit(pushes[3].3.take().ok_or("writer ready")?) + .await?; + let bytes = bound + .coordinator + .submit(pushes[4].3.take().ok_or("third ready")?) + .await + .err() + .ok_or("mixed byte quota bypassed")?; + assert_eq!(bytes.reason, PublicationScheduleError::Capacity); + pushes[4].3 = Some(bytes.ready); + // One 8 KiB checkpoint plus three 8 MiB push commands. Another account + // still has an operation slot, but no byte credit for its full command. + assert_eq!( + bound.coordinator.reservations_for_test().await, + (4, (24 << 20) + 8192, 2) + ); + release + .send(()) + .map_err(|_| "mixed dispatch worker stopped")?; + assert!( + registered(timeout(Duration::from_secs(10), input.wait()).await?)? + .custody + .is_ok() + ); + for ticket in [owner, writer_a, writer_b] { + finished(timeout(Duration::from_secs(10), ticket.wait()).await?)?; + assert_eq!(ticket.response().await?, refused()); + } + assert_eq!(bound.coordinator.reservations_for_test().await, (0, 0, 0)); + let replay = bound + .coordinator + .submit(pushes[1].3.take().ok_or("retained owner ready")?) + .await?; + finished(timeout(Duration::from_secs(10), replay.wait()).await?)?; + assert!(bound.coordinator.close_and_drain().await.is_empty()); + for (prepared, root, budget, ready) in pushes { + drop(ready); + drop(prepared); + cleaned(root.path(), &budget).await?; + } + bound.fixture.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/inputs/custody.rs b/crates/canopy-server/src/packs/publication/tests/inputs/custody.rs new file mode 100644 index 0000000..23f57b1 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/inputs/custody.rs @@ -0,0 +1,597 @@ +use super::super::{ + prepare::{cleaned, physical}, + publishing::{plan, update}, +}; +use super::*; +use crate::packs::{ + catalog::{CatalogFileLimits, CatalogFiles, CatalogIndexes, CatalogReader}, + metadata::{MetadataSegment, tests::limits}, + verification::{ + PhysicalPackWitness, PhysicalVerifier, + physical::tests::{independence::git_input, physical_limits, prepared_for_store}, + }, +}; +use crate::{ObjectId, ObjectKind}; +use cellule_ltx::DiskBudget; + +struct Recovered { + fixture: Fixture, + coordinator: StagingCoordinator, + ticket: StagingTicket, + store: Arc, + provider: Arc, + prior: NativeInputCertificate, + adopted: NativeInputCertificate, + native: NativePackDescriptor, +} +impl Recovered { + async fn new(format: ObjectFormat) -> Result { + Self::new_with_inventory(format, false).await + } + async fn new_with_inventory(format: ObjectFormat, alter_index: bool) -> Result { + let fixture = Fixture::new(format).await?; + let (old_coordinator, old_ticket) = active(&fixture, [184; 16]).await?; + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new(provider.clone(), fixture.repository)); + let captured = store.clone(); + let task = old_ticket.spawn(move |context| async move { + capture_real(context, captured) + .await + .map_err(StagingError::Input) + })?; + let mut prior = task.wait().await.map_err(|e| e.to_string())?; + let index = NativeInputIndex::new(store.clone(), format); + let mut cursor = index.cursor(prior.root()?, None)?; + let native = cursor.next().await?.ok_or("native input")?; + assert!(cursor.next().await?.is_none()); + if alter_index { + let mut different = native; + different.index.manifest_digest[0] ^= 1; + let provider = store.clone(); + let task = old_ticket.spawn(move |context| async move { + context + .seal_native_inputs(provider, [different]) + .await + .map_err(|e| StagingError::Input(Box::new(e))) + })?; + prior = task.wait().await.map_err(|e| e.to_string())?; + } + let checkpoint = old_ticket + .register_inputs(prior.clone(), identity()?) + .map_err(|(e, _)| e)?; + checkpoint.wait().await.map_err(|e| e.to_string())?; + old_ticket.stop(); + assert!(old_coordinator.close_and_drain().await.is_empty()); + let coordinator = + StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let ready = ReadyStaging::claim( + fixture.client(), + fixture.target.clone(), + LeaseRequest { + check: LeaseCheck { + token: prior.token()?, + actor: "owner".into(), + }, + lease_ms: DEFAULT_LEASE_MS, + }, + identity()?, + ) + .await?; + let ticket = coordinator.submit(ready).map_err(|(e, _)| e)?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait()).await?, + StagingState::Active(_) + )); + let old = prior.clone(); + let adopt_store = store.clone(); + let task = ticket.spawn(move |context| async move { + context + .adopt_native_inputs(adopt_store, &old) + .await + .map_err(|e| StagingError::Input(Box::new(e))) + })?; + let adopted = task.wait().await.map_err(|e| e.to_string())?; + + Ok(Self { + fixture, + coordinator, + ticket, + store, + provider, + prior, + adopted, + native, + }) + } + async fn register(&self) -> Result { + let observer = self + .ticket + .register_inputs(self.adopted.clone(), identity()?) + .map_err(|(e, _)| e)?; + observer.wait().await.map_err(|e| e.to_string())?; + assert!(matches!( + timeout(Duration::from_secs(10), self.ticket.wait()).await?, + StagingState::Active(_) + )); + Ok(()) + } + async fn base(&self) -> Result<(Arc, Arc)> { + self.ticket.seal()?; + assert!(matches!( + timeout(Duration::from_secs(10), self.ticket.wait_terminal()).await?, + StagingState::Bound(_) + )); + let indexes = Arc::new(CatalogIndexes::new(self.store.clone(), self.fixture.format)); + let files = Arc::new(CatalogFiles::new( + self.fixture.root.path(), + DiskBudget::new(64 << 20), + self.store.clone(), + self.fixture.format, + CatalogFileLimits::default(), + )?); + let base = Arc::new(self.ticket.open_base(indexes.clone(), files).await?); + Ok((base, indexes)) + } + async fn physical( + &self, + root: &std::path::Path, + budget: DiskBudget, + ) -> Result<(PhysicalPackWitness, Arc, ObjectId)> { + let mut verifier = PhysicalVerifier::download( + root, + budget, + &self.store, + self.native, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + let segment = verifier + .inspect_next_shard(self.native.object_count) + .await?; + let tip = segment + .headers_after(None)? + .into_iter() + .find(|h| h.object.kind == ObjectKind::Commit) + .ok_or("tip")? + .object + .oid; + Ok((verifier.finish().await?, segment, tip)) + } + async fn publish_background(&self, operation: [u8; 16], name: &str, blobs: usize) -> Result { + let (base, _, _) = + super::super::prepare::opened(&self.fixture, operation, self.store.clone()).await?; + let native = prepared_for_store( + self.fixture.format, + blobs, + base.context().operation, + self.provider.clone(), + self.store.clone(), + ) + .await?; + let tip = native + .fixture + .objects + .values() + .find(|(o, _)| o.kind == ObjectKind::Commit) + .ok_or("background tip")? + .0 + .oid; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let mut builder = + CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + let (witness, segments) = physical(&native, root.path(), budget.clone()).await?; + builder.begin_pack(witness)?; + for segment in segments { + builder.add_segment(segment).await?; + } + builder.finish_pack().await?; + let prepared = builder.finish().await?; + let proof = prepared + .ref_proof( + plan(vec![update(name, None, Some(tip))]), + root.path(), + budget.clone(), + limits(), + ) + .await?; + assert!(matches!( + self.fixture + .client() + .command::(&self.fixture.target, identity()?, proof) + .await? + .output, + PublicationReply::Published(_) + )); + drop((prepared, native)); + cleaned(root.path(), &budget).await?; + Ok(()) + } + async fn close(self) -> Result { + self.ticket.stop(); + assert!(self.coordinator.close_and_drain().await.is_empty()); + self.fixture.runtime.shutdown().await?; + Ok(()) + } +} +fn digest(proof: &NativeInputCertificate) -> Result<[u8; 32]> { + let mut e = BoundedEncoder::new(CERTIFICATE_BYTES)?; + proof.encode(&mut e)?; + Ok(*blake3::hash(&e.finish()).as_bytes()) +} + +#[tokio::test] +async fn retained_input_custody_reconciles_over_moving_nonempty_base_and_cold_clones() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let recovered = Recovered::new(format).await?; + recovered.register().await?; + mutate( + &recovered.fixture.handle, + format!( + "UPDATE catalog_leases SET expires_at_ms=0 WHERE admission_sequence={}", + recovered.prior.token()?.attempt + ), + ) + .await?; + recovered + .publish_background([185; 16], "refs/heads/base", 2) + .await?; + let (base, indexes) = recovered.base().await?; + assert_eq!(base.generation_fact().generation, 1); + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let mut builder = + CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + let (witness, segment, tip) = recovered.physical(root.path(), budget.clone()).await?; + builder.begin_retained_pack(witness).await?; + assert!(indexes.input_stats().loaded_nodes > 0); + builder.add_segment(segment).await?; + builder.finish_pack().await?; + let prepared = builder.finish().await?; + assert_eq!( + prepared + .certificate() + .await? + .data()? + .input_checkpoint_digest, + Some(digest(&recovered.adopted)?) + ); + recovered + .publish_background([186; 16], "refs/heads/advance", 3) + .await?; + let next = prepared.reconcile().await?; + assert_eq!(next.base().generation, 2); + let proof = next + .ref_proof( + plan(vec![update("refs/heads/main", None, Some(tip))]), + root.path(), + budget.clone(), + limits(), + ) + .await?; + assert_eq!( + proof.certificate.data()?.input_checkpoint_digest, + Some(digest(&recovered.adopted)?) + ); + assert!(proof.certificate.bytes()?.len() <= CERTIFICATE_BYTES as usize); + let mut worst = proof.certificate.data()?; + worst.actor = "z".repeat(64); + worst.completion_digest = Some([93; 32]); + assert!( + CatalogCertificate::seal(&worst, &[16; 32])?.bytes()?.len() + <= CERTIFICATE_BYTES as usize + ); + let mutation = identity()?; + let committed = recovered + .fixture + .client() + .command::(&recovered.fixture.target, mutation, proof.clone()) + .await?; + let replay = recovered + .fixture + .client() + .command::(&recovered.fixture.target, mutation, proof) + .await?; + assert_eq!(committed.receipt, replay.receipt); + assert!(matches!( + committed.output, + PublicationReply::Published(PublishedRefs { generation: 3, .. }) + )); + let catalog = next.catalog(); + drop((next, prepared)); + cleaned(root.path(), &budget).await?; + let cold_root = tempfile::TempDir::new()?; + let cold_budget = DiskBudget::new(64 << 20); + let reader = CatalogReader::open( + Arc::new(CatalogIndexes::new(recovered.store.clone(), format)), + catalog, + ) + .await?; + let files = CatalogFiles::new( + cold_root.path(), + cold_budget.clone(), + recovered.store.clone(), + format, + CatalogFileLimits::default(), + )?; + let selected = reader + .lookup(tip, &files, &files) + .await? + .ok_or("retained tip")?; + assert_eq!(selected.source.record.native(), recovered.native); + let backend = crate::git_http::GitHttpBackend::initialize( + cold_root.path().into(), + cold_budget, + "refs/heads/main", + format, + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + backend + .cache + .download_native(&recovered.store, selected.source.record.native()) + .await?; + backend + .cache + .store_refs(&std::collections::BTreeMap::from([( + "refs/heads/main".into(), + crate::RefExpectation { + oid: Some(tip), + version: 1, + }, + )])) + .await?; + let client = tempfile::TempDir::new()?; + let clone = client.path().join("clone"); + git_input( + client.path(), + &[ + "-c", + "protocol.file.allow=always", + "clone", + "--no-local", + backend.git_dir().to_str().ok_or("cache path")?, + clone.to_str().ok_or("clone path")?, + ], + &[], + ) + .await?; + git_input(&clone, &["fsck", "--full"], &[]).await?; + recovered.close().await?; + } + Ok(()) +} + +#[tokio::test] +async fn retained_input_custody_requires_current_successor_checkpoint_and_keeps_raw_namespace_guard() +-> Result { + let recovered = Recovered::new(ObjectFormat::Sha256).await?; + let (base, _) = recovered.base().await?; + for raw in [true, false] { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let mut builder = + CatalogPreparation::new(root.path(), budget.clone(), base.clone(), limits()).await?; + let (witness, segment, _) = recovered.physical(root.path(), budget.clone()).await?; + let result = if raw { + builder.begin_pack(witness) + } else { + builder.begin_retained_pack(witness).await + }; + assert!(result.is_err()); + assert!(builder.finish().await.is_err()); + drop(segment); + cleaned(root.path(), &budget).await?; + } + recovered.close().await?; + Ok(()) +} + +#[tokio::test] +async fn retained_input_custody_rejects_native_pair_missing_from_authenticated_inventory() -> Result +{ + let recovered = Recovered::new(ObjectFormat::Sha256).await?; + // A fresh checkpoint in the successor namespace cannot authorize the old + // physical pair simply because both belong to the same logical request. + let wrong = seal( + &recovered.fixture, + &recovered.ticket, + recovered.store.clone(), + 1, + ) + .await?; + let observer = recovered + .ticket + .register_inputs(wrong, identity()?) + .map_err(|(e, _)| e)?; + observer.wait().await.map_err(|e| e.to_string())?; + let (base, _) = recovered.base().await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let mut builder = CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + let (witness, segment, _) = recovered.physical(root.path(), budget.clone()).await?; + assert!(builder.begin_retained_pack(witness).await.is_err()); + assert!(builder.finish().await.is_err()); + drop(segment); + cleaned(root.path(), &budget).await?; + recovered.close().await?; + Ok(()) +} + +#[tokio::test] +async fn retained_input_custody_final_certificate_digest_is_checked_and_old_domain_rejected() +-> Result { + let recovered = Recovered::new(ObjectFormat::Sha256).await?; + recovered.register().await?; + let (base, _) = recovered.base().await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let mut builder = CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + let (witness, segment, tip) = recovered.physical(root.path(), budget.clone()).await?; + builder.begin_retained_pack(witness).await?; + builder.add_segment(segment).await?; + builder.finish_pack().await?; + let prepared = builder.finish().await?; + let proof = prepared + .ref_proof( + plan(vec![update("refs/heads/main", None, Some(tip))]), + root.path(), + budget.clone(), + limits(), + ) + .await?; + let mut data = proof.certificate.data()?; + data.input_checkpoint_digest = Some([92; 32]); + let wrong = CatalogCertificate::seal(&data, &[16; 32])?; + assert!( + matches!(recovered.fixture.client().command::(&recovered.fixture.target,identity()?,wrong.clone()).await,Err(InvocationError::Rejected(outcome)) if outcome.output==AttestationOutcome::Denied(PreparationDenial::Conflict)) + ); + let mut wrong_proof = proof.clone(); + wrong_proof.certificate = wrong; + assert!( + matches!(recovered.fixture.client().command::(&recovered.fixture.target,identity()?,wrong_proof).await,Err(InvocationError::Rejected(outcome)) if outcome.output==PublicationReply::Denied(PreparationDenial::Conflict)) + ); + let mut old = proof.certificate.clone(); + let domain = b"canopy.catalog-attestation.v4\0"; + let at = old + .0 + .body + .windows(domain.len()) + .position(|b| b == domain) + .ok_or("catalog domain")?; + old.0.body[at + domain.len() - 2] = b'3'; + assert!(old.bytes().is_err()); + let mut e = BoundedEncoder::new(CERTIFICATE_BYTES)?; + old.0.encode(&mut e)?; + let bytes = e.finish(); + let mut d = BoundedDecoder::new(&bytes, CERTIFICATE_BYTES)?; + assert!(CatalogCertificate::decode(&mut d).is_err()); + assert!(matches!( + recovered + .fixture + .client() + .command::(&recovered.fixture.target, identity()?, proof) + .await? + .output, + PublicationReply::Published(_) + )); + drop(prepared); + cleaned(root.path(), &budget).await?; + recovered.close().await?; + Ok(()) +} + +#[tokio::test] +async fn retained_input_custody_matches_exact_index_incarnation_not_just_pack_key() -> Result { + let recovered = Recovered::new_with_inventory(ObjectFormat::Sha256, true).await?; + recovered.register().await?; + let (base, indexes) = recovered.base().await?; + let key = crate::packs::directory::SegmentKey { + operation: recovered.native.operation, + digest: recovered.native.pack.digest, + }; + let selected = indexes + .inputs() + .find(recovered.adopted.root()?, key) + .await? + .ok_or("indexed pair")?; + assert_eq!(selected.pack, recovered.native.pack); + assert_ne!( + selected.index.manifest_digest, + recovered.native.index.manifest_digest + ); + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let mut builder = CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + let (witness, segment, _) = recovered.physical(root.path(), budget.clone()).await?; + assert!(builder.begin_retained_pack(witness).await.is_err()); + assert!(builder.finish().await.is_err()); + drop(segment); + cleaned(root.path(), &budget).await?; + recovered.close().await?; + Ok(()) +} + +#[tokio::test] +async fn retained_input_custody_issuer_and_final_command_recheck_revocation_expiry_and_claim() +-> Result { + for loss in [0, 1, 2] { + let recovered = Recovered::new(ObjectFormat::Sha256).await?; + recovered.register().await?; + let (base, _) = recovered.base().await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let mut builder = + CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + let (witness, segment, tip) = recovered.physical(root.path(), budget.clone()).await?; + builder.begin_retained_pack(witness).await?; + builder.add_segment(segment).await?; + builder.finish_pack().await?; + let prepared = builder.finish().await?; + let proof = prepared + .ref_proof( + plan(vec![update("refs/heads/main", None, Some(tip))]), + root.path(), + budget.clone(), + limits(), + ) + .await?; + let denial = match loss { + 0 => { + mutate( + &recovered.fixture.handle, + "UPDATE repository_identity SET owner='other' WHERE singleton=1".into(), + ) + .await?; + PreparationDenial::Unauthorized + } + 1 => { + mutate(&recovered.fixture.handle,format!("UPDATE catalog_operations SET expires_at_ms=0; UPDATE catalog_leases SET expires_at_ms=0 WHERE admission_sequence={}",prepared.token().attempt)).await?; + PreparationDenial::Expired + } + _ => { + recovered + .fixture + .client() + .command::( + &recovered.fixture.target, + identity()?, + LeaseRequest { + check: LeaseCheck { + token: prepared.token(), + actor: "owner".into(), + }, + lease_ms: DEFAULT_LEASE_MS, + }, + ) + .await?; + PreparationDenial::Stale + } + }; + assert!(prepared.certificate().await.is_err()); + assert!( + matches!(recovered.fixture.client().command::(&recovered.fixture.target,identity()?,proof).await,Err(InvocationError::Rejected(outcome)) if outcome.output==PublicationReply::Denied(denial)) + ); + let state = recovered + .fixture + .handle + .query(0, 32, |c| { + let generation: u64 = c.query_row( + "SELECT generation FROM catalog_state WHERE singleton=1", + [], + |r| r.get(0), + )?; + let refs: u64 = c.query_row("SELECT count(*) FROM refs", [], |r| r.get(0))?; + Ok([generation.to_be_bytes(), refs.to_be_bytes()].concat()) + }) + .await?; + assert_eq!(state, vec![0; 16]); + drop(prepared); + cleaned(root.path(), &budget).await?; + recovered.close().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/inputs/requests.rs b/crates/canopy-server/src/packs/publication/tests/inputs/requests.rs new file mode 100644 index 0000000..ae37e7d --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/inputs/requests.rs @@ -0,0 +1,620 @@ +use super::*; +use crate::{git_gateway::preflight::EncodedPush, git_http::GitHttpRequest, git_input::GitInput}; +use cellule_ltx::DiskBudget; +use std::io::Write; +mod results; + +struct Request { + fixture: Fixture, + coordinator: StagingCoordinator, + ticket: StagingTicket, + store: Arc, + provider: Arc, + directory: tempfile::TempDir, + disk: DiskBudget, + raw: Vec, + encoded_size: u64, + proof: NativeInputCertificate, +} +fn raw_request(format: ObjectFormat, large: bool) -> Vec { + let mut bytes = Vec::new(); + super::super::completion::packet(&mut bytes, b"push-cert\0report-status push-options\n"); + for line in [ + "certificate version 0.1\n".into(), + "push-option canopy.note=retained\n".into(), + "\n".into(), + format!( + "{} {} refs/heads/main\n", + "00".repeat(format.bytes()), + "12".repeat(format.bytes()) + ), + "-----BEGIN SSH SIGNATURE-----\n".into(), + "unverified-request-fixture\n".into(), + "-----END SSH SIGNATURE-----\n".into(), + "push-cert-end\n".into(), + ] { + super::super::completion::packet(&mut bytes, line.as_bytes()); + } + bytes.extend(b"0000"); + super::super::completion::packet(&mut bytes, b"canopy.note=retained"); + bytes.extend(b"0000PACKunvalidated-request-fixture"); + if large { + bytes.resize(canopy_object_storage::external::PART_BYTES + 1024, b'x'); + } + bytes +} +impl Request { + async fn new( + format: ObjectFormat, + gzip: bool, + large: bool, + operation: [u8; 16], + ) -> Result { + Self::new_for_actor(format, gzip, large, operation, "owner").await + } + async fn new_for_actor( + format: ObjectFormat, + gzip: bool, + large: bool, + operation: [u8; 16], + actor: &str, + ) -> Result { + let fixture = Fixture::new(format).await?; + if actor != "owner" { + let actor = actor.to_owned(); + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([218; 32]), + sql::now(0)?, + 64, + 0, + move |tx| { + tx.execute("UPDATE repository_identity SET owner=?1", [actor])?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + } + let directory = tempfile::TempDir::new()?; + let disk = DiskBudget::new(32 << 20); + let raw = raw_request(format, large); + let wire = if gzip { + let mut encoder = + flate2::write::GzEncoder::new(Vec::new(), flate2::Compression::default()); + encoder.write_all(&raw)?; + encoder.finish()? + } else { + raw.clone() + }; + let encoded_size = wire.len() as u64; + let encoded = EncodedPush::new( + GitHttpRequest { + method: "POST".into(), + path_info: "/repo.git/git-receive-pack".into(), + query: String::new(), + content_type: Some("application/x-git-receive-pack-request".into()), + gzip, + protocol_v2: false, + authenticated: true, + body: GitInput::receive( + axum::body::Body::from(wire), + directory.path(), + &disk, + None, + None, + ) + .await?, + }, + &fixture.target, + fixture.repository, + format, + actor, + operation, + ) + .await?; + let coordinator = + StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let ticket = coordinator + .submit( + ReadyStaging::new( + fixture.client(), + fixture.target.clone(), + encoded.identity().clone(), + identity()?, + ) + .await?, + ) + .map_err(|(error, _)| error)?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait()).await?, + StagingState::Active(_) + )); + let provider = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new(provider.clone(), fixture.repository)); + let upload = store.clone(); + let missing_root = directory.path().to_owned(); + let missing_disk = disk.clone(); + let worker = ticket.spawn(move |context| async move { + assert!( + context + .reopen_push_request(&upload, &missing_root, &missing_disk, None, None) + .await + .is_err() + ); + let (_encoded, saved) = encoded + .retain(&context, &upload) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + context + .seal_push_inputs(upload, std::iter::empty(), saved) + .await + .map_err(|error| StagingError::Input(Box::new(error))) + })?; + let proof = worker.wait().await.map_err(|error| error.to_string())?; + assert!(proof.root()?.is_none()); + assert!(proof.wire_request()?.is_some()); + assert_eq!(disk.used(), 0); + ticket + .register_inputs(proof.clone(), identity()?) + .map_err(|(error, _)| error)? + .wait() + .await + .map_err(|error| error.to_string())?; + Ok(Self { + fixture, + coordinator, + ticket, + store, + provider, + directory, + disk, + raw, + encoded_size, + proof, + }) + } + async fn verify(&self) -> Result { + let store = self.store.clone(); + let root = self.directory.path().to_owned(); + let disk = self.disk.clone(); + let expected = self.raw.clone(); + let expected_digest = self.proof.token()?.request_digest; + let size = self.encoded_size; + self.ticket + .spawn(move |context| async move { + assert!( + context + .reopen_push_request(&store, &root, &disk, Some(size - 1), None) + .await + .is_err() + ); + let encoded = context + .reopen_push_request(&store, &root, &disk, Some(size), None) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + assert_eq!(encoded.identity().request_digest, expected_digest); + let preflight = encoded + .decode(&root, &disk, None) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + assert_eq!(preflight.identity().request_digest, expected_digest); + let request = preflight.into_native_request(); + assert!(!request.gzip); + assert_eq!( + request + .body + .prefix(expected.len()) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?, + expected + ); + Ok(()) + })? + .wait() + .await + .map_err(|error| error.to_string())?; + assert_eq!(self.disk.used(), 0); + Ok(()) + } +} + +#[tokio::test] +async fn request_checkpoint_recovers_large_plain_and_gzip_intent_and_appends_native_inventory() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for gzip in [false, true] { + let request = Request::new(format, gzip, true, [201; 16]).await?; + request.verify().await?; + let wire = request.proof.wire_request()?.unwrap(); + assert!(wire.artifact().size < 1024); + let record = wire.read(&request.store).await?; + assert_eq!(record.request.body.size, request.encoded_size); + let store = request.store.clone(); + let prior = request.proof.clone(); + let repository = request.fixture.repository; + let next = request + .ticket + .spawn(move |context| async move { + let token = context.token()?; + context + .append_native_inputs( + store, + &prior, + records(repository, token.artifact_operation, format, 2), + ) + .await + .map_err(|error| StagingError::Input(Box::new(error))) + })? + .wait() + .await + .map_err(|error| error.to_string())?; + assert_eq!(next.wire_request()?, Some(wire)); + assert_eq!(next.root()?.unwrap().record_count, 2); + let mutation = identity()?; + request + .ticket + .register_inputs(next.clone(), mutation) + .map_err(|(error, _)| error)? + .wait() + .await + .map_err(|error| error.to_string())?; + let replay = request + .fixture + .client() + .command::(&request.fixture.target, mutation, next.clone()) + .await?; + assert!(matches!(replay.output, StagingReply::Granted(_))); + denied( + request + .fixture + .client() + .command::( + &request.fixture.target, + identity()?, + request.proof.clone(), + ) + .await, + PreparationDenial::Conflict, + ); + let store = request.store.clone(); + let prior = next.clone(); + let same = request + .ticket + .spawn(move |context| async move { + let token = context.token()?; + context + .append_native_inputs( + store, + &prior, + records(repository, token.artifact_operation, format, 2), + ) + .await + .map_err(|error| StagingError::Input(Box::new(error))) + })? + .wait() + .await + .map_err(|error| error.to_string())?; + assert_eq!(same, next); + let token = next.token()?; + let revision=request.fixture.handle.query(0,32,move|connection|Ok(connection.query_row("SELECT input_checkpoint_revision FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2",rusqlite::params![token.owner.incarnation.as_bytes().as_slice(),token.attempt as i64],|row|row.get::<_,i64>(0))?.to_be_bytes().to_vec())).await?; + assert_eq!(revision, 1i64.to_be_bytes()); + request.ticket.seal()?; + let StagingState::Bound(_) = + timeout(Duration::from_secs(10), request.ticket.wait_terminal()).await? + else { + return Err("bound".into()); + }; + let encoded = request + .ticket + .bound_session()? + .reopen_push_request( + &request.store, + request.directory.path(), + &request.disk, + None, + None, + ) + .await?; + assert_eq!(encoded.identity().request_digest, token.request_digest); + drop(encoded); + assert_eq!(request.disk.used(), 0); + assert!(request.coordinator.close_and_drain().await.is_empty()); + request.fixture.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn request_checkpoint_owner_restore_adopts_original_bytes_after_source_pin_expiry() -> Result +{ + let request = Request::new(ObjectFormat::Sha256, true, false, [202; 16]).await?; + let native = results::completion("owner", ObjectFormat::Sha256, 65, false); + let expected_plan = native.plan.clone(); + let expected_response = native.response.clone(); + let expected_options = native.options.clone(); + let retained = results::retain(&request, native).await?; + request.ticket.stop(); + assert!(request.coordinator.close_and_drain().await.is_empty()); + request.fixture.handle.drain().await?; + request.fixture.runtime.shutdown().await?; + let session = SessionId::from_bytes([203; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, session)?; + let authority = CellAuthority::new(request.fixture.layout.clone()); + let idle = authority + .load(request.fixture.target.cell_id()) + .await? + .ok_or("idle")?; + let provision = CellCatalog::new( + request.fixture.layout.clone(), + request.fixture.target.tenant(), + ) + .lookup(request.fixture.target.cell_id()) + .await? + .ok_or("provision")?; + let directory = tempfile::TempDir::new()?; + let handle = runtime + .acquire_idle_restored( + provision, + request.fixture.replica.clone(), + authority, + idle, + directory.path().join("restored.sqlite"), + Owner { + session, + endpoint: "https://request-restored.invalid".into(), + }, + ) + .await?; + let client = CellClient::local(request.fixture.registry.clone(), handle.clone()); + let coordinator = + StagingCoordinator::new(request.fixture.target.clone(), StagingLimits::default())?; + let old = retained.token()?; + let ticket = coordinator + .submit( + ReadyStaging::claim( + client.clone(), + request.fixture.target.clone(), + LeaseRequest { + check: LeaseCheck { + token: old, + actor: "owner".into(), + }, + lease_ms: DEFAULT_LEASE_MS, + }, + identity()?, + ) + .await?, + ) + .map_err(|(error, _)| error)?; + let StagingState::Active(current) = timeout(Duration::from_secs(10), ticket.wait()).await? + else { + return Err("claim".into()); + }; + assert_ne!(current.token.owner, old.owner); + let store = request.store.clone(); + let prior = retained.clone(); + let proof = ticket + .spawn(move |context| async move { + context + .adopt_native_inputs(store, &prior) + .await + .map_err(|error| StagingError::Input(Box::new(error))) + })? + .wait() + .await + .map_err(|error| error.to_string())?; + assert_eq!(proof.wire_request()?, request.proof.wire_request()?); + assert_eq!(proof.native_result()?, retained.native_result()?); + assert_eq!(proof.root()?, retained.root()?); + ticket + .register_inputs(proof, identity()?) + .map_err(|(error, _)| error)? + .wait() + .await + .map_err(|error| error.to_string())?; + handle.execute(identity()?,Digest::from_bytes([204;32]),sql::now(0)?,64,0,move|tx|{tx.execute("UPDATE catalog_leases SET expires_at_ms=0 WHERE incarnation=?1 AND admission_sequence=?2",rusqlite::params![old.owner.incarnation.as_bytes().as_slice(),old.attempt as i64])?;Ok(cellule_runtime::cell::executor::HandlerOutcome::Success(Vec::new()))}).await?; + assert!( + check(&client, &request.fixture.target, old) + .await? + .is_none() + ); + drop(request.directory); + let disk = DiskBudget::new(1 << 20); + let restored_path = directory.path().to_owned(); + let work_disk = disk.clone(); + let store = request.store.clone(); + let expected = request.raw; + ticket + .spawn(move |context| async move { + let encoded = context + .reopen_push_request(&store, &restored_path, &work_disk, None, None) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + assert_eq!(encoded.identity().request_digest, old.request_digest); + let native = encoded + .decode(&restored_path, &work_disk, None) + .await + .map_err(|error| StagingError::Input(Box::new(error)))? + .into_native_request(); + assert_eq!( + native + .body + .prefix(expected.len()) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?, + expected + ); + drop(native); + let recovered = context + .reopen_native_result(&store, &restored_path, &work_disk, None) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + assert_eq!(recovered.plan, expected_plan); + assert_eq!(recovered.response, expected_response); + assert_eq!(recovered.options, expected_options); + assert!(recovered.certificate.is_none()); + Ok(()) + })? + .wait() + .await + .map_err(|error| error.to_string())?; + assert_eq!(disk.used(), 0); + assert!(coordinator.close_and_drain().await.is_empty()); + runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn request_checkpoint_rejects_late_part_corruption_and_revoked_custody() -> Result { + use object_store::ObjectStoreExt; + let request = Request::new(ObjectFormat::Sha256, false, true, [205; 16]).await?; + let record = request + .proof + .wire_request()? + .unwrap() + .read(&request.store) + .await?; + let path = request + .store + .path(record.body_key(), record.request.body.digest)?; + assert!(record.request.body.size > canopy_object_storage::external::PART_BYTES as u64); + let part = canopy_object_storage::external::part(&path, 1); + request + .provider + .put(&part, bytes::Bytes::from(vec![b'y'; 1024]).into()) + .await?; + let store = request.store.clone(); + let root = request.directory.path().to_owned(); + let disk = request.disk.clone(); + request + .ticket + .spawn(move |context| async move { + // First authenticated part is spooled before the corrupt second part. + assert!( + context + .reopen_push_request(&store, &root, &disk, None, None) + .await + .is_err() + ); + assert_eq!(disk.used(), 0); + Ok(()) + })? + .wait() + .await + .map_err(|error| error.to_string())?; + request + .provider + .put(&part, bytes::Bytes::from(vec![b'x'; 1024]).into()) + .await?; + request.verify().await?; + mutate( + &request.fixture.handle, + "UPDATE repository_identity SET owner='other' WHERE singleton=1".into(), + ) + .await?; + let store = request.store.clone(); + let root = request.directory.path().to_owned(); + let disk = request.disk.clone(); + request + .ticket + .spawn(move |context| async move { + assert!( + context + .reopen_push_request(&store, &root, &disk, None, None) + .await + .is_err() + ); + assert_eq!(disk.used(), 0); + Ok(()) + })? + .wait() + .await + .map_err(|error| error.to_string())?; + assert!(request.coordinator.close_and_drain().await.is_empty()); + request.fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn request_checkpoint_append_recovers_exact_uncertain_registration_and_old_observers() +-> Result { + for fault in [1, 2, 3] { + let request = Request::new(ObjectFormat::Sha1, false, false, [206; 16]).await?; + let previous = request.ticket.pending_inputs().ok_or("previous observer")?; + let original = previous.wait().await.map_err(|error| error.to_string())?; + let store = request.store.clone(); + let prior = request.proof.clone(); + let repository = request.fixture.repository; + let next = request + .ticket + .spawn(move |context| async move { + let token = context.token()?; + context + .append_native_inputs( + store, + &prior, + records(repository, token.artifact_operation, ObjectFormat::Sha1, 1), + ) + .await + .map_err(|error| StagingError::Input(Box::new(error))) + })? + .wait() + .await + .map_err(|error| error.to_string())?; + let mutation = identity()?; + request.coordinator.fault_for_test(fault); + let observer = request + .ticket + .register_inputs(next.clone(), mutation) + .map_err(|(error, _)| error)?; + drop(observer); + assert!(matches!( + timeout(Duration::from_secs(10), request.ticket.wait_terminal()).await?, + StagingState::Uncertain(_) + )); + assert!( + request + .ticket + .register_inputs(next.clone(), identity()?) + .is_err() + ); + assert_eq!(request.coordinator.stats().command_bytes, 12 << 10); + request.coordinator.recover(&request.ticket)?; + let registered = request + .ticket + .pending_inputs() + .ok_or("append observer")? + .wait() + .await + .map_err(|error| error.to_string())?; + let replay = request + .fixture + .client() + .command::(&request.fixture.target, mutation, next.clone()) + .await?; + assert_eq!(registered, replay.receipt); + assert_eq!( + previous.wait().await.map_err(|error| error.to_string())?, + original + ); + assert!(registered.commit_sequence > original.commit_sequence); + assert_eq!( + check( + &request.fixture.client(), + &request.fixture.target, + next.token()? + ) + .await?, + Some(next) + ); + request.verify().await?; + assert!(request.coordinator.close_and_drain().await.is_empty()); + request.fixture.runtime.shutdown().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/inputs/requests/results.rs b/crates/canopy-server/src/packs/publication/tests/inputs/requests/results.rs new file mode 100644 index 0000000..8bddb41 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/inputs/requests/results.rs @@ -0,0 +1,263 @@ +//! Synthetic completions qualify retention/recovery; native CGI validity is +//! covered independently by the receive/publication/cold-clone composition. +use super::*; +use crate::git_http::GitHttpResponse; +use canopy_object_storage::artifact::{ArtifactKey, ArtifactKind}; +use object_store::ObjectStoreExt; +mod signed; + +pub(super) fn completion( + actor: &str, + format: ObjectFormat, + count: u32, + progress: bool, +) -> PushCompletionRequest { + let oid = crate::ObjectId::try_from(&[23; 32][..format.bytes()]).unwrap(); + let plan = crate::PushPlan { + actor: actor.into(), + updates: (0..count) + .map(|n| crate::RefUpdate { + name: format!("refs/heads/{n:06}{}", "x".repeat(200)), + expected: (n % 3 != 0).then_some(crate::RefExpectation { + oid: (n % 3 == 1).then_some(oid), + version: n as i64 + 1, + }), + new_oid: Some(oid), + }) + .collect(), + }; + let mut report = Vec::new(); + super::super::super::completion::packet(&mut report, b"unpack ok\n"); + for update in &plan.updates { + super::super::super::completion::packet( + &mut report, + format!("ok {}\n", update.name).as_bytes(), + ); + } + report.extend(b"0000"); + let body = if progress { + let mut body = Vec::new(); + while body.len() <= canopy_object_storage::external::PART_BYTES { + super::super::super::completion::packet( + &mut body, + &[&[2][..], &vec![b'p'; 60_000]].concat(), + ); + } + for chunk in report.chunks(60_000) { + super::super::super::completion::packet(&mut body, &[&[1][..], chunk].concat()); + } + body.extend(b"0000"); + body + } else { + report + }; + PushCompletionRequest { + plan: Some(plan), + response: GitHttpResponse { + status: 200, + headers: vec![ + ( + "Content-Type".into(), + "application/x-git-receive-pack-result".into(), + ), + ("Content-Length".into(), body.len().to_string()), + ("X-Native-Test".into(), "preserved".into()), + ], + body, + }, + options: vec!["canopy.note=result".into()], + certificate: None, + } +} +pub(super) async fn retain( + request: &Request, + native: PushCompletionRequest, +) -> Result { + let store = request.store.clone(); + let prior = request.proof.clone(); + let directory = request.directory.path().to_owned(); + let disk = request.disk.clone(); + let repository = request.fixture.repository; + let format = request.fixture.format; + let proof = request + .ticket + .spawn(move |context| async move { + let result = context + .retain_native_result(&store, &prior, native, &directory, &disk) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + let token = context.token()?; + let proof = context + .append_native_result( + store.clone(), + &prior, + records(repository, token.artifact_operation, format, 1), + result, + ) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + // Artifacts and an issued MAC are insufficient until the exact + // checkpoint is committed to this lease. + assert!( + context + .reopen_native_result(&store, &directory, &disk, None) + .await + .is_err() + ); + Ok(proof) + })? + .wait() + .await + .map_err(|error| error.to_string())?; + assert_eq!(request.disk.used(), 0); + request + .ticket + .register_inputs(proof.clone(), identity()?) + .map_err(|(error, _)| error)? + .wait() + .await + .map_err(|error| error.to_string())?; + Ok(proof) +} +#[tokio::test] +async fn native_result_checkpoint_preserves_large_framed_plan_response_and_final_inventory() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let request = Request::new(format, true, false, [208; 16]).await?; + let native = completion("owner", format, 20_000, true); + let plan = native.plan.clone(); + let expected = native.response.clone(); + let options = native.options.clone(); + let mut inline = BoundedEncoder::new(4 << 20)?; + assert!(plan.as_ref().unwrap().encode(&mut inline).is_err()); + let proof = retain(&request, native).await?; + assert_eq!(proof.wire_request()?, request.proof.wire_request()?); + assert!(proof.native_result()?.is_some()); + let mut e = BoundedEncoder::new(CERTIFICATE_BYTES)?; + proof.encode(&mut e)?; + assert!(e.finish().len() < 1024); + let store = request.store.clone(); + let directory = request.directory.path().to_owned(); + let disk = request.disk.clone(); + let prior = proof.clone(); + let repository = request.fixture.repository; + let recovered = request + .ticket + .spawn(move |context| async move { + let token = context.token()?; + assert!( + context + .append_native_inputs( + store.clone(), + &prior, + records(repository, token.artifact_operation, format, 2) + ) + .await + .is_err() + ); + context + .reopen_native_result(&store, &directory, &disk, None) + .await + .map_err(|error| StagingError::Input(Box::new(error))) + })? + .wait() + .await + .map_err(|error| error.to_string())?; + assert_eq!(recovered.plan, plan); + assert_eq!(recovered.response, expected); + assert_eq!(recovered.options, options); + assert!(recovered.certificate.is_none()); + drop(recovered); + assert_eq!(request.disk.used(), 0); + request.ticket.seal()?; + assert!(matches!( + timeout(Duration::from_secs(10), request.ticket.wait_terminal()).await?, + StagingState::Bound(_) + )); + let recovered = request + .ticket + .bound_session()? + .reopen_native_result( + &request.store, + request.directory.path(), + &request.disk, + None, + ) + .await?; + assert_eq!(recovered.plan, plan); + assert_eq!(recovered.response, expected); + drop(recovered); + assert_eq!(request.disk.used(), 0); + assert!(request.coordinator.close_and_drain().await.is_empty()); + request.fixture.runtime.shutdown().await?; + } + Ok(()) +} +#[tokio::test] +async fn native_result_checkpoint_requires_registered_custody_and_rejects_corrupt_bodies() -> Result +{ + let request = Request::new(ObjectFormat::Sha256, false, false, [209; 16]).await?; + let native = completion("owner", ObjectFormat::Sha256, 1, true); + let digest = *blake3::hash(&native.response.body).as_bytes(); + let suffix = native.response.body[canopy_object_storage::external::PART_BYTES..].to_vec(); + let proof = retain(&request, native).await?; + let key = ArtifactKey { + operation: proof.token()?.artifact_operation, + binding_digest: digest, + kind: ArtifactKind::InputBody, + }; + let path = request.store.path(key, digest)?; + let part = canopy_object_storage::external::part(&path, 1); + request + .provider + .put(&part, bytes::Bytes::from(vec![b'!'; suffix.len()]).into()) + .await?; + let store = request.store.clone(); + let directory = request.directory.path().to_owned(); + let disk = request.disk.clone(); + request + .ticket + .spawn(move |context| async move { + assert!( + context + .reopen_native_result(&store, &directory, &disk, None) + .await + .is_err() + ); + assert_eq!(disk.used(), 0); + Ok(()) + })? + .wait() + .await + .map_err(|error| error.to_string())?; + request + .provider + .put(&part, bytes::Bytes::from(suffix).into()) + .await?; + mutate( + &request.fixture.handle, + "UPDATE repository_identity SET owner='other' WHERE singleton=1".into(), + ) + .await?; + let store = request.store.clone(); + let directory = request.directory.path().to_owned(); + let disk = request.disk.clone(); + request + .ticket + .spawn(move |context| async move { + assert!( + context + .reopen_native_result(&store, &directory, &disk, None) + .await + .is_err() + ); + assert_eq!(disk.used(), 0); + Ok(()) + })? + .wait() + .await + .map_err(|error| error.to_string())?; + assert!(request.coordinator.close_and_drain().await.is_empty()); + request.fixture.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/inputs/requests/results/signed.rs b/crates/canopy-server/src/packs/publication/tests/inputs/requests/results/signed.rs new file mode 100644 index 0000000..9cdae28 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/inputs/requests/results/signed.rs @@ -0,0 +1,237 @@ +use super::*; +use crate::{ + CanopyApplication, build_descriptor, + directory::{ + self, DirectoryCell, DirectoryModule, SshKey, SshKeyChange, TokenAuthority, TokenScope, + }, + push::VerifiedPushCertificate, +}; +use cellule_app::{ApplicationHandle, CellApplication}; + +struct Signers { + runtime: CellRuntime, + cell: DirectoryCell, + handle: CellHandle, + key: SshKey, + _files: tempfile::TempDir, +} +fn authority() -> TokenAuthority<'static> { + TokenAuthority { + actor_digest: [1; 32], + site_owner: "owner", + account: "owner", + } +} +impl Signers { + async fn new(target: &CellTarget) -> Result { + let app = Arc::new(CanopyApplication::compile(build_descriptor( + include_bytes!("../../../../../../../../../Cargo.lock"), + "native-result-signers", + ))?); + let target = directory::directory_target(target.tenant(), target.application())?; + let session = SessionId::from_bytes([219; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, session)?; + let layout = CellStorageLayout::new( + Store::new(Arc::new(InMemory::new())), + object_store::path::Path::from("native-result-signers"), + *target.application().as_bytes(), + ); + let provision = CellCatalog::new(layout.clone(), target.tenant()) + .provision(CatalogEntry::new( + &target, + CatalogRole::Sql, + app.registry() + .module_code(DirectoryModule::NAME) + .ok_or("directory code")?, + 1, + )?) + .await?; + let incarnation = IncarnationId::from_bytes([220; 16]); + let control = CellAuthority::new(layout.clone()) + .create_initial( + &provision, + incarnation, + Owner { + session, + endpoint: "https://native-result-signers.invalid".into(), + }, + ) + .await?; + let replica = CellReplica::new( + layout.clone(), + *target.cell_id().as_bytes(), + *incarnation.as_bytes(), + Limits::default(), + )?; + let files = tempfile::TempDir::new()?; + let handle = runtime + .bootstrap( + provision, + replica, + CellAuthority::new(layout), + control, + files.path().join("directory.sqlite"), + |tx| { + tx.execute_batch(directory::SCHEMA)?; + Ok(()) + }, + ) + .await?; + let application = ApplicationHandle::::new( + CellClient::local(app.registry(), handle.clone()), + app, + target.tenant(), + target.application(), + )?; + let cell = DirectoryCell::new(&application, target)?; + cell.create_account(identity()?, "owner", [1; 32], TokenScope::Admin) + .await?; + let bytes = ed25519_dalek::SigningKey::from_bytes(&[7; 32]) + .verifying_key() + .to_bytes(); + let key = SshKey::parse( + &ssh_key::PublicKey::from(ssh_key::public::Ed25519PublicKey(bytes)).to_openssh()?, + )?; + assert_eq!( + cell.register_ssh_key(identity()?, authority(), [1; 16], &key, TokenScope::Write) + .await? + .output, + SshKeyChange::Applied + ); + Ok(Self { + runtime, + cell, + handle, + key, + _files: files, + }) + } +} + +#[tokio::test] +async fn native_result_checkpoint_rechecks_scoped_signer_authority() -> Result { + let request = Request::new(ObjectFormat::Sha256, false, false, [221; 16]).await?; + let signers = Signers::new(&request.fixture.target).await?; + let foreign = crate::repository_target( + TenantId::from_bytes([222; 16]), + request.fixture.target.application(), + request.fixture.repository, + )?; + let foreign_signers = Signers::new(&foreign).await?; + let mut native = completion("owner", ObjectFormat::Sha256, 1, false); + // Synthetic opaque witness tests custody, not cryptographic verification. + // Native signature verification has independent real-SSH integration tests. + let signed_body = vec![b's'; canopy_object_storage::external::PART_BYTES + 33]; + let fingerprint = signers.key.fingerprint().to_owned(); + native.certificate = Some(VerifiedPushCertificate { + target: request.fixture.target.clone(), + request_digest: request.proof.token()?.request_digest, + signer: "owner".into(), + key: fingerprint.clone(), + body: signed_body.clone(), + }); + retain(&request, native).await?; + request.ticket.seal()?; + assert!(matches!( + timeout(Duration::from_secs(10), request.ticket.wait_terminal()).await?, + StagingState::Bound(_) + )); + let session = request.ticket.bound_session()?; + assert!( + session + .reopen_native_result( + &request.store, + request.directory.path(), + &request.disk, + None + ) + .await + .is_err() + ); + assert!( + session + .reopen_native_result( + &request.store, + request.directory.path(), + &request.disk, + Some(&foreign_signers.cell) + ) + .await + .is_err() + ); + let recovered = session + .reopen_native_result( + &request.store, + request.directory.path(), + &request.disk, + Some(&signers.cell), + ) + .await?; + let certificate = recovered.certificate.ok_or("missing certificate")?; + assert_eq!(certificate.body, signed_body); + assert_eq!(certificate.key, fingerprint); + assert_eq!(certificate.signer, "owner"); + assert_eq!(certificate.target, request.fixture.target); + assert_eq!( + certificate.request_digest, + request.proof.token()?.request_digest + ); + mutate( + &signers.handle, + "UPDATE accounts SET enabled=0 WHERE name='owner'".into(), + ) + .await?; + assert!( + session + .reopen_native_result( + &request.store, + request.directory.path(), + &request.disk, + Some(&signers.cell) + ) + .await + .is_err() + ); + mutate( + &signers.handle, + "UPDATE accounts SET enabled=1 WHERE name='owner'; UPDATE ssh_keys SET scope='read'".into(), + ) + .await?; + assert!( + session + .reopen_native_result( + &request.store, + request.directory.path(), + &request.disk, + Some(&signers.cell) + ) + .await + .is_err() + ); + mutate(&signers.handle, "UPDATE ssh_keys SET scope='write'".into()).await?; + assert_eq!( + signers + .cell + .revoke_ssh_key(identity()?, authority(), [1; 16]) + .await? + .output, + SshKeyChange::Applied + ); + assert!( + session + .reopen_native_result( + &request.store, + request.directory.path(), + &request.disk, + Some(&signers.cell) + ) + .await + .is_err() + ); + assert_eq!(request.disk.used(), 0); + assert!(request.coordinator.close_and_drain().await.is_empty()); + request.fixture.runtime.shutdown().await?; + signers.runtime.shutdown().await?; + foreign_signers.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/namespaces.rs b/crates/canopy-server/src/packs/publication/tests/namespaces.rs new file mode 100644 index 0000000..6c759ed --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/namespaces.rs @@ -0,0 +1,206 @@ +use super::*; +use canopy_object_storage::artifact::{ArtifactKey, ArtifactKind, ArtifactStore}; +use object_store::{ObjectStore, ObjectStoreExt}; + +#[test] +fn namespace_watermark_and_pin_identity_cannot_be_reset_or_rebound() -> Result { + let mut connection = rusqlite::Connection::open_in_memory()?; + connection.execute_batch("PRAGMA foreign_keys=ON")?; + connection.execute_batch(SCHEMA)?; + connection.execute("INSERT INTO repository_identity(singleton,repository_id,object_format,owner,push_cert_seed,artifact_sequence) VALUES(1,?1,'sha256','owner',zeroblob(32),1)", [uuid::Uuid::new_v4().as_bytes().as_slice()])?; + for sql in [ + "UPDATE repository_identity SET artifact_sequence=0", + "UPDATE repository_identity SET artifact_sequence=1", + "UPDATE repository_identity SET artifact_sequence=3", + "DELETE FROM repository_identity", + "INSERT OR REPLACE INTO repository_identity(singleton,repository_id,object_format,owner,push_cert_seed,artifact_sequence) VALUES(1,zeroblob(16),'sha256','owner',zeroblob(32),0)", + ] { + assert!(connection.execute(sql, []).is_err()); + } + connection.execute("UPDATE repository_identity SET artifact_sequence=2", [])?; + // A valid alternate generation ensures the mutation fails because the + // pin binding is immutable, rather than because its target is absent. + connection.execute( + "INSERT INTO catalog_generations(generation,catalog,certificate) VALUES(1,x'01',zeroblob(32))", + [], + )?; + connection.execute("INSERT INTO catalog_leases(incarnation,admission_sequence,operation,owner_epoch,artifact_operation,generation,expires_at_ms) VALUES(zeroblob(16),1,zeroblob(16),x'0000000000000001',?1,0,100)", [artifact_number(1).as_slice()])?; + assert!(connection.execute("INSERT OR REPLACE INTO catalog_leases(incarnation,admission_sequence,operation,owner_epoch,artifact_operation,generation,expires_at_ms) VALUES(zeroblob(16),1,zeroblob(16),x'0000000000000001',?1,0,100)", [artifact_number(2).as_slice()]).is_err()); + assert!(connection.execute("INSERT OR REPLACE INTO catalog_leases(incarnation,admission_sequence,operation,owner_epoch,artifact_operation,generation,expires_at_ms) VALUES(zeroblob(16),2,zeroblob(16),x'0000000000000001',?1,0,100)", [artifact_number(1).as_slice()]).is_err()); + for sql in [ + "UPDATE catalog_leases SET incarnation=randomblob(16)", + "UPDATE catalog_leases SET admission_sequence=2", + "UPDATE catalog_leases SET operation=randomblob(16)", + "UPDATE catalog_leases SET owner_epoch=x'0000000000000002'", + "UPDATE catalog_leases SET artifact_operation=randomblob(16)", + "UPDATE catalog_leases SET generation=1", + ] { + assert!(connection.execute(sql, []).is_err()); + } + // Each deferred binding differs from the existing pin by one field. + for (logical, epoch, namespace) in [ + ([1u8; 16], 1u64, artifact_number(1)), + ([0u8; 16], 2, artifact_number(1)), + ([0u8; 16], 1, artifact_number(2)), + ] { + let tx = connection.transaction()?; + tx.execute("INSERT INTO catalog_operations(id,actor,request_digest,incarnation,owner_epoch,admission_sequence,artifact_operation,generation,expires_at_ms) VALUES(?1,'owner',zeroblob(32),zeroblob(16),?2,1,?3,0,100)", rusqlite::params![logical.as_slice(),epoch.to_be_bytes().as_slice(),namespace.as_slice()])?; + assert!(tx.commit().is_err()); + } + connection.execute( + "UPDATE catalog_leases SET attestation=x'01',attestation_digest=zeroblob(32)", + [], + )?; + for sql in [ + "UPDATE catalog_leases SET attestation=NULL,attestation_digest=NULL", + "UPDATE catalog_leases SET attestation=x'02'", + "UPDATE catalog_leases SET attestation_digest=randomblob(32)", + ] { + assert!(connection.execute(sql, []).is_err()); + } + connection.execute( + "UPDATE catalog_leases SET attestation=x'01',attestation_digest=zeroblob(32)", + [], + )?; + Ok(()) +} + +#[tokio::test] +async fn delayed_old_attempt_deletes_cannot_remove_recreated_logical_request_bytes() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let client = fixture.client(); + let input = fixture.begin([58; 16]); + let mutation = identity()?; + let first = client + .command::(&fixture.target, mutation, input.clone()) + .await?; + let old = lease(first.output.clone())?; + let provider: Arc = Arc::new(InMemory::new()); + let artifacts = ArtifactStore::new(Arc::clone(&provider), fixture.repository); + let body = b"identical immutable artifact bytes"; + let digest = *blake3::hash(body).as_bytes(); + let mut pending = Vec::new(); + for kind in [ + ArtifactKind::Pack, + ArtifactKind::Index, + ArtifactKind::Metadata, + ArtifactKind::DirectoryRun, + ArtifactKind::CatalogNode, + ] { + let key = ArtifactKey { + operation: old.token.artifact_operation, + binding_digest: digest, + kind, + }; + artifacts + .put(key, body.len() as u64, digest, &mut body.as_slice()) + .await?; + let path = artifacts.path(key, digest)?; + pending.push(canopy_object_storage::external::part(&path, 0)); + pending.push(path); + } + let claimed = lease( + client + .command::(&fixture.target, identity()?, request(old.token)) + .await? + .output, + )?; + assert_eq!(claimed.token.operation, old.token.operation); + assert_ne!( + claimed.token.artifact_operation, + old.token.artifact_operation + ); + client + .command::(&fixture.target, identity()?, check(claimed.token)) + .await?; + let next = lease( + client + .command::(&fixture.target, identity()?, input.clone()) + .await? + .output, + )?; + assert_eq!(next.token.operation, old.token.operation); + assert_eq!(next.token.artifact_operation, artifact_number(3)); + assert_eq!(fixture.counts().await?, (1, 3)); + let replay = client + .command::(&fixture.target, mutation, input) + .await?; + assert_eq!(replay.output, first.output); + assert_eq!(replay.receipt, first.receipt); + let mut successors = Vec::new(); + for kind in [ + ArtifactKind::Pack, + ArtifactKind::Index, + ArtifactKind::Metadata, + ArtifactKind::DirectoryRun, + ArtifactKind::CatalogNode, + ] { + let key = ArtifactKey { + operation: next.token.artifact_operation, + binding_digest: digest, + kind, + }; + let stored = artifacts + .put(key, body.len() as u64, digest, &mut body.as_slice()) + .await?; + successors.push((key, stored)); + } + // Fault injection only: the production collector still needs retained-root + // and reader-drain proofs before scheduling any delete. + for path in pending { + provider.delete(&path).await?; + } + for (key, stored) in successors { + let mut read = artifacts.read(key, stored).await?; + assert_eq!(read.next().await?.ok_or("part")?.as_ref(), body); + assert!(read.next().await?.is_none()); + } + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn exhausted_namespace_allocator_preserves_live_attempts_and_exact_replay() -> Result { + let fixture = Fixture::with_artifact_sequence(ObjectFormat::Sha1, i64::MAX - 1).await?; + let client = fixture.client(); + let input = fixture.begin([59; 16]); + let mutation = identity()?; + let first = client + .command::(&fixture.target, mutation, input.clone()) + .await?; + let old = lease(first.output.clone())?; + assert_eq!( + old.token.artifact_operation, + artifact_number(i64::MAX as u64) + ); + let duplicate = client + .command::(&fixture.target, identity()?, input.clone()) + .await?; + assert_eq!(lease(duplicate.output)?.token, old.token); + let replay = client + .command::(&fixture.target, mutation, input) + .await?; + assert_eq!(replay.output, first.output); + assert_eq!(replay.receipt, first.receipt); + assert!( + client + .command::(&fixture.target, identity()?, fixture.begin([60; 16])) + .await + .is_err() + ); + assert!( + client + .command::(&fixture.target, identity()?, request(old.token)) + .await + .is_err() + ); + assert_eq!(fixture.counts().await?, (1, 1)); + let active = client + .query::(&fixture.target, None, check(old.token)) + .await? + .output + .ok_or("live attempt")?; + assert_eq!(active.token, old.token); + fixture.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/native_capture.rs b/crates/canopy-server/src/packs/publication/tests/native_capture.rs new file mode 100644 index 0000000..e7ab470 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/native_capture.rs @@ -0,0 +1,654 @@ +use super::*; +use super::{ + completion::packet, + prepare::{cleaned, opened}, + publishing::{plan, update}, +}; +use crate::{ + git_http::{GitHttpBackend, GitHttpRequest, NativeCaptureError}, + git_input::GitInput, + native_resources::{NativeClass, NativeResources}, + packs::{ + catalog::{CatalogFileLimits, CatalogFiles, CatalogIndexes, CatalogReader}, + metadata::tests::{fixture as input_fixture, limits}, + verification::{ + PhysicalVerifier, + physical::tests::{independence::git_input, physical_limits}, + }, + }, +}; +use canopy_object_storage::artifact::ArtifactStore; +use cellule_ltx::DiskBudget; +use tokio::time::{Duration, timeout}; + +#[tokio::test] +async fn native_receive_stages_verifies_and_publishes_then_clones_after_cache_loss() -> Result { + native_receive(false).await +} +#[tokio::test] +async fn native_receive_prepares_bounded_immutable_root_completion_from_registered_custody() +-> Result { + native_receive(true).await +} +async fn native_receive(rooted: bool) -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + if rooted { + let (base, _, _) = opened(&fixture, [159; 16], store.clone()).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let empty = CatalogPreparation::new(root.path(), budget.clone(), base, limits()) + .await? + .finish() + .await?; + fixture + .client() + .command::( + &fixture.target, + identity()?, + empty.empty_ref_initialization().await?, + ) + .await?; + drop(empty); + cleaned(root.path(), &budget).await?; + } + let source = input_fixture(format, 4).await?; + let tip = source + .objects + .values() + .find(|(object, _)| object.kind == crate::ObjectKind::Commit) + .ok_or("tip")? + .0 + .oid; + let pack_path = std::fs::read_dir(source.root.path().join("objects/pack"))? + .find_map(|entry| { + entry + .ok() + .map(|e| e.path()) + .filter(|p| p.extension().is_some_and(|e| e == "pack")) + }) + .ok_or("pack")?; + let work_root = tempfile::TempDir::new()?; + let disk = DiskBudget::new(256 << 20); + let native = NativeResources::default(); + let backend = GitHttpBackend::initialize( + work_root.path().into(), + disk.clone(), + "refs/heads/main", + format, + native.scope(NativeClass::Foreground), + ) + .await?; + let cache_weak = Arc::downgrade(&backend.cache); + let mut body = Vec::new(); + packet( + &mut body, + format!( + "{} {} refs/heads/main\0report-status ofs-delta object-format={}\n", + "0".repeat(format.bytes() * 2), + hex::encode(tip), + format.as_str() + ) + .as_bytes(), + ); + let mut changes = vec![update("refs/heads/main", None, Some(tip))]; + if rooted { + for i in 1..257 { + let name = format!("refs/heads/topic/{i:06}"); + packet( + &mut body, + format!( + "{} {} {name}\n", + "0".repeat(format.bytes() * 2), + hex::encode(tip) + ) + .as_bytes(), + ); + changes.push(update(&name, None, Some(tip))); + } + } + let command_bytes = body.len() + 4; + body.extend_from_slice(b"0000"); + body.extend_from_slice(&std::fs::read(pack_path)?); + let encoded = crate::git_gateway::preflight::EncodedPush::new( + GitHttpRequest { + method: "POST".into(), + path_info: "/repo.git/git-receive-pack".into(), + query: String::new(), + content_type: Some("application/x-git-receive-pack-request".into()), + gzip: false, + protocol_v2: false, + body: GitInput::receive( + axum::body::Body::from(body), + work_root.path(), + &disk, + Some(4 << 20), + None, + ) + .await?, + authenticated: true, + }, + &fixture.target, + fixture.repository, + format, + "owner", + [160; 16], + ) + .await?; + let request_digest = encoded.identity().request_digest; + let coordinator = + StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let ready = ReadyStaging::new( + fixture.client(), + fixture.target.clone(), + encoded.identity().clone(), + identity()?, + ) + .await?; + let ticket = coordinator.submit(ready).map_err(|(error, _)| error)?; + let StagingState::Active(initial) = timeout(Duration::from_secs(10), ticket.wait()).await? + else { + return Err("staging not active".into()); + }; + assert_eq!(initial.token.request_digest, request_digest); + let upload_store = Arc::clone(&store); + let retained = ticket.spawn(move |context| async move { + let (encoded, saved) = encoded + .retain(&context, &upload_store) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + let request = context + .seal_push_inputs(upload_store, std::iter::empty(), saved) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + Ok((encoded, request)) + })?; + let (encoded, request_checkpoint) = retained + .wait() + .await + .map_err(|error| format!("native receive stage: {error:?}"))?; + ticket + .register_inputs(request_checkpoint.clone(), identity()?) + .map_err(|(error, _)| error)? + .wait() + .await + .map_err(|error| format!("native receive stage: {error:?}"))?; + assert!(request_checkpoint.root()?.is_none()); + assert!(request_checkpoint.wire_request()?.is_some()); + let producer = backend.clone(); + let upload_store = Arc::clone(&store); + let request_root = work_root.path().to_owned(); + let request_disk = disk.clone(); + let task = ticket.spawn(move |context| async move { + let preflight = encoded + .decode(&request_root, &request_disk, None) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + let response = producer + .run_native_receive(preflight.into_native_request()) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + let inputs = producer + .stage_native_packs(&context, &upload_store, physical_limits()) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + let expected = response.clone(); + let result = context + .retain_native_result( + &upload_store, + &request_checkpoint, + PushCompletionRequest { + plan: Some(plan(changes)), + response, + options: Vec::new(), + certificate: None, + }, + &request_root, + &request_disk, + ) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + let certificate = context + .append_native_result( + upload_store, + &request_checkpoint, + inputs.iter().copied(), + result, + ) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + Ok((inputs, certificate, expected)) + })?; + let (inputs, input_certificate, response) = task + .wait() + .await + .map_err(|error| format!("capture: {error:?}"))?; + assert_eq!(response.status, 200); + assert!( + response + .body + .windows(b"ok refs/heads/main".len()) + .any(|bytes| bytes == b"ok refs/heads/main") + ); + // Small native receives must remain pack/index pairs, not loose bodies. + assert_eq!( + std::fs::read_dir(backend.git_dir().join("objects"))?.count(), + 2 + ); + assert!(input_certificate.wire_request()?.is_some()); + assert!(input_certificate.native_result()?.is_some()); + assert_eq!(inputs.len(), 1); + assert_eq!(inputs[0].operation, initial.token.artifact_operation); + assert_ne!(inputs[0].pack.manifest_digest, [0; 32]); + assert_ne!(inputs[0].index.manifest_digest, [0; 32]); + let checkpoint = ticket + .register_inputs(input_certificate.clone(), identity()?) + .map_err(|(error, _)| error)?; + let checkpoint_receipt = checkpoint + .wait() + .await + .map_err(|error| format!("native receive stage: {error:?}"))?; + assert_eq!( + fixture + .client() + .query::( + &fixture.target, + Some(checkpoint_receipt), + LeaseCheck { + token: initial.token, + actor: "owner".into() + } + ) + .await? + .output, + Some(input_certificate) + ); + drop((backend, source)); + assert!(cache_weak.upgrade().is_none()); + cleaned(work_root.path(), &disk).await?; + let recover_store = store.clone(); + let recover_root = work_root.path().to_owned(); + let recover_disk = disk.clone(); + let recovered = ticket.spawn(move |context| async move { + let encoded = context + .reopen_push_request(&recover_store, &recover_root, &recover_disk, None, None) + .await + .map_err(|error| StagingError::Input(Box::new(error)))?; + assert_eq!(encoded.identity().request_digest, request_digest); + let request = encoded + .decode(&recover_root, &recover_disk, None) + .await + .map_err(|error| StagingError::Input(Box::new(error)))? + .into_native_request(); + if rooted { + assert!(command_bytes > 4096); + assert!(request.body.packet_prefix(4096).await.is_err()); + } + assert!( + request + .body + .packet_prefix(command_bytes) + .await + .map_err(|error| StagingError::Input(Box::new(error)))? + .windows(b"refs/heads/main".len()) + .any(|part| part == b"refs/heads/main") + ); + context + .reopen_native_result(&recover_store, &recover_root, &recover_disk, None) + .await + .map_err(|error| StagingError::Input(Box::new(error))) + })?; + let recovered = recovered + .wait() + .await + .map_err(|error| format!("native receive stage: {error:?}"))?; + assert_eq!(recovered.response, response); + drop(response); + cleaned(work_root.path(), &disk).await?; + let physical_root = Arc::new(tempfile::TempDir::new()?); + let physical_disk = DiskBudget::new(256 << 20); + let mut verifier = PhysicalVerifier::download( + physical_root.path(), + physical_disk.clone(), + &store, + inputs[0], + physical_limits(), + native.scope(NativeClass::Foreground), + ) + .await?; + let segment = verifier.inspect_next_shard(inputs[0].object_count).await?; + let witness = verifier.finish().await?; + ticket.seal()?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait_terminal()).await?, + StagingState::Bound(_) + )); + let indexes = Arc::new(CatalogIndexes::new(Arc::clone(&store), format)); + let files = Arc::new(CatalogFiles::new( + fixture.root.path(), + DiskBudget::new(64 << 20), + Arc::clone(&store), + format, + CatalogFileLimits::default(), + )?); + let base = Arc::new(ticket.open_base(indexes, files).await?); + if rooted { + let mut builder = CatalogPreparation::new( + physical_root.path(), + physical_disk.clone(), + base, + limits(), + ) + .await?; + builder.begin_retained_pack(witness).await?; + builder.add_segment(segment).await?; + builder.finish_pack().await?; + let prepared = builder.finish().await?; + super::root_completion::qualify( + &fixture, + &prepared, + &store, + recovered, + physical_root.path(), + physical_disk.clone(), + ) + .await + .map_err(|error| format!("root composition: {error:?}"))?; + drop(prepared); + cleaned(physical_root.path(), &physical_disk).await?; + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + continue; + } + let expected = recovered.response.clone(); + let producer_root = Arc::clone(&physical_root); + let producer_disk = physical_disk.clone(); + let publication_identity = identity()?; + let work = ticket.spawn_bound(move |_| async move { + let result = async { + let mut builder = CatalogPreparation::new( + producer_root.path(), + producer_disk.clone(), + base, + limits(), + ) + .await?; + builder.begin_pack(witness)?; + builder.add_segment(segment).await?; + builder.finish_pack().await?; + let prepared = Arc::new(builder.finish().await?); + let ready = Box::pin(prepared.ready_push( + publication_identity, + recovered, + producer_root.path(), + producer_disk.clone(), + limits(), + )) + .await?; + Ok::<_, Box>((prepared, ready)) + } + .await; + result.map_err(StagingError::Input) + })?; + let (prepared, ready) = work + .wait() + .await + .map_err(|error| format!("native receive stage: {error:?}"))?; + let publications = + PublicationCoordinator::new(fixture.target.clone(), PublicationLimits::default())?; + let observer = ticket.publish(&publications, ready)?; + let completed = + super::coordinator::finished(timeout(Duration::from_secs(10), observer.wait()).await?)?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait_terminal()).await?, + StagingState::Published(Ok(_)) + )); + assert!(matches!( + completed.output, + CatalogCompletionReply::Completed(CompletedCatalogPush { + rejected: false, + publication: Some(PublishedRefs { + generation: 1, + ref_generation: 1, + .. + }), + .. + }) + )); + assert_eq!(observer.response().await?, expected); + let sql = cellule_runtime::primitives::sql::SqlCell::::new( + fixture.client(), + fixture.target.clone(), + )?; + let observed = sql + .query( + Some(completed.receipt), + super::super::sql::statement(super::super::sql::CURRENT, vec![]), + ) + .await?; + let published = + super::super::sql::generation(&observed.output, fixture.repository, format)? + .catalog + .ok_or("published root")?; + assert_eq!(published, prepared.catalog()); + drop(prepared); + cleaned(physical_root.path(), &physical_disk).await?; + assert!(coordinator.close_and_drain().await.is_empty()); + assert!(publications.close_and_drain().await.is_empty()); + // A new admitted workspace reconstructs from the committed catalog's + // source, with no producer descriptors/cache or SQL Git object records. + let cold_root = tempfile::TempDir::new()?; + let cold_disk = DiskBudget::new(64 << 20); + let indexes = Arc::new(CatalogIndexes::new(Arc::clone(&store), format)); + let files = CatalogFiles::new( + cold_root.path(), + cold_disk.clone(), + Arc::clone(&store), + format, + CatalogFileLimits::default(), + )?; + let reader = CatalogReader::open(indexes, published).await?; + let selected = reader + .lookup(tip, &files, &files) + .await? + .ok_or("canonical tip")?; + let cold = GitHttpBackend::initialize( + cold_root.path().into(), + cold_disk.clone(), + "refs/heads/main", + format, + native.scope(NativeClass::Foreground), + ) + .await?; + cold.cache + .download_native(&store, selected.source.record.native()) + .await?; + cold.cache + .store_refs(&std::collections::BTreeMap::from([( + "refs/heads/main".into(), + crate::RefExpectation { + oid: Some(tip), + version: 1, + }, + )])) + .await?; + let client = tempfile::TempDir::new()?; + let clone_path = client.path().join("clone"); + git_input( + client.path(), + &[ + "-c", + "protocol.file.allow=always", + "clone", + "--no-local", + cold.git_dir().to_str().ok_or("cache path")?, + clone_path.to_str().ok_or("clone path")?, + ], + &[], + ) + .await?; + git_input(&clone_path, &["fsck", "--full"], &[]).await?; + let cloned = git_input(&clone_path, &["rev-parse", "HEAD"], &[]).await?; + assert_eq!(String::from_utf8(cloned)?.trim(), hex::encode(tip)); + fixture.handle.query(0,1024,|db| { + assert_eq!(db.query_row("SELECT count(*) FROM sqlite_schema WHERE name IN ('objects','object_edges','object_closure','git_packs')",[],|r|r.get::<_,u64>(0))?,0); + Ok(Vec::new()) + }).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn native_capture_rejects_scope_limits_mutation_and_active_native_workers() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let source = input_fixture(ObjectFormat::Sha1, 4).await?; + let root = tempfile::TempDir::new()?; + let disk = DiskBudget::new(16 << 20); + let native = NativeResources::default(); + let backend = GitHttpBackend::initialize( + root.path().into(), + disk.clone(), + "refs/heads/main", + ObjectFormat::Sha1, + native.scope(NativeClass::Foreground), + ) + .await?; + for entry in std::fs::read_dir(source.root.path().join("objects/pack"))? { + let entry = entry?; + std::fs::copy( + entry.path(), + backend + .git_dir() + .join("objects/pack") + .join(entry.file_name()), + )?; + } + backend.cache.reconcile().await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let coordinator = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let ready = ReadyStaging::new( + fixture.client(), + fixture.target.clone(), + fixture.begin([161; 16]), + identity()?, + ) + .await?; + let ticket = coordinator.submit(ready).map_err(|(error, _)| error)?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait()).await?, + StagingState::Active(_) + )); + let task = ticket.spawn(move |context| async move { + rejection_checks(context, backend, store) + .await + .map_err(StagingError::Input) + })?; + task.wait() + .await + .map_err(|error| format!("capture checks: {error:?}"))?; + ticket.seal()?; + assert!(matches!( + timeout(Duration::from_secs(10), ticket.wait_terminal()).await?, + StagingState::Bound(_) + )); + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + Ok(()) +} + +async fn rejection_checks( + context: StagingContext, + backend: GitHttpBackend, + store: Arc, +) -> std::result::Result<(), Box> { + let other = ArtifactStore::new(Arc::new(InMemory::new()), [99; 16]); + assert!(matches!( + backend + .stage_native_packs(&context, &other, physical_limits()) + .await, + Err(NativeCaptureError::Context) + )); + let small = crate::packs::verification::PhysicalLimits { + max_pack_bytes: 1, + ..physical_limits() + }; + assert!(matches!( + backend.stage_native_packs(&context, &store, small).await, + Err(NativeCaptureError::Limit) + )); + let path = std::fs::read_dir(backend.git_dir().join("objects/pack"))? + .find_map(|entry| { + entry + .ok() + .map(|e| e.path()) + .filter(|p| p.extension().is_some_and(|e| e == "pack")) + }) + .ok_or("pack")?; + let loose = backend.git_dir().join("objects/aa"); + std::fs::create_dir(&loose)?; + assert!(matches!( + backend + .stage_native_packs(&context, &store, physical_limits()) + .await, + Err(NativeCaptureError::Context) + )); + std::fs::remove_dir(&loose)?; + #[cfg(unix)] + { + use std::os::unix::fs::PermissionsExt; + std::fs::set_permissions(&path, std::fs::Permissions::from_mode(0o600))?; + } + #[cfg(not(unix))] + { + let mut permissions = std::fs::metadata(&path)?.permissions(); + permissions.set_readonly(false); + std::fs::set_permissions(&path, permissions)?; + } + let original = std::fs::read(&path)?; + let mut corrupted = original.clone(); + corrupted[12] ^= 1; + std::fs::write(&path, &corrupted)?; + assert!(matches!( + backend + .stage_native_packs(&context, &store, physical_limits()) + .await, + Err(NativeCaptureError::Binding(_)) + )); + std::fs::write(&path, original)?; + let mut command = crate::native_git::command(&backend.git_dir())?; + command + .args(["hash-object", "--stdin"]) + .stdin(std::process::Stdio::piped()) + .stdout(std::process::Stdio::null()) + .stderr(std::process::Stdio::null()); + let mut worker = crate::native_git::process::GitProcess::spawn( + command, + Arc::clone(&backend.cache), + backend + .cache + .native + .try_admit(crate::native_resources::NativeWork::Read)?, + )?; + let blocked = matches!(backend.stage_native_packs(&context,&store,physical_limits()).await,Err(NativeCaptureError::Io(error)) if error.kind()==std::io::ErrorKind::WouldBlock); + drop(worker.child.stdin.take()); + let status = timeout(Duration::from_secs(5), worker.wait()).await??; + drop(worker); + assert!(blocked && status.success()); + let first = backend + .stage_native_packs(&context, &store, physical_limits()) + .await?; + let replay = backend + .stage_native_packs(&context, &store, physical_limits()) + .await?; + assert_eq!(first, replay); + assert_eq!(first.len(), 1); + Ok::<_, Box>(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/prepare.rs b/crates/canopy-server/src/packs/publication/tests/prepare.rs new file mode 100644 index 0000000..97d16ca --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/prepare.rs @@ -0,0 +1,478 @@ +use super::*; +use crate::packs::{ + catalog::{CatalogFileLimits, CatalogFiles, CatalogIndexes, CatalogReader, CatalogSnapshot}, + closure::ClosureError, + directory::snapshot::DirectorySnapshot, + metadata::{MetadataSegment, tests::limits}, + verification::{ + PhysicalPackWitness, PhysicalVerifier, + physical::tests::{ + Prepared, + independence::{git_input, upload_pair_for_operation}, + physical_limits, prepared_for_context, prepared_for_store, + }, + }, +}; +use canopy_object_storage::artifact::ArtifactStore; +use cellule_ltx::DiskBudget; +use std::{path::Path, time::Duration}; +type NativeAttempt = ( + Prepared, + Arc, + Arc, + Arc, +); +pub(super) async fn opened_native( + fixture: &Fixture, + operation: [u8; 16], + blobs: usize, +) -> Result { + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new(provider.clone(), fixture.repository)); + let (base, files, indexes) = opened(fixture, operation, Arc::clone(&store)).await?; + let native = prepared_for_store( + fixture.format, + blobs, + base.context().operation, + provider, + store, + ) + .await?; + Ok((native, base, files, indexes)) +} + +pub(super) async fn opened( + fixture: &Fixture, + operation: [u8; 16], + store: Arc, +) -> Result<( + Arc, + Arc, + Arc, +)> { + let started = fixture + .client() + .command::(&fixture.target, identity()?, fixture.begin(operation)) + .await?; + let token = lease(started.output)?.token; + let indexes = Arc::new(CatalogIndexes::new(Arc::clone(&store), fixture.format)); + let files = Arc::new(CatalogFiles::new( + fixture.root.path(), + DiskBudget::new(64 << 20), + store, + fixture.format, + CatalogFileLimits::default(), + )?); + let base = Arc::new( + PreparationBaseResolver::open( + fixture.client(), + fixture.target.clone(), + check(token), + Arc::clone(&indexes), + Arc::clone(&files), + Some(started.receipt), + ) + .await?, + ); + Ok((base, files, indexes)) +} +pub(super) async fn physical( + prepared: &Prepared, + root: &Path, + budget: DiskBudget, +) -> Result<(PhysicalPackWitness, Vec>)> { + let mut physical = PhysicalVerifier::download( + root, + budget, + &prepared.store, + prepared.descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + let halfway = prepared.descriptor.object_count / 2; + let mut segments = Vec::new(); + for count in [1, halfway, prepared.descriptor.object_count - halfway - 1] { + segments.push(physical.inspect_next_shard(count).await?); + } + Ok((physical.finish().await?, segments)) +} +pub(super) async fn cleaned(root: &Path, budget: &DiskBudget) -> Result { + tokio::time::timeout(Duration::from_secs(5), async { + while budget.used() != 0 { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + assert_eq!(std::fs::read_dir(root)?.count(), 0); + Ok(()) +} + +#[tokio::test] +async fn complete_physical_partitions_build_exact_catalogs_and_reuse_the_certified_base() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let (prepared, base, files, indexes) = opened_native(&fixture, [2; 16], 1600).await?; + let token = base.context_token(); + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(256 << 20); + let run_limits = crate::packs::metadata::MetadataLimits { + max_file_bytes: 16 << 10, + cache_kib: 16, + }; + let mut assembler = CatalogPreparation::new_with_run_limits( + root.path(), + budget.clone(), + base, + limits(), + run_limits, + ) + .await?; + let mut expected_segments = Vec::new(); + // Repeated exact inputs must not add source leaves or duplicate objects. + for _ in 0..2 { + let (witness, segments) = physical(&prepared, root.path(), budget.clone()).await?; + assembler.begin_pack(witness)?; + expected_segments = segments.iter().map(|s| s.descriptor()).collect(); + for segment in segments { + assembler.add_segment(segment).await?; + } + assembler.finish_pack().await?; + } + let proof = assembler.finish().await?; + assert_eq!(proof.token(), token); + assert_eq!(proof.base().generation, 0); + assert_eq!(proof.object_count(), prepared.fixture.objects.len() as u64); + assert_eq!(proof.input_count(), 1); + assert_eq!( + proof.edge_count(), + prepared + .fixture + .objects + .values() + .map(|(_, edges)| edges + .iter() + .map(|e| (e.child, e.expected_kind)) + .collect::>() + .len() as u64) + .sum::() + ); + proof.ensure_live()?; + let stored = proof.catalog(); + let snapshot = CatalogSnapshot::download(&prepared.store, stored).await?; + let directory = DirectorySnapshot::download(&prepared.store, snapshot.directory).await?; + assert_eq!(directory.level_zero.len(), 1); + let run_root = directory.level_zero[0]; + assert!(run_root.record_count > crate::packs::directory::snapshot::LEVEL_ZERO_ROOTS as u64); + assert_eq!(run_root.object_count, proof.object_count()); + let mut run_cursor = indexes.ranges().cursor(Some(run_root), None)?; + let mut run_count = 0; + while let Some(run) = run_cursor.next().await? { + assert!(run.run.size <= run_limits.max_file_bytes); + run_count += 1; + } + assert_eq!(run_count, run_root.record_count); + let sources = indexes.sources(); + let mut cursor = sources.cursor(snapshot.sources, None)?; + let mut found = Vec::new(); + while let Some(record) = cursor.next().await? { + assert_eq!(record.native(), prepared.descriptor); + found.push(record.metadata.segment); + } + found.sort_by_key(|segment| segment.identity.first_ordinal); + assert_eq!(found, expected_segments); + let reader = CatalogReader::open(Arc::clone(&indexes), stored).await?; + for ids in prepared + .fixture + .objects + .keys() + .copied() + .collect::>() + .chunks(512) + { + let headers = reader.headers(ids, &*files, &*files).await?; + for (oid, header) in ids.iter().zip(headers) { + assert_eq!( + header.ok_or("header")?.object, + prepared.fixture.objects[oid].0 + ); + } + } + drop(proof); + cleaned(root.path(), &budget).await?; + // Fixture-only publication supplies a trusted base for the next stage; + // production certificate issuance/final publication is still separate. + fixture.install_catalog(1, stored).await?; + let (base, _, _) = opened(&fixture, [47; 16], Arc::clone(&prepared.store)).await?; + let next = CatalogPreparation::new(root.path(), budget.clone(), base, limits()) + .await? + .finish() + .await?; + assert_eq!(next.object_count(), 0); + assert_eq!(next.input_count(), 0); + assert_eq!(next.base().catalog, Some(stored)); + let unchanged = CatalogSnapshot::download(&prepared.store, next.catalog()).await?; + assert_eq!(unchanged.sources, snapshot.sources); + assert_eq!( + DirectorySnapshot::download(&prepared.store, unchanged.directory).await?, + DirectorySnapshot::download(&prepared.store, snapshot.directory).await? + ); + drop(next); + cleaned(root.path(), &budget).await?; + let commit = prepared + .fixture + .objects + .values() + .find(|(object, _)| object.kind == crate::ObjectKind::Commit) + .ok_or("commit")? + .0; + let pack = git_input( + prepared.fixture.root.path(), + &["pack-objects", "--stdout", "--no-reuse-delta"], + format!("{}\n", hex::encode(commit.oid)).as_bytes(), + ) + .await?; + let (base, files, indexes) = + opened(&fixture, [50; 16], Arc::clone(&prepared.store)).await?; + let descriptor = + upload_pair_for_operation(&prepared, commit.oid, &pack, base.context().operation) + .await?; + let mut physical = PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + let segment = physical.inspect_next_shard(1).await?; + let mut assembler = + CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + assembler.begin_pack(physical.finish().await?)?; + assembler.add_segment(segment).await?; + assembler.finish_pack().await?; + let extension = assembler.finish().await?; + assert_eq!(extension.base().catalog, Some(stored)); + assert_eq!(extension.object_count(), 1); + let extended = CatalogSnapshot::download(&prepared.store, extension.catalog()).await?; + let old_directory = + DirectorySnapshot::download(&prepared.store, snapshot.directory).await?; + let new_directory = + DirectorySnapshot::download(&prepared.store, extended.directory).await?; + assert_eq!( + &new_directory.level_zero[..old_directory.level_zero.len()], + old_directory.level_zero.as_slice() + ); + assert_eq!( + new_directory.level_zero.len(), + old_directory.level_zero.len() + 1 + ); + let reader = CatalogReader::open(indexes, extension.catalog()).await?; + assert_eq!( + reader.headers(&[commit.oid], &*files, &*files).await?[0] + .ok_or("commit header")? + .object, + commit + ); + drop(extension); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn incomplete_or_out_of_order_inputs_poison_catalog_preparation() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let (prepared, base, _, _) = opened_native(&fixture, [2; 16], 4).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(256 << 20); + for order in [false, true] { + let (witness, segments) = physical(&prepared, root.path(), budget.clone()).await?; + let mut assembler = + CatalogPreparation::new(root.path(), budget.clone(), Arc::clone(&base), limits()) + .await?; + assembler.begin_pack(witness)?; + if order { + assert!( + assembler + .add_segment(Arc::clone(&segments[1])) + .await + .is_err() + ); + } else { + assembler.add_segment(Arc::clone(&segments[0])).await?; + assert!(assembler.finish_pack().await.is_err()); + } + assert!(assembler.finish().await.is_err()); + drop(segments); + cleaned(root.path(), &budget).await?; + } + // Successful physical verification in a different operation grants no + // authority to inject that source into this catalog attempt. + let wrong = prepared_for_context(fixture.format, 4, fixture.repository, [3; 16]).await?; + let (witness, segments) = physical(&wrong, root.path(), budget.clone()).await?; + let mut assembler = + CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + assert!(assembler.begin_pack(witness).is_err()); + assert!(assembler.finish().await.is_err()); + drop(segments); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn physical_validity_cannot_publish_missing_graph_dependencies() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let (prepared, base, _, _) = opened_native(&fixture, [2; 16], 4).await?; + let commit = prepared + .fixture + .objects + .values() + .find(|(o, _)| o.kind == crate::ObjectKind::Commit) + .ok_or("commit")? + .0; + let pack = git_input( + prepared.fixture.root.path(), + &["pack-objects", "--stdout", "--no-reuse-delta"], + format!("{}\n", hex::encode(commit.oid)).as_bytes(), + ) + .await?; + let descriptor = + upload_pair_for_operation(&prepared, commit.oid, &pack, base.context().operation) + .await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(256 << 20); + let mut physical = PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + let segment = physical.inspect_next_shard(1).await?; + let mut assembler = + CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + assembler.begin_pack(physical.finish().await?)?; + assembler.add_segment(segment).await?; + assembler.finish_pack().await?; + assert!(matches!( + assembler.finish().await, + Err(CatalogPreparationError::Closure(ClosureError::Missing(_))) + )); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[test] +fn canceled_catalog_finish_keeps_the_private_workspace_until_queued_work_drains() -> Result { + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .max_blocking_threads(1) + .build()?; + runtime.block_on(async { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let (base, _, _) = opened(&fixture, [48; 16], store).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let assembler = + CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + let workspace = std::fs::read_dir(root.path())? + .next() + .ok_or("workspace")?? + .path(); + let (release_tx, release_rx) = std::sync::mpsc::channel(); + let started = Arc::new(tokio::sync::Notify::new()); + let notify = Arc::clone(&started); + let blocked = tokio::task::spawn_blocking(move || { + notify.notify_one(); + release_rx.recv().unwrap(); + }); + started.notified().await; + let mut finish = Box::pin(assembler.finish()); + assert!( + tokio::time::timeout(Duration::from_millis(25), &mut finish) + .await + .is_err() + ); + drop(finish); + assert!(workspace.exists()); + assert_eq!( + budget.used(), + crate::packs::metadata::growth::INITIAL_BYTES * 3 + ); + release_tx.send(())?; + blocked.await?; + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + Ok::<_, Box>(()) + }) +} + +#[tokio::test] +async fn catalog_admission_failure_leaves_no_private_workspace_or_proof() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let store = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let (base, _, _) = opened(&fixture, [49; 16], store).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(1); + assert!( + CatalogPreparation::new(root.path(), budget.clone(), base, limits()) + .await + .is_err() + ); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn a_different_artifact_backend_cannot_supply_publication_input_proofs() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let destination = Arc::new(ArtifactStore::new( + Arc::new(InMemory::new()), + fixture.repository, + )); + let (base, _, _) = opened(&fixture, [2; 16], destination).await?; + let source = prepared_for_context( + fixture.format, + 4, + fixture.repository, + base.context().operation, + ) + .await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(256 << 20); + let (witness, segments) = physical(&source, root.path(), budget.clone()).await?; + // A clone of the exact service capability retains the physical binding. + witness.verify_store(&source.store.clone())?; + let mut assembler = + CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + assert!(matches!( + assembler.begin_pack(witness), + Err(CatalogPreparationError::Physical(_)) + )); + assert!(assembler.finish().await.is_err()); + drop(segments); + cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/publishing.rs b/crates/canopy-server/src/packs/publication/tests/publishing.rs new file mode 100644 index 0000000..39599e5 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/publishing.rs @@ -0,0 +1,926 @@ +use super::prepare::{cleaned, opened, opened_native, physical}; +use super::*; +use crate::packs::{ + catalog::CatalogReader, + metadata::{MetadataError, tests::limits}, + verification::physical::tests::{Prepared, independence::git_input}, +}; +use crate::{ObjectId, ObjectKind, PushPlan, RefExpectation, RefUpdate}; +use canopy_object_storage::artifact::{ArtifactKey, ArtifactKind}; +use cellule_ltx::DiskBudget; + +pub(super) struct Graph { + pub(super) prepared: PreparedCatalog, + pub(super) root: tempfile::TempDir, + pub(super) budget: DiskBudget, + pub(super) initial: ObjectId, + pub(super) tip: ObjectId, + pub(super) other: ObjectId, + pub(super) blob: ObjectId, + pub(super) store: Arc, +} +pub(super) fn update(name: &str, old: Option<(ObjectId, i64)>, new: Option) -> RefUpdate { + RefUpdate { + name: name.into(), + expected: old.map(|(oid, version)| RefExpectation { + oid: Some(oid), + version, + }), + new_oid: new, + } +} +pub(super) fn plan(updates: Vec) -> PushPlan { + PushPlan { + actor: "owner".into(), + updates, + } +} +pub(super) async fn assembled( + fixture: &Fixture, + operation: [u8; 16], + depth: usize, +) -> Result { + let (mut native, base, _, _) = opened_native(fixture, operation, 4).await?; + let initial = native + .fixture + .objects + .values() + .find(|(object, _)| object.kind == ObjectKind::Commit) + .ok_or("commit")? + .0 + .oid; + let blob = native + .fixture + .objects + .values() + .find(|(object, _)| object.kind == ObjectKind::Blob) + .ok_or("blob")? + .0 + .oid; + let (tip, other) = if depth == 0 { + (initial, initial) + } else { + history(&mut native, initial, depth).await? + }; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(256 << 20); + let mut builder = CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + let (witness, segments) = physical(&native, root.path(), budget.clone()).await?; + builder.begin_pack(witness)?; + for segment in segments { + builder.add_segment(segment).await?; + } + builder.finish_pack().await?; + Ok(Graph { + prepared: builder.finish().await?, + root, + budget, + initial, + tip, + other, + blob, + store: native.store, + }) +} +async fn history( + native: &mut Prepared, + initial: ObjectId, + count: usize, +) -> Result<(ObjectId, ObjectId)> { + let mut input = Vec::new(); + for n in 1..=count { + let parent = if n == 1 { + hex::encode(initial) + } else { + format!(":{}", n - 1) + }; + input.extend_from_slice(format!("commit refs/heads/chain\nmark :{n}\ncommitter Test {} +0000\ndata 1\nx\nfrom {parent}\n\n",n+1).as_bytes()); + } + input.extend_from_slice( + b"commit refs/heads/other\ncommitter Test 1 +0000\ndata 1\ny\n\n", + ); + git_input( + native.fixture.root.path(), + &["fast-import", "--quiet"], + &input, + ) + .await?; + let parse = |bytes: Vec| -> Result { + Ok(ObjectId::try_from( + hex::decode(String::from_utf8(bytes)?.trim())?.as_slice(), + )?) + }; + let tip = parse( + git_input( + native.fixture.root.path(), + &["rev-parse", "refs/heads/chain"], + b"", + ) + .await?, + )?; + let other = parse( + git_input( + native.fixture.root.path(), + &["rev-parse", "refs/heads/other"], + b"", + ) + .await?, + )?; + git_input(native.fixture.root.path(), &["repack", "-ad"], b"").await?; + let path = std::fs::read_dir(native.fixture.root.path().join("objects/pack"))? + .find_map(|entry| { + entry + .ok() + .map(|entry| entry.path()) + .filter(|path| path.extension().is_some_and(|ext| ext == "idx")) + }) + .ok_or("index")?; + let index = crate::git_format::pack_index::PackIndex::open(&path, native.descriptor.format)?; + let pack = std::fs::read(path.with_extension("pack"))?; + let digest = *blake3::hash(&pack).as_bytes(); + let key = |kind| ArtifactKey { + operation: native.descriptor.operation, + binding_digest: digest, + kind, + }; + let pack = native + .store + .put( + key(ArtifactKind::Pack), + pack.len() as u64, + digest, + &mut &pack[..], + ) + .await?; + let index_bytes = std::fs::read(path)?; + let artifact_index = native + .store + .put( + key(ArtifactKind::Index), + index_bytes.len() as u64, + *blake3::hash(&index_bytes).as_bytes(), + &mut &index_bytes[..], + ) + .await?; + native.descriptor.pack = pack; + native.descriptor.index = artifact_index; + native.descriptor.git_checksum = index.pack_checksum(); + native.descriptor.object_count = index.len(); + Ok((tip, other)) +} +async fn proof(graph: &Graph, updates: Vec) -> Result { + Ok(Box::pin(graph.prepared.ref_proof( + plan(updates), + graph.root.path(), + graph.budget.clone(), + limits(), + )) + .await?) +} +pub(super) async fn state(handle: &CellHandle) -> Result> { + Ok(handle.query(0, 64 << 10, |connection| { + let mut refs = connection.prepare("SELECT name,oid,version FROM refs ORDER BY name")?; + let refs = refs.query_map([], |row| Ok((row.get::<_,String>(0)?,row.get::<_,Option>>(1)?,row.get::<_,i64>(2)?)))?.collect::>>()?; + let mut catalog = connection.prepare("SELECT generation,catalog,certificate,refs FROM catalog_generations ORDER BY generation")?; + let catalog = catalog.query_map([], |row| Ok((row.get::<_,u64>(0)?,row.get::<_,Option>>(1)?,row.get::<_,Option>>(2)?,row.get::<_,Option>>(3)?)))?.collect::>>()?; + let generations = connection.query_row("SELECT (SELECT generation FROM catalog_state),(SELECT generation FROM ref_generation)", [], |row| Ok((row.get::<_,u64>(0)?,row.get::<_,u64>(1)?)))?; + let pushes = connection.query_row("SELECT count(*) FROM pushes WHERE publication IS NOT NULL", [], |row|row.get::<_,u64>(0))?; + let mut checkpoints=connection.prepare("SELECT artifact_operation,attestation,attestation_digest FROM catalog_leases ORDER BY artifact_operation")?; + let checkpoints=checkpoints.query_map([],|row|Ok((row.get::<_,Vec>(0)?,row.get::<_,Option>>(1)?,row.get::<_,Option>>(2)?)))?.collect::>>()?; + let mut initial=connection.prepare("SELECT id,actor,request_digest,verification_digest,result FROM catalog_initialization")?; + let initial=initial.query_map([],|row|Ok((row.get::<_,Vec>(0)?,row.get::<_,String>(1)?,row.get::<_,Vec>(2)?,row.get::<_,Vec>(3)?,row.get::<_,Vec>(4)?)))?.collect::>>()?; + let policy=connection.query_row("SELECT (SELECT version FROM ref_policy_epoch),(SELECT watches FROM ref_policy_budget)",[],|row|Ok((row.get::<_,u64>(0)?,row.get::<_,u64>(1)?)))?; + let mut hash=blake3::Hasher::new();hash.update(b"fixture.ref-policy-state.v1\0"); + let mut guards=connection.prepare("SELECT id,scope,token,policy_epoch,total,next,valid FROM ref_policy_guards ORDER BY id")?; + let mut rows=guards.query([])?; + while let Some(row)=rows.next()? { + let record=(row.get::<_,Vec>(0)?,row.get::<_,Vec>(1)?,row.get::<_,Vec>(2)?,row.get::<_,u64>(3)?,row.get::<_,u64>(4)?,row.get::<_,u64>(5)?,row.get::<_,u8>(6)?); + hash.update(&serde_json::to_vec(&record).map_err(|_|Error::Command("fixture guard hash"))?); + } + hash.update(b"\0watches\0"); + let mut watches=connection.prepare("SELECT guard,oid,context,context_version,run_number FROM ref_policy_watches ORDER BY guard,oid,context,context_version,run_number")?; + let mut rows=watches.query([])?; + while let Some(row)=rows.next()? { + let record=(row.get::<_,Vec>(0)?,row.get::<_,Vec>(1)?,row.get::<_,String>(2)?,row.get::<_,u64>(3)?,row.get::<_,u64>(4)?); + hash.update(&serde_json::to_vec(&record).map_err(|_|Error::Command("fixture watch hash"))?); + } + let policy_hash=*hash.finalize().as_bytes(); + serde_json::to_vec(&(refs,catalog,generations,pushes,checkpoints,initial,policy,policy_hash)).map_err(|_| Error::Command("fixture publication state")) + }).await?) +} +fn published(reply: PublicationReply) -> Result { + match reply { + PublicationReply::Published(value) => Ok(value), + PublicationReply::Denied(reason) => Err(format!("denied {reason:?}").into()), + } +} +async fn reject( + fixture: &Fixture, + input: RefPublicationProof, + reason: PreparationDenial, +) -> Result { + let before = state(&fixture.handle).await?; + let result = Box::pin(fixture.client().command::( + &fixture.target, + identity()?, + input, + )) + .await; + assert!( + matches!(result,Err(InvocationError::Rejected(ref value)) if value.output==PublicationReply::Denied(reason)), + "{result:?}" + ); + assert_eq!(state(&fixture.handle).await?, before); + Ok(()) +} +pub(super) async fn edit(fixture: &Fixture, sql: &str) -> Result { + let sql = sql.to_owned(); + fixture + .handle + .execute( + identity()?, + Digest::from_bytes(*blake3::hash(sql.as_bytes()).as_bytes()), + sql::now(0)?, + sql.len(), + 0, + move |tx| { + tx.execute_batch(&sql)?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + Ok(()) +} +pub(super) async fn next_graph( + fixture: &Fixture, + old: &Graph, + operation: [u8; 16], +) -> Result { + let (base, _, _) = opened(fixture, operation, Arc::clone(&old.store)).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(256 << 20); + let prepared = CatalogPreparation::new(root.path(), budget.clone(), base, limits()) + .await? + .finish() + .await?; + Ok(Graph { + prepared, + root, + budget, + initial: old.initial, + tip: old.tip, + other: old.other, + blob: old.blob, + store: Arc::clone(&old.store), + }) +} + +#[tokio::test] +async fn catalog_refs_publish_atomically_for_both_formats_and_replay_exact_outcomes() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let graph = assembled(&fixture, [70; 16], 0).await?; + let before = state(&fixture.handle).await?; + let input = proof( + &graph, + vec![ + update("refs/heads/main", None, Some(graph.initial)), + update("refs/tags/blob", None, Some(graph.blob)), + ], + ) + .await?; + let changed_plan = proof( + &graph, + vec![update("refs/heads/other", None, Some(graph.initial))], + ) + .await?; + assert_eq!(state(&fixture.handle).await?, before); + let mut e = BoundedEncoder::new(4 << 20)?; + input.encode(&mut e)?; + let bytes = e.finish(); + let mut d = BoundedDecoder::new(&bytes, 4 << 20)?; + assert_eq!(RefPublicationProof::decode(&mut d)?, input); + d.finish()?; + let mutation = identity()?; + let first = Box::pin(fixture.client().command::( + &fixture.target, + mutation, + input.clone(), + )) + .await?; + let result = published(first.output)?; + assert_eq!((result.generation, result.ref_generation), (1, 1)); + assert_eq!( + result.certificate_digest, + *blake3::hash(&input.certificate.bytes()?).as_bytes() + ); + assert_eq!(fixture.counts().await?, (0, 1)); + let after = state(&fixture.handle).await?; + let replay = Box::pin(fixture.client().command::( + &fixture.target, + mutation, + input.clone(), + )) + .await?; + assert_eq!(replay.output, first.output); + assert_eq!(replay.receipt, first.receipt); + let logical = Box::pin(fixture.client().command::( + &fixture.target, + identity()?, + input, + )) + .await?; + assert_eq!(logical.output, first.output); + assert_eq!(state(&fixture.handle).await?, after); + reject(&fixture, changed_plan, PreparationDenial::Conflict).await?; + let (_, files, indexes) = opened(&fixture, [71; 16], Arc::clone(&graph.store)).await?; + let reader = CatalogReader::open(indexes, graph.prepared.catalog()).await?; + let headers = reader + .headers(&[graph.initial, graph.blob], &*files, &*files) + .await?; + assert!(headers.iter().all(Option::is_some)); + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn catalog_ref_membership_kind_and_tampered_bindings_cannot_publish() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let graph = assembled(&fixture, [72; 16], 0).await?; + for updates in [ + vec![update("refs/heads/main", None, Some(graph.blob))], + vec![update( + "refs/heads/main", + None, + Some(ObjectId::Sha1([99; 20])), + )], + vec![update("refs/canopy/forbidden", None, Some(graph.initial))], + vec![ + update("refs/heads/main", None, Some(graph.initial)), + update("refs/heads/main", None, Some(graph.initial)), + ], + ] { + assert!(matches!( + graph + .prepared + .ref_proof( + plan(updates), + graph.root.path(), + graph.budget.clone(), + limits() + ) + .await, + Err(RefProofError::Invalid) + )); + } + let input = proof( + &graph, + vec![update("refs/heads/main", None, Some(graph.initial))], + ) + .await?; + let mut edited = input.clone(); + edited.plan.updates[0].name = "refs/heads/edited".into(); + reject(&fixture, edited, PreparationDenial::Unauthorized).await?; + let mut edited = input.clone(); + edited.ancestry[0] = 0; + reject(&fixture, edited, PreparationDenial::Unauthorized).await?; + let mut edited = input.clone(); + // Change a signed digest byte while preserving both optional-field flags. + let digest_byte = edited.certificate.0.body.len() - 2; + edited.certificate.0.body[digest_byte] ^= 1; + reject(&fixture, edited, PreparationDenial::Unauthorized).await?; + let mut edited = input.clone(); + edited.certificate = graph.prepared.certificate().await?; + reject(&fixture, edited, PreparationDenial::Unauthorized).await?; + let mut invalid = input.clone(); + invalid.ancestry[0] |= 0x80; + let mut e = BoundedEncoder::new(4 << 20)?; + assert!(invalid.encode(&mut e).is_err()); + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn final_current_policies_use_certified_ancestry_and_current_check_versions() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let graph = assembled(&fixture, [73; 16], 1200).await?; + let initial = proof( + &graph, + vec![ + update("refs/heads/main", None, Some(graph.initial)), + update("refs/heads/other", None, Some(graph.other)), + ], + ) + .await?; + fixture + .client() + .command::(&fixture.target, identity()?, initial) + .await?; + edit(&fixture,"INSERT INTO branch_rules VALUES('refs/heads/main',1,1,1,1,0,0); INSERT INTO branch_rules VALUES('refs/heads/other',1,1,1,1,0,0);").await?; + let next = next_graph(&fixture, &graph, [74; 16]).await?; + let combined = proof( + &next, + vec![ + update("refs/heads/main", Some((graph.initial, 1)), Some(graph.tip)), + update("refs/heads/other", Some((graph.other, 1)), Some(graph.tip)), + ], + ) + .await?; + assert!(super::super::ref_proof::proven(&combined.ancestry, 0)); + assert!(!super::super::ref_proof::proven(&combined.ancestry, 1)); + reject(&fixture, combined, PreparationDenial::Conflict).await?; + let valid = proof( + &next, + vec![update( + "refs/heads/main", + Some((graph.initial, 1)), + Some(graph.tip), + )], + ) + .await?; + edit(&fixture,"INSERT INTO check_contexts VALUES('test','ci',1,1); INSERT INTO branch_required_checks VALUES('refs/heads/main','test');").await?; + reject(&fixture, valid.clone(), PreparationDenial::Conflict).await?; + let oid = hex::encode(graph.tip); + edit(&fixture,&format!("INSERT INTO check_runs(id,oid,context,context_version,reporter,state,version,summary,created_ms,updated_ms) VALUES(zeroblob(16),x'{oid}','test',1,'ci','success',1,'',0,0);")).await?; + // A current reporter/context change invalidates an earlier successful run. + edit( + &fixture, + "UPDATE check_contexts SET version=2,reporter='ci-new' WHERE name='test';", + ) + .await?; + reject(&fixture, valid.clone(), PreparationDenial::Conflict).await?; + edit(&fixture,&format!("INSERT INTO check_runs(id,oid,context,context_version,reporter,state,version,summary,created_ms,updated_ms) VALUES(x'01010101010101010101010101010101',x'{oid}','test',2,'ci-new','success',1,'',0,0);")).await?; + let result = published( + fixture + .client() + .command::(&fixture.target, identity()?, valid) + .await? + .output, + )?; + assert_eq!((result.generation, result.ref_generation), (2, 2)); + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + drop(next.prepared); + cleaned(next.root.path(), &next.budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn ancestry_growth_reuses_pairs_only_in_one_exact_native_catalog() -> Result { + use super::super::ref_proof::{RefProofError, ancestry::Walker}; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let graph = assembled(&fixture, [151; 16], 1200).await?; + let reader = + CatalogReader::open(graph.prepared.base.indexes(), graph.prepared.catalog()).await?; + let files = graph.prepared.base.files(); + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(4 << 20); + let mut walk = Walker::new(root.path(), budget.clone(), limits()).await?; + assert_eq!( + budget.used(), + crate::packs::metadata::growth::INITIAL_BYTES * 3 + ); + assert!( + walk.is_ancestor( + &reader, + &files, + graph.initial, + graph.tip, + &graph.prepared.base + ) + .await? + ); + assert!( + !walk + .is_ancestor( + &reader, + &files, + graph.other, + graph.tip, + &graph.prepared.base + ) + .await? + ); + assert!(budget.used() > crate::packs::metadata::growth::INITIAL_BYTES * 3); + // Both cached answers retain their meaning after another queue was used. + assert!( + walk.is_ancestor( + &reader, + &files, + graph.initial, + graph.tip, + &graph.prepared.base + ) + .await? + ); + assert!( + !walk + .is_ancestor( + &reader, + &files, + graph.other, + graph.tip, + &graph.prepared.base + ) + .await? + ); + let other = assembled(&fixture, [152; 16], 4).await?; + let foreign = + CatalogReader::open(other.prepared.base.indexes(), other.prepared.catalog()).await?; + assert!(matches!( + walk.is_ancestor( + &foreign, + &other.prepared.base.files(), + other.initial, + other.tip, + &other.prepared.base + ) + .await, + Err(RefProofError::Invalid) + )); + assert!(matches!( + walk.is_ancestor( + &reader, + &files, + graph.initial, + graph.tip, + &graph.prepared.base + ) + .await, + Err(RefProofError::Canceled) + )); + drop(walk); + cleaned(root.path(), &budget).await?; + drop((reader, foreign, files, graph.prepared, other.prepared)); + cleaned(graph.root.path(), &graph.budget).await?; + cleaned(other.root.path(), &other.budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn denied_native_ancestry_growth_cannot_become_a_negative_or_reused_answer() -> Result { + use super::super::ref_proof::{RefProofError, ancestry::Walker}; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let graph = assembled(&fixture, [153; 16], 1200).await?; + let reader = + CatalogReader::open(graph.prepared.base.indexes(), graph.prepared.catalog()).await?; + let files = graph.prepared.base.files(); + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(crate::packs::metadata::growth::INITIAL_BYTES * 3); + let mut walk = Walker::new(root.path(), budget.clone(), limits()).await?; + assert!(matches!( + walk.is_ancestor( + &reader, + &files, + graph.other, + graph.tip, + &graph.prepared.base + ) + .await, + Err(RefProofError::Metadata(MetadataError::Budget(_))) + )); + assert_eq!( + budget.used(), + crate::packs::metadata::growth::INITIAL_BYTES * 3 + ); + assert!(matches!( + walk.is_ancestor( + &reader, + &files, + graph.initial, + graph.tip, + &graph.prepared.base + ) + .await, + Err(RefProofError::Canceled) + )); + drop(walk); + cleaned(root.path(), &budget).await?; + drop((reader, files, graph.prepared)); + cleaned(graph.root.path(), &graph.budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn canceled_native_ancestry_fences_reuse_and_retains_queued_worker_credit() -> Result { + use super::super::ref_proof::{RefProofError, ancestry::Walker}; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let graph = assembled(&fixture, [154; 16], 4).await?; + let reader = + CatalogReader::open(graph.prepared.base.indexes(), graph.prepared.catalog()).await?; + let files = graph.prepared.base.files(); + // Warm membership so cancellation occurs with the walk awaiting scratch. + reader + .headers(&[graph.initial, graph.tip], &*files, &*files) + .await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(4 << 20); + let mut walk = Walker::new(root.path(), budget.clone(), limits()).await?; + let (entered, started) = tokio::sync::oneshot::channel(); + let (release, gate) = std::sync::mpsc::channel(); + let worker = walk.test_blocker(entered, gate); + started.await?; + let held = budget.used(); + let mut pending = Box::pin(walk.is_ancestor( + &reader, + &files, + graph.initial, + graph.tip, + &graph.prepared.base, + )); + assert!( + tokio::time::timeout(std::time::Duration::from_millis(25), &mut pending) + .await + .is_err() + ); + drop(pending); + assert!(matches!( + walk.is_ancestor( + &reader, + &files, + graph.initial, + graph.tip, + &graph.prepared.base + ) + .await, + Err(RefProofError::Canceled) + )); + drop(walk); + assert_eq!(budget.used(), held); + assert_eq!(std::fs::read_dir(root.path())?.count(), 1); + release.send(())?; + worker.await??; + cleaned(root.path(), &budget).await?; + drop((reader, files, graph.prepared)); + cleaned(graph.root.path(), &graph.budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn later_policy_changes_and_late_transaction_failures_publish_nothing() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let graph = assembled(&fixture, [75; 16], 8).await?; + let initial = proof( + &graph, + vec![update("refs/heads/main", None, Some(graph.initial))], + ) + .await?; + fixture + .client() + .command::(&fixture.target, identity()?, initial) + .await?; + let next = next_graph(&fixture, &graph, [76; 16]).await?; + let no_ancestry = proof( + &next, + vec![update( + "refs/heads/main", + Some((graph.initial, 1)), + Some(graph.tip), + )], + ) + .await?; + assert!(!super::super::ref_proof::proven(&no_ancestry.ancestry, 0)); + edit( + &fixture, + "INSERT INTO branch_rules VALUES('refs/heads/main',1,1,1,1,0,0);", + ) + .await?; + reject(&fixture, no_ancestry, PreparationDenial::Conflict).await?; + let valid = proof( + &next, + vec![update( + "refs/heads/main", + Some((graph.initial, 1)), + Some(graph.tip), + )], + ) + .await?; + edit(&fixture,"UPDATE branch_rules SET version=2,require_pull_request=1 WHERE reference='refs/heads/main';").await?; + reject(&fixture, valid.clone(), PreparationDenial::Conflict).await?; + edit(&fixture,"UPDATE branch_rules SET version=3,require_pull_request=0 WHERE reference='refs/heads/main'; CREATE TRIGGER forced_publication_failure BEFORE INSERT ON pushes BEGIN SELECT RAISE(ABORT,'forced late publication failure'); END;").await?; + let before = state(&fixture.handle).await?; + let failed = fixture + .client() + .command::(&fixture.target, identity()?, valid.clone()) + .await; + assert!(failed.is_err(), "{failed:?}"); + assert_eq!(state(&fixture.handle).await?, before); + edit(&fixture, "DROP TRIGGER forced_publication_failure;").await?; + fixture + .client() + .command::(&fixture.target, identity()?, valid) + .await?; + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + drop(next.prepared); + cleaned(next.root.path(), &next.budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn catalog_cas_and_authority_races_preserve_inputs_and_optional_checkpoints() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let first = assembled(&fixture, [77; 16], 0).await?; + let second = assembled(&fixture, [78; 16], 0).await?; + let input = proof( + &first, + vec![update("refs/heads/main", None, Some(first.initial))], + ) + .await?; + let stale = proof( + &second, + vec![update("refs/heads/second", None, Some(second.initial))], + ) + .await?; + // A catalog-only checkpoint remains immutable and compatible with the + // stronger final ref proof; it adds no third command to the ordinary path. + first.prepared.attest(identity()?).await?; + let mutation = identity()?; + let committed = fixture + .client() + .command::(&fixture.target, mutation, input.clone()) + .await?; + reject(&fixture, stale, PreparationDenial::Conflict).await?; + let next = next_graph(&fixture, &first, [79; 16]).await?; + let pending = proof( + &next, + vec![update("refs/heads/pending", None, Some(first.initial))], + ) + .await?; + edit( + &fixture, + "UPDATE repository_identity SET owner='successor' WHERE singleton=1;", + ) + .await?; + reject(&fixture, pending, PreparationDenial::Unauthorized).await?; + let after = state(&fixture.handle).await?; + let replay = fixture + .client() + .command::(&fixture.target, mutation, input.clone()) + .await?; + assert_eq!(replay.output, committed.output); + assert_eq!(replay.receipt, committed.receipt); + let logical = fixture + .client() + .command::(&fixture.target, identity()?, input) + .await?; + assert_eq!(logical.output, committed.output); + assert_eq!(state(&fixture.handle).await?, after); + drop(first.prepared); + cleaned(first.root.path(), &first.budget).await?; + drop(second.prepared); + cleaned(second.root.path(), &second.budget).await?; + drop(next.prepared); + cleaned(next.root.path(), &next.budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn publication_acknowledgements_survive_owner_restore_and_stale_attempts_cannot_write() +-> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let first = assembled(&fixture, [80; 16], 0).await?; + let second = assembled(&fixture, [81; 16], 0).await?; + let input = proof( + &first, + vec![update("refs/heads/main", None, Some(first.initial))], + ) + .await?; + let pending = proof( + &second, + vec![update("refs/heads/pending", None, Some(second.initial))], + ) + .await?; + let mutation = identity()?; + let committed = fixture + .client() + .command::(&fixture.target, mutation, input.clone()) + .await?; + let before = state(&fixture.handle).await?; + fixture.handle.drain().await?; + fixture.runtime.shutdown().await?; + let session = SessionId::from_bytes([82; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, session)?; + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await? + .ok_or("idle")?; + let provision = CellCatalog::new(fixture.layout.clone(), fixture.target.tenant()) + .lookup(fixture.target.cell_id()) + .await? + .ok_or("provision")?; + let handle = runtime + .acquire_idle_restored( + provision, + fixture.replica.clone(), + authority, + idle, + fixture.root.path().join("publication-b.sqlite"), + Owner { + session, + endpoint: "https://publication-b.invalid".into(), + }, + ) + .await?; + assert!(handle.owner_fence().epoch > first.prepared.token().owner.epoch); + let client = CellClient::local(Arc::clone(&fixture.registry), handle.clone()); + assert_eq!(state(&handle).await?, before); + let replay = client + .command::(&fixture.target, mutation, input.clone()) + .await?; + assert_eq!(replay.output, committed.output); + assert_eq!(replay.receipt, committed.receipt); + let logical = client + .command::(&fixture.target, identity()?, input) + .await?; + assert_eq!(logical.output, committed.output); + let stale = client + .command::(&fixture.target, identity()?, pending) + .await; + assert!( + matches!(stale,Err(InvocationError::Rejected(ref value)) if value.output==PublicationReply::Denied(PreparationDenial::Stale)), + "{stale:?}" + ); + assert_eq!(state(&handle).await?, before); + drop(first.prepared); + cleaned(first.root.path(), &first.budget).await?; + drop(second.prepared); + cleaned(second.root.path(), &second.budget).await?; + runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn expired_and_claimed_proofs_and_mutable_publication_facts_fail_closed() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let graph = assembled(&fixture, [83; 16], 0).await?; + let input = proof( + &graph, + vec![update("refs/heads/main", None, Some(graph.initial))], + ) + .await?; + edit( + &fixture, + "UPDATE catalog_operations SET expires_at_ms=0; UPDATE catalog_leases SET expires_at_ms=0;", + ) + .await?; + reject(&fixture, input.clone(), PreparationDenial::Expired).await?; + fixture + .client() + .command::( + &fixture.target, + identity()?, + request(graph.prepared.token()), + ) + .await?; + reject(&fixture, input, PreparationDenial::Stale).await?; + let fresh = assembled(&fixture, [84; 16], 0).await?; + let input = proof( + &fresh, + vec![update("refs/heads/main", None, Some(fresh.initial))], + ) + .await?; + fixture + .client() + .command::(&fixture.target, identity()?, input) + .await?; + let before = state(&fixture.handle).await?; + for sql in [ + "UPDATE pushes SET publication=zeroblob(53)", + "UPDATE pushes SET publication_plan_digest=zeroblob(32)", + "UPDATE pushes SET actor='outsider'", + "INSERT OR REPLACE INTO pushes SELECT id,actor,request_digest,options,response_id,rejected,rejection_reason,publication,publication_plan_digest FROM pushes", + "INSERT OR REPLACE INTO catalog_generations SELECT * FROM catalog_generations WHERE generation=1", + ] { + assert!(edit(&fixture, sql).await.is_err(), "{sql}"); + assert_eq!(state(&fixture.handle).await?, before); + } + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + drop(fresh.prepared); + cleaned(fresh.root.path(), &fresh.budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/reconcile.rs b/crates/canopy-server/src/packs/publication/tests/reconcile.rs new file mode 100644 index 0000000..2567e0a --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/reconcile.rs @@ -0,0 +1,711 @@ +use super::*; +use super::{ + prepare::{cleaned, opened, physical}, + publishing::{Graph, edit, plan, update}, +}; +use crate::{ + ObjectKind, + git_http::GitHttpResponse, + packs::{ + catalog::{CatalogReader, CatalogSnapshot}, + metadata::tests::limits, + verification::{ + PhysicalVerifier, + physical::tests::{ + independence::{git_input, upload_pair_for_operation}, + physical_limits, prepared_for_store, + }, + }, + }, +}; +use canopy_object_storage::artifact::ArtifactStore; +use cellule_ltx::DiskBudget; + +pub(super) async fn graph( + fixture: &Fixture, + provider: Arc, + store: Arc, + operation: [u8; 16], + blobs: usize, +) -> Result { + graph_with_run_limits(fixture, provider, store, operation, blobs, limits()).await +} +pub(super) async fn graph_with_run_limits( + fixture: &Fixture, + provider: Arc, + store: Arc, + operation: [u8; 16], + blobs: usize, + run_limits: crate::packs::metadata::MetadataLimits, +) -> Result { + let (base, _, _) = opened(fixture, operation, Arc::clone(&store)).await?; + let native = prepared_for_store( + fixture.format, + blobs, + base.context().operation, + provider, + Arc::clone(&store), + ) + .await?; + let initial = native + .fixture + .objects + .values() + .find(|(object, _)| object.kind == ObjectKind::Commit) + .ok_or("commit")? + .0 + .oid; + let blob = native + .fixture + .objects + .values() + .find(|(object, _)| object.kind == ObjectKind::Blob) + .ok_or("blob")? + .0 + .oid; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(256 << 20); + let mut builder = CatalogPreparation::new_with_run_limits( + root.path(), + budget.clone(), + base, + limits(), + run_limits, + ) + .await?; + let (witness, segments) = physical(&native, root.path(), budget.clone()).await?; + builder.begin_pack(witness)?; + for segment in segments { + builder.add_segment(segment).await?; + } + builder.finish_pack().await?; + Ok(Graph { + prepared: builder.finish().await?, + root, + budget, + initial, + tip: initial, + other: initial, + blob, + store, + }) +} +async fn publish( + fixture: &Fixture, + prepared: &PreparedCatalog, + root: &std::path::Path, + budget: DiskBudget, + name: &str, + tip: crate::ObjectId, +) -> Result { + let proof = Box::pin(prepared.ref_proof( + plan(vec![update(name, None, Some(tip))]), + root, + budget, + limits(), + )) + .await?; + assert!(proof.certificate.bytes()?.len() <= CERTIFICATE_BYTES as usize); + match fixture + .client() + .command::(&fixture.target, identity()?, proof) + .await? + .output + { + PublicationReply::Published(value) => Ok(value), + value => Err(format!("unexpected publication {value:?}").into()), + } +} + +#[tokio::test] +async fn concurrent_native_inputs_reconcile_without_claim_and_publish_exact_network_outcome() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new( + Arc::clone(&provider), + fixture.repository, + )); + let first = graph( + &fixture, + Arc::clone(&provider), + Arc::clone(&store), + [80; 16], + 4, + ) + .await?; + let second = graph( + &fixture, + Arc::clone(&provider), + Arc::clone(&store), + [81; 16], + 5, + ) + .await?; + let third = graph(&fixture, provider, Arc::clone(&store), [82; 16], 6).await?; + let original = second.prepared.token(); + second.prepared.attest(identity()?).await?; + let checkpoint = fixture + .handle + .query(0, 4096, move |connection| { + Ok(connection.query_row( + "SELECT attestation FROM catalog_leases WHERE artifact_operation=?1", + [original.artifact_operation.as_slice()], + |row| row.get::<_, Vec>(0), + )?) + }) + .await?; + assert_eq!( + publish( + &fixture, + &first.prepared, + first.root.path(), + first.budget.clone(), + "refs/heads/a", + first.initial + ) + .await? + .generation, + 1 + ); + let stale = second + .prepared + .ref_proof( + plan(vec![update("refs/heads/b", None, Some(second.initial))]), + second.root.path(), + second.budget.clone(), + limits(), + ) + .await?; + assert!( + matches!(fixture.client().command::(&fixture.target,identity()?,stale).await,Err(InvocationError::Rejected(value)) if value.output==PublicationReply::Denied(PreparationDenial::Conflict)) + ); + let before = fixture.counts().await?; + let reconciled = second.prepared.reconcile().await?; + assert_eq!(fixture.counts().await?, before); + assert_eq!(reconciled.token(), original); + assert_eq!(reconciled.base().generation, 1); + assert_eq!(reconciled.base.retention_floor().generation, 0); + assert_eq!(reconciled.inputs_digest(), second.prepared.inputs_digest()); + assert_eq!( + reconciled.inventory_digest(), + second.prepared.inventory_digest() + ); + assert_eq!(reconciled.object_count(), second.prepared.object_count()); + // A trusted fault fixture MAC cannot substitute the selected generation + // for the operation's immutable original retention floor. + let mut wrong = reconciled + .ref_proof( + plan(vec![update("refs/heads/b", None, Some(second.initial))]), + second.root.path(), + second.budget.clone(), + limits(), + ) + .await?; + let mut wrong_data = wrong.certificate.data()?; + wrong_data.retention_floor = wrong_data.base.generation; + wrong_data.retention_certificate = wrong_data.base.certificate; + wrong.certificate = CatalogCertificate::seal(&wrong_data, &[16; 32])?; + let unchanged = super::publishing::state(&fixture.handle).await?; + assert!( + matches!(fixture.client().command::(&fixture.target,identity()?,wrong).await,Err(InvocationError::Rejected(value)) if value.output==PublicationReply::Denied(PreparationDenial::Conflict)) + ); + assert_eq!(super::publishing::state(&fixture.handle).await?, unchanged); + assert_eq!( + publish( + &fixture, + &reconciled, + second.root.path(), + second.budget.clone(), + "refs/heads/b", + second.initial + ) + .await? + .generation, + 2 + ); + let preserved = fixture + .handle + .query(0, 4096, move |connection| { + Ok(connection.query_row( + "SELECT attestation FROM catalog_leases WHERE artifact_operation=?1", + [original.artifact_operation.as_slice()], + |row| row.get::<_, Vec>(0), + )?) + }) + .await?; + assert_eq!(preserved, checkpoint); // immutable optional input checkpoint + let final_catalog = third.prepared.reconcile().await?; + assert_eq!(final_catalog.base().generation, 2); + let mut body = Vec::new(); + for line in ["unpack ok\n", "ok refs/heads/c\n"] { + body.extend_from_slice(format!("{:04x}{line}", line.len() + 4).as_bytes()); + } + body.extend_from_slice(b"0000"); + let response = GitHttpResponse { + status: 200, + headers: vec![], + body, + }; + let completed = final_catalog + .complete_push( + identity()?, + PushCompletionRequest { + plan: Some(plan(vec![update( + "refs/heads/c", + None, + Some(third.initial), + )])), + response: response.clone(), + options: vec![], + certificate: None, + }, + third.root.path(), + third.budget.clone(), + limits(), + ) + .await?; + assert!( + matches!(completed.output,CatalogCompletionReply::Completed(ref value) if value.publication.is_some_and(|result|result.generation==3)&&!value.rejected) + ); + assert_eq!( + final_catalog.completed_push_response(&completed).await?, + response + ); + let indexes = final_catalog.base.indexes(); + let files = final_catalog.base.files(); + let reader = CatalogReader::open(Arc::clone(&indexes), final_catalog.catalog()).await?; + let tips = [first.initial, second.initial, third.initial]; + assert!( + reader + .headers(&tips, &*files, &*files) + .await? + .iter() + .all(Option::is_some) + ); + let snapshot = CatalogSnapshot::download(&store, final_catalog.catalog()).await?; + let sources = indexes.sources(); + let mut cursor = sources.cursor(snapshot.sources, None)?; + let mut namespaces = std::collections::BTreeSet::new(); + while let Some(record) = cursor.next().await? { + namespaces.insert(record.native().operation); + } + assert_eq!( + namespaces, + [ + first.prepared.token().artifact_operation, + second.prepared.token().artifact_operation, + third.prepared.token().artifact_operation + ] + .into_iter() + .collect() + ); + drop(reconciled); + drop(final_catalog); + for graph in [first, second, third] { + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + } + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn reconciliation_preserves_external_dependencies_and_rejects_their_removal() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new( + Arc::clone(&provider), + fixture.repository, + )); + let first = graph( + &fixture, + Arc::clone(&provider), + Arc::clone(&store), + [83; 16], + 4, + ) + .await?; + publish( + &fixture, + &first.prepared, + first.root.path(), + first.budget.clone(), + "refs/heads/a", + first.initial, + ) + .await?; + let (base, _, _) = opened(&fixture, [84; 16], Arc::clone(&store)).await?; + let native = prepared_for_store( + format, + 4, + base.context().operation, + provider, + Arc::clone(&store), + ) + .await?; + let commit = native + .fixture + .objects + .values() + .find(|(object, _)| object.kind == ObjectKind::Commit) + .ok_or("commit")? + .0; + let bytes = git_input( + native.fixture.root.path(), + &["pack-objects", "--stdout", "--no-reuse-delta"], + format!("{}\n", hex::encode(commit.oid)).as_bytes(), + ) + .await?; + let descriptor = + upload_pair_for_operation(&native, commit.oid, &bytes, base.context().operation) + .await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(256 << 20); + let mut physical = PhysicalVerifier::download( + root.path(), + budget.clone(), + &store, + descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + let segment = physical.inspect_next_shard(1).await?; + let mut builder = + CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + builder.begin_pack(physical.finish().await?)?; + builder.add_segment(segment).await?; + builder.finish_pack().await?; + let prepared = builder.finish().await?; + // A fixture-only new certified root changes the generation but preserves + // canonical dependencies. Physical work and original floor are reused. + fixture.install_catalog(2, first.prepared.catalog()).await?; + let reconciled = prepared.reconcile().await?; + assert_eq!(reconciled.base().generation, 2); + assert_eq!(reconciled.base.retention_floor().generation, 1); + assert_eq!(reconciled.object_count(), 1); + let mut data = reconciled.certificate().await?.data()?; + assert_eq!(data.retention_floor, 1); + assert_eq!( + data.retention_certificate, + prepared.base.retention_floor().certificate + ); + data.actor = "a".repeat(64); + data.refs_digest = Some([1; 32]); + data.completion_digest = Some([2; 32]); + let maximum = CatalogCertificate::seal(&data, &[16; 32])?; + assert!(maximum.bytes()?.len() <= CERTIFICATE_BYTES as usize); + // Fault fixture: current certification loses the external tree. The + // old floor still retains its bytes, but the proposed new root must not + // silently refer to that absent dependency. + let directory = + crate::packs::directory::snapshot::DirectorySnapshot::empty(fixture.repository, format) + .upload(&store, [85; 16]) + .await?; + let empty = CatalogSnapshot { + directory, + sources: None, + } + .upload(&store, [85; 16]) + .await?; + fixture.install_catalog(3, empty).await?; + assert!(matches!( + prepared.reconcile().await, + Err(CatalogPreparationError::Closure( + crate::packs::closure::ClosureError::Missing(_) + )) + )); + assert!(prepared.ensure_live().is_ok()); + drop(reconciled); + drop(prepared); + cleaned(root.path(), &budget).await?; + drop(first.prepared); + cleaned(first.root.path(), &first.budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn reconciliation_rechecks_revocation_expiry_and_claim_without_releasing_inputs() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new( + Arc::clone(&provider), + fixture.repository, + )); + let graph = graph(&fixture, provider, store, [86; 16], 4).await?; + let original = graph.prepared.token(); + let held = graph.budget.used(); + assert!(held > 0); + edit( + &fixture, + "UPDATE repository_identity SET owner='replacement'", + ) + .await?; + assert!(graph.prepared.reconcile().await.is_err()); + assert_eq!(graph.budget.used(), held); + edit(&fixture, "UPDATE repository_identity SET owner='owner'").await?; + let same = graph.prepared.reconcile().await?; + assert_eq!(same.catalog(), graph.prepared.catalog()); + fixture + .client() + .command::(&fixture.target, identity()?, request(original)) + .await?; + assert!(graph.prepared.reconcile().await.is_err()); + drop(same); + let fresh = super::publishing::next_graph(&fixture, &graph, [89; 16]).await?; + let fresh_held = fresh.budget.used(); + edit( + &fixture, + "UPDATE catalog_operations SET expires_at_ms=0; UPDATE catalog_leases SET expires_at_ms=0;", + ) + .await?; + assert!(fresh.prepared.reconcile().await.is_err()); + assert_eq!(fresh.budget.used(), fresh_held); + drop(fresh.prepared); + cleaned(fresh.root.path(), &fresh.budget).await?; + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn reconciliation_rejects_canonical_body_and_graph_conflicts_in_the_new_base() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for graph_conflict in [false, true] { + let fixture = Fixture::new(format).await?; + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new( + Arc::clone(&provider), + fixture.repository, + )); + let incoming = graph( + &fixture, + Arc::clone(&provider), + Arc::clone(&store), + [87; 16], + 4, + ) + .await?; + let (base, _, _) = opened(&fixture, [88; 16], Arc::clone(&store)).await?; + let mut fake = prepared_for_store( + format, + 4, + base.context().operation, + provider, + Arc::clone(&store), + ) + .await?; + let altered = fake + .fixture + .objects + .values_mut() + .find(|(object, _)| { + object.kind + == if graph_conflict { + ObjectKind::Tree + } else { + ObjectKind::Blob + } + }) + .ok_or("altered object")?; + if graph_conflict { + assert!(!altered.1.is_empty()); + altered.1.clear(); + } else { + altered.0.digest[0] ^= 1; + } + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let mut metadata = crate::packs::metadata::MetadataBuilder::new( + root.path(), + budget.clone(), + fake.fixture.identity, + limits(), + )?; + crate::packs::metadata::tests::fill( + &mut metadata, + &fake.fixture.objects.values().cloned().collect::>(), + )?; + let segment = Arc::new(metadata.seal(&fake.fixture.index)?); + let mut directory = crate::packs::directory::DirectoryBuilder::new( + root.path(), + budget.clone(), + fixture.repository, + base.context().operation, + format, + limits(), + )?; + directory.add_segment(&segment)?; + let directory = Arc::new(directory.seal()?); + let run = Arc::clone(&directory).upload(&store).await?; + let sources = base.indexes().sources(); + let source = sources + .insert( + None, + base.context().operation, + crate::packs::sources::SourceRecord { + metadata: Arc::clone(&segment).upload(&store).await?, + pack: fake.descriptor.pack, + index: fake.descriptor.index, + pack_object_count: fake.descriptor.object_count, + }, + ) + .await?; + let mut snapshot = crate::packs::directory::snapshot::DirectorySnapshot::empty( + fixture.repository, + format, + ); + let indexes = base.indexes(); + let run_root = indexes + .ranges() + .insert(None, base.context().operation, run) + .await?; + snapshot.append(indexes.ranges(), run_root).await?; + let catalog = CatalogSnapshot { + directory: snapshot.upload(&store, base.context().operation).await?, + sources: Some(source), + } + .upload(&store, base.context().operation) + .await?; + // Trusted fault injection, not a route for public certification: + // source and directory agree but the canonical identity is false. + fixture.install_catalog(1, catalog).await?; + let held = incoming.budget.used(); + assert!(matches!( + incoming.prepared.reconcile().await, + Err(CatalogPreparationError::Closure( + crate::packs::closure::ClosureError::Metadata( + crate::packs::metadata::MetadataError::IdentityConflict + ) + )) + )); + assert_eq!(incoming.budget.used(), held); + assert_eq!(incoming.prepared.base().generation, 0); + drop(segment); + drop(directory); + cleaned(root.path(), &budget).await?; + drop(incoming.prepared); + cleaned(incoming.root.path(), &incoming.budget).await?; + fixture.runtime.shutdown().await?; + } + } + Ok(()) +} + +#[tokio::test] +async fn reconciled_publication_replays_after_owner_restore_and_pending_old_proofs_are_fenced() +-> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new( + Arc::clone(&provider), + fixture.repository, + )); + let first = graph( + &fixture, + Arc::clone(&provider), + Arc::clone(&store), + [90; 16], + 4, + ) + .await?; + let second = graph( + &fixture, + Arc::clone(&provider), + Arc::clone(&store), + [91; 16], + 5, + ) + .await?; + let pending = graph(&fixture, provider, store, [92; 16], 6).await?; + publish( + &fixture, + &first.prepared, + first.root.path(), + first.budget.clone(), + "refs/heads/a", + first.initial, + ) + .await?; + let second_ready = second.prepared.reconcile().await?; + let proof = second_ready + .ref_proof( + plan(vec![update("refs/heads/b", None, Some(second.initial))]), + second.root.path(), + second.budget.clone(), + limits(), + ) + .await?; + let mutation = identity()?; + let committed = fixture + .client() + .command::(&fixture.target, mutation, proof.clone()) + .await?; + let pending_ready = pending.prepared.reconcile().await?; + let stale = pending_ready + .ref_proof( + plan(vec![update("refs/heads/c", None, Some(pending.initial))]), + pending.root.path(), + pending.budget.clone(), + limits(), + ) + .await?; + assert_eq!(pending_ready.base().generation, 2); + fixture.handle.drain().await?; + fixture.runtime.shutdown().await?; + let session = SessionId::from_bytes([93; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, session)?; + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await? + .ok_or("idle")?; + let provision = CellCatalog::new(fixture.layout.clone(), fixture.target.tenant()) + .lookup(fixture.target.cell_id()) + .await? + .ok_or("provision")?; + let handle = runtime + .acquire_idle_restored( + provision, + fixture.replica.clone(), + authority, + idle, + fixture.root.path().join("reconciled-restore.sqlite"), + Owner { + session, + endpoint: "https://reconciled-owner.invalid".into(), + }, + ) + .await?; + let client = CellClient::local(Arc::clone(&fixture.registry), handle.clone()); + let exact = client + .command::(&fixture.target, mutation, proof.clone()) + .await?; + assert_eq!(exact.receipt, committed.receipt); + assert_eq!(exact.output, committed.output); + let logical = client + .command::(&fixture.target, identity()?, proof) + .await?; + assert_eq!(logical.output, committed.output); + let before = super::publishing::state(&handle).await?; + assert!( + matches!(client.command::(&fixture.target,identity()?,stale).await,Err(InvocationError::Rejected(value)) if value.output==PublicationReply::Denied(PreparationDenial::Stale)) + ); + assert_eq!(super::publishing::state(&handle).await?, before); + drop(second_ready); + drop(pending_ready); + for graph in [first, second, pending] { + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + } + runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/ref_policy.rs b/crates/canopy-server/src/packs/publication/tests/ref_policy.rs new file mode 100644 index 0000000..795035a --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/ref_policy.rs @@ -0,0 +1,196 @@ +use super::*; +use super::{ + prepare::{cleaned, opened}, + publishing::{Graph, edit, plan, state, update}, + reconcile::graph, +}; +use crate::packs::metadata::tests::limits; +use canopy_object_storage::artifact::ArtifactStore; +use cellule_ltx::DiskBudget; + +mod cleanup; +mod freshness; +mod pages; + +async fn rooted(format: ObjectFormat) -> Result<(Fixture, Graph)> { + let fixture = Fixture::new(format).await?; + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new(provider.clone(), fixture.repository)); + let (base, _, _) = opened(&fixture, [230; 16], store.clone()).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 20); + let empty = CatalogPreparation::new(root.path(), budget.clone(), base, limits()) + .await? + .finish() + .await?; + let proof = empty.empty_ref_initialization().await?; + fixture + .client() + .command::(&fixture.target, identity()?, proof) + .await?; + drop(empty); + cleaned(root.path(), &budget).await?; + let graph = graph(&fixture, provider, store, [231; 16], 4).await?; + Ok((fixture, graph)) +} +async fn close(fixture: Fixture, graph: Graph) -> Result { + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + fixture.runtime.shutdown().await?; + Ok(()) +} +fn run_sql(oid: crate::ObjectId, context: &str, version: u64, state: &str) -> String { + format!( + "INSERT INTO check_runs(id,oid,context,context_version,reporter,state,version,summary,created_ms,updated_ms) VALUES(X'{}',X'{}','{context}',{version},'owner','{state}',1,'',0,0)", + hex::encode(uuid::Uuid::new_v4().as_bytes()), + hex::encode(oid) + ) +} +async fn protect(fixture: &Fixture, graph: &Graph) -> Result { + edit(fixture,&format!("INSERT INTO check_contexts VALUES('ci','owner',1,1); INSERT INTO branch_rules VALUES('refs/heads/main',1,1,0,1,0,0); INSERT INTO branch_required_checks VALUES('refs/heads/main','ci'); {}; {}",run_sql(graph.tip,"ci",1,"queued"),run_sql(graph.tip,"ci",1,"success"))).await +} +async fn registered( + fixture: &Fixture, + graph: &Graph, + changes: crate::PushPlan, +) -> Result<(RefPolicyPreparation, RefPolicyPage)> { + fn send(value: T) -> T { + value + } + let pending = send(graph.prepared.ref_policy_preparation( + changes, + graph.root.path(), + graph.budget.clone(), + limits(), + )) + .await?; + let page = send(pending.page(&graph.prepared, 0)).await?; + let reply = fixture + .client() + .command::(&fixture.target, identity()?, page.clone()) + .await?; + assert!(matches!(reply.output,RefPolicyReply::Registered(progress) if progress.ready())); + Ok((pending, page)) +} +#[tokio::test] +async fn guarded_root_binds_catalog_conditional_refs_and_original_intent_and_survives_unrelated_rebase() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (fixture, graph) = rooted(format).await?; + protect(&fixture, &graph).await?; + let changes = plan(vec![ + update("refs/heads/main", None, Some(graph.tip)), + update("refs/tags/example", None, Some(graph.initial)), + ]); + let (pending, page) = registered(&fixture, &graph, changes.clone()).await?; + let receipt = fixture + .client() + .command::(&fixture.target, identity()?, page.clone()) + .await?; + let replay = fixture + .client() + .command::(&fixture.target, identity()?, page.clone()) + .await?; + assert_eq!(receipt.output, replay.output); + let guard = pending.ready(&graph.prepared).await?; + let old = graph.prepared.base().refs; + fixture + .install_generation(2, graph.prepared.catalog(), old) + .await?; + let rebased = graph.prepared.reconcile().await?; + assert!(pending.ready(&rebased).await.is_ok()); + fn send(value: T) -> T { + value + } + let proof = send(rebased.guarded_ref_snapshot( + &guard, + changes.clone(), + graph.root.path(), + graph.budget.clone(), + limits(), + )) + .await?; + assert_eq!(proof.certificate.data()?.base.generation, 2); + assert_eq!( + proof.guard.plan_digest, + super::super::ref_proof::plan_digest(&changes)? + ); + assert_eq!(proof.snapshot.read(&graph.store).await?.generation, 1); + let mut e = BoundedEncoder::new(2048)?; + proof.encode(&mut e)?; + let bytes = e.finish(); + assert!(bytes.len() < 2048); + let mut d = BoundedDecoder::new(&bytes, 2048)?; + assert_eq!(RefRootPublicationProof::decode(&mut d)?, proof); + d.finish()?; + for cut in 0..bytes.len() { + assert!( + RefRootPublicationProof::decode(&mut BoundedDecoder::new(&bytes[..cut], 2048)?) + .is_err() + ); + } + let before = state(&fixture.handle).await?; + let legacy = RefPublicationProof { + plan: changes, + ancestry: page.proof.ancestry, + certificate: proof.certificate, + }; + let result = fixture + .client() + .command::(&fixture.target, identity()?, legacy) + .await; + assert!( + matches!(result,Err(InvocationError::Rejected(value)) if value.output==PublicationReply::Denied(PreparationDenial::Unauthorized)) + ); + assert_eq!(before, state(&fixture.handle).await?); + drop(rebased); + close(fixture, graph).await?; + } + Ok(()) +} + +#[tokio::test] +async fn exact_check_dependencies_ignore_unrelated_and_older_reports_and_refuse_newer_or_replaced_attempts() +-> Result { + let (fixture, graph) = rooted(ObjectFormat::Sha256).await?; + protect(&fixture, &graph).await?; + let changes = plan(vec![update("refs/heads/main", None, Some(graph.tip))]); + let (pending, _) = registered(&fixture, &graph, changes.clone()).await?; + assert_ne!(graph.blob, graph.tip); + edit(&fixture,&format!("{}; {}; UPDATE check_runs SET state='failure',version=2 WHERE number=1; DELETE FROM check_runs WHERE number=1",run_sql(graph.blob,"ci",1,"success"),run_sql(graph.tip,"ci",2,"failure"))).await?; + assert!(pending.ready(&graph.prepared).await.is_ok()); + edit(&fixture, &run_sql(graph.tip, "ci", 1, "queued")).await?; + assert!(matches!( + pending.ready(&graph.prepared).await, + Err(RefPolicyPreparationError::Context) + )); + edit(&fixture,"UPDATE check_runs SET state='success',version=2 WHERE number=(SELECT max(number) FROM check_runs WHERE context_version=1)").await?; + let (pending, _) = registered(&fixture, &graph, changes.clone()).await?; + edit(&fixture,&format!("INSERT OR REPLACE INTO check_runs SELECT number,id,X'{}',context,context_version,reporter,state,version,summary,created_ms,updated_ms FROM check_runs WHERE oid=X'{}' AND context_version=1 ORDER BY number DESC LIMIT 1",hex::encode(graph.blob),hex::encode(graph.tip))).await?; + assert!(matches!( + pending.ready(&graph.prepared).await, + Err(RefPolicyPreparationError::Context) + )); + edit(&fixture, &run_sql(graph.tip, "ci", 1, "success")).await?; + let (pending, _) = registered(&fixture, &graph, changes.clone()).await?; + edit(&fixture,&format!("INSERT OR REPLACE INTO check_runs(id,oid,context,context_version,reporter,state,version,summary,created_ms,updated_ms) SELECT id,X'{}',context,context_version,reporter,state,version,summary,created_ms,updated_ms FROM check_runs WHERE oid=X'{}' AND context_version=1 ORDER BY number DESC LIMIT 1",hex::encode(graph.blob),hex::encode(graph.tip))).await?; + assert!(matches!( + pending.ready(&graph.prepared).await, + Err(RefPolicyPreparationError::Context) + )); + edit(&fixture, &run_sql(graph.tip, "ci", 1, "success")).await?; + let (pending, _) = registered(&fixture, &graph, changes).await?; + edit( + &fixture, + &format!( + "DELETE FROM check_runs WHERE oid=X'{}' AND context_version=1", + hex::encode(graph.tip) + ), + ) + .await?; + assert!(matches!( + pending.ready(&graph.prepared).await, + Err(RefPolicyPreparationError::Context) + )); + close(fixture, graph).await +} diff --git a/crates/canopy-server/src/packs/publication/tests/ref_policy/cleanup.rs b/crates/canopy-server/src/packs/publication/tests/ref_policy/cleanup.rs new file mode 100644 index 0000000..c973cf6 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/ref_policy/cleanup.rs @@ -0,0 +1,197 @@ +use super::*; + +fn request(fixture: &Fixture, id: [u8; 16]) -> RefPolicyReap { + RefPolicyReap { + maintenance: MaintenanceRequest { + repository: fixture.repository, + actor: "owner".into(), + owner: fixture.handle.owner_fence(), + }, + id, + } +} + +#[tokio::test] +async fn cleanup_is_bounded_transactional_and_retains_live_invalid_tombstones() -> Result { + let (fixture, graph) = rooted(ObjectFormat::Sha256).await?; + protect(&fixture, &graph).await?; + let (pending, page) = registered( + &fixture, + &graph, + plan(vec![update("refs/heads/main", None, Some(graph.tip))]), + ) + .await?; + let id = pending.intent().id; + let plans=fixture.handle.query(0,8192,move|connection| { + let statements=[ + "EXPLAIN QUERY PLAN SELECT guard FROM ref_policy_watches WHERE oid=?1 AND context=?2 AND context_version=?3 AND run_number<=?4", + "EXPLAIN QUERY PLAN SELECT oid,context,context_version,run_number FROM ref_policy_watches WHERE guard=?1 ORDER BY oid,context,context_version,run_number LIMIT 512", + ]; + let mut result=String::new(); + for sql in statements { + let mut q=connection.prepare(sql)?; + let count=q.parameter_count(); + let mut rows=q.query(rusqlite::params_from_iter(std::iter::repeat_n(rusqlite::types::Value::Null,count)))?; + while let Some(row)=rows.next()? { + result.push_str(&row.get::<_,String>(3)?); + result.push('\n'); + } + } + Ok(result.into_bytes()) + }).await?; + let plans = String::from_utf8(plans)?; + assert!(plans.contains("ref_policy_watches_by_check"), "{plans}"); + assert!(plans.contains("USING PRIMARY KEY (guard=?)"), "{plans}"); + assert!(!plans.contains("SCAN ref_policy_watches"), "{plans}"); + let input = request(&fixture, id); + let before = state(&fixture.handle).await?; + assert!(matches!(fixture.client().command::( + &fixture.target,identity()?,input.clone()).await, + Err(InvocationError::Rejected(value)) if value.output==RefPolicyReapReply::Denied(PreparationDenial::Conflict))); + assert_eq!(state(&fixture.handle).await?, before); + // Trusted fixture injection exercises the indexed 512-row boundary without + // pretending that 600 synthetic contexts prove large-team capacity. + edit(&fixture,&format!("WITH RECURSIVE seq(x) AS (VALUES(1) UNION ALL SELECT x+1 FROM seq WHERE x<599) INSERT INTO ref_policy_watches SELECT X'{}',X'{}',printf('context-%04d',x),1,2 FROM seq; UPDATE ref_policy_budget SET watches=600; UPDATE ref_policy_guards SET valid=0",hex::encode(id),hex::encode(graph.tip))).await?; + edit(&fixture,"CREATE TRIGGER fail_watch_budget BEFORE UPDATE ON ref_policy_budget BEGIN SELECT RAISE(ABORT,'late watch budget fault'); END").await?; + let before = state(&fixture.handle).await?; + let mutation = identity()?; + assert!( + fixture + .client() + .command::(&fixture.target, mutation, input.clone()) + .await + .is_err() + ); + assert_eq!(state(&fixture.handle).await?, before); + edit(&fixture, "DROP TRIGGER fail_watch_budget").await?; + assert_eq!( + fixture + .client() + .command::(&fixture.target, mutation, input.clone()) + .await? + .output, + RefPolicyReapReply::Reaped { + watches: 512, + removed: false + } + ); + assert_eq!( + fixture + .client() + .command::(&fixture.target, identity()?, input.clone()) + .await? + .output, + RefPolicyReapReply::Reaped { + watches: 88, + removed: false + } + ); + let before = state(&fixture.handle).await?; + assert!(matches!(fixture.client().command::( + &fixture.target,identity()?,page.clone()).await, + Err(InvocationError::Rejected(value)) if value.output==RefPolicyReply::Denied(PreparationDenial::Conflict))); + assert_eq!(state(&fixture.handle).await?, before); + fixture + .client() + .command::(&fixture.target, identity()?, check(graph.prepared.token())) + .await?; + assert_eq!( + fixture + .client() + .command::(&fixture.target, identity()?, input.clone()) + .await? + .output, + RefPolicyReapReply::Reaped { + watches: 0, + removed: true + } + ); + let before = state(&fixture.handle).await?; + assert!(matches!(fixture.client().command::( + &fixture.target,identity()?,page).await, + Err(InvocationError::Rejected(value)) if value.output==RefPolicyReply::Denied(PreparationDenial::Missing))); + assert_eq!(state(&fixture.handle).await?, before); + close(fixture, graph).await +} + +#[tokio::test] +async fn restored_owner_cannot_replay_old_guard_into_new_writes_and_can_reap_its_watches() -> Result +{ + let (fixture, graph) = rooted(ObjectFormat::Sha1).await?; + protect(&fixture, &graph).await?; + let (pending, page) = registered( + &fixture, + &graph, + plan(vec![update("refs/heads/main", None, Some(graph.tip))]), + ) + .await?; + let mutation = identity()?; + let original = fixture + .client() + .command::(&fixture.target, mutation, page.clone()) + .await?; + fixture.handle.drain().await?; + fixture.runtime.shutdown().await?; + let session = SessionId::from_bytes([232; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, session)?; + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await? + .ok_or("idle")?; + let provision = CellCatalog::new(fixture.layout.clone(), fixture.target.tenant()) + .lookup(fixture.target.cell_id()) + .await? + .ok_or("provision")?; + let handle = runtime + .acquire_idle_restored( + provision, + fixture.replica.clone(), + authority, + idle, + fixture.root.path().join("ref-policy-restored.sqlite"), + Owner { + session, + endpoint: "https://ref-policy-restored.invalid".into(), + }, + ) + .await?; + assert!(handle.owner_fence().epoch > graph.prepared.token().owner.epoch); + let client = CellClient::local(Arc::clone(&fixture.registry), handle.clone()); + let exact = client + .command::(&fixture.target, mutation, page.clone()) + .await?; + assert_eq!( + (exact.output, exact.receipt), + (original.output, original.receipt) + ); + let before = state(&handle).await?; + assert!(matches!(client.command::( + &fixture.target,identity()?,page).await, + Err(InvocationError::Rejected(value)) if value.output==RefPolicyReply::Denied(PreparationDenial::Stale))); + assert_eq!(state(&handle).await?, before); + assert_eq!( + client + .command::( + &fixture.target, + identity()?, + RefPolicyReap { + maintenance: MaintenanceRequest { + owner: handle.owner_fence(), + ..request(&fixture, pending.intent().id).maintenance + }, + id: pending.intent().id, + } + ) + .await? + .output, + RefPolicyReapReply::Reaped { + watches: 1, + removed: true + } + ); + drop(graph.prepared); + cleaned(graph.root.path(), &graph.budget).await?; + runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/ref_policy/freshness.rs b/crates/canopy-server/src/packs/publication/tests/ref_policy/freshness.rs new file mode 100644 index 0000000..65af05c --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/ref_policy/freshness.rs @@ -0,0 +1,143 @@ +use super::*; + +#[tokio::test] +async fn current_acl_and_new_direct_push_rules_are_checked_before_registration() -> Result { + let (fixture, graph) = rooted(ObjectFormat::Sha256).await?; + protect(&fixture, &graph).await?; + let pending = graph + .prepared + .ref_policy_preparation( + plan(vec![update("refs/heads/main", None, Some(graph.tip))]), + graph.root.path(), + graph.budget.clone(), + limits(), + ) + .await?; + let page = pending.page(&graph.prepared, 0).await?; + for (sql, reason) in [ + ( + "UPDATE repository_identity SET owner='another'", + PreparationDenial::Unauthorized, + ), + ( + "UPDATE repository_identity SET owner='owner'; UPDATE branch_rules SET require_pull_request=1,version=version+1", + PreparationDenial::Conflict, + ), + ] { + edit(&fixture, sql).await?; + let before = state(&fixture.handle).await?; + assert!(matches!(fixture.client().command::( + &fixture.target,identity()?,page.clone()).await, + Err(InvocationError::Rejected(value)) if value.output==RefPolicyReply::Denied(reason))); + assert_eq!(state(&fixture.handle).await?, before); + assert!(pending.ready(&graph.prepared).await.is_err()); + } + // Freshly certified pages cannot bypass a pull-request rule either. + let pending = graph + .prepared + .ref_policy_preparation( + plan(vec![update("refs/heads/main", None, Some(graph.tip))]), + graph.root.path(), + graph.budget.clone(), + limits(), + ) + .await?; + let page = pending.page(&graph.prepared, 0).await?; + let before = state(&fixture.handle).await?; + assert!(matches!(fixture.client().command::( + &fixture.target,identity()?,page).await, + Err(InvocationError::Rejected(value)) if value.output==RefPolicyReply::Denied(PreparationDenial::Conflict))); + assert_eq!(state(&fixture.handle).await?, before); + close(fixture, graph).await +} + +#[tokio::test] +async fn watched_updates_and_rule_context_edits_invalidate_fresh_readiness() -> Result { + let (fixture, graph) = rooted(ObjectFormat::Sha256).await?; + protect(&fixture, &graph).await?; + let changes = plan(vec![update("refs/heads/main", None, Some(graph.tip))]); + let watched = format!( + "oid=X'{}' AND context='ci' AND context_version=1", + hex::encode(graph.tip) + ); + // Even successful-to-successful updates invalidate their exact witness. + // A new preparation observes and validates the replacement facts. + for change in [ + "state='failure',version=version+1".to_string(), + "reporter='another',version=version+1".into(), + format!("oid=X'{}',version=version+1", hex::encode(graph.blob)), + "context_version=2,version=version+1".into(), + "state='success',version=version+1".into(), + ] { + let (pending, _) = registered(&fixture, &graph, changes.clone()).await?; + edit(&fixture,&format!("UPDATE check_runs SET {change} WHERE number=(SELECT max(number) FROM check_runs WHERE {watched})")).await?; + assert!(matches!( + pending.ready(&graph.prepared).await, + Err(RefPolicyPreparationError::Context) + )); + edit(&fixture, &run_sql(graph.tip, "ci", 1, "success")).await?; + } + for change in [ + "UPDATE branch_rules SET deny_deletions=1,version=version+1", + "UPDATE branch_rules SET require_pull_request=1,version=version+1", + "DELETE FROM branch_required_checks WHERE reference='refs/heads/main'", + "UPDATE check_contexts SET reporter='another',version=version+1", + ] { + let (pending, page) = registered(&fixture, &graph, changes.clone()).await?; + edit(&fixture, change).await?; + assert!(matches!( + pending.ready(&graph.prepared).await, + Err(RefPolicyPreparationError::Context) + )); + let before = state(&fixture.handle).await?; + assert!(matches!(fixture.client().command::( + &fixture.target, identity()?, page).await, + Err(InvocationError::Rejected(value)) if value.output==RefPolicyReply::Denied(PreparationDenial::Conflict))); + assert_eq!(state(&fixture.handle).await?, before); + // Restore configuration through actual mutations; do not reset epochs. + edit(&fixture,"UPDATE branch_rules SET require_pull_request=0,version=version+1; INSERT OR IGNORE INTO branch_required_checks VALUES('refs/heads/main','ci'); UPDATE check_contexts SET reporter='owner',version=1").await?; + edit(&fixture, &run_sql(graph.tip, "ci", 1, "success")).await?; + } + close(fixture, graph).await +} + +#[tokio::test] +async fn integer_counters_watch_immutability_and_invalid_guard_resurrection_are_refused() -> Result +{ + let (fixture, graph) = rooted(ObjectFormat::Sha1).await?; + protect(&fixture, &graph).await?; + let (pending, _) = registered( + &fixture, + &graph, + plan(vec![update("refs/heads/main", None, Some(graph.tip))]), + ) + .await?; + for sql in [ + "UPDATE ref_policy_guards SET next=next+0.5,total=total+0.5", + "UPDATE ref_policy_guards SET policy_epoch=policy_epoch+0.5", + "UPDATE ref_policy_watches SET context_version=context_version+0.5", + "UPDATE ref_policy_watches SET run_number=run_number+0.5", + "DELETE FROM ref_policy_watches", + "UPDATE ref_policy_epoch SET version=version+2", + "UPDATE ref_policy_budget SET watches=1.5", + "INSERT OR REPLACE INTO ref_policy_guards SELECT * FROM ref_policy_guards", + "INSERT OR REPLACE INTO ref_policy_watches SELECT * FROM ref_policy_watches", + "INSERT OR REPLACE INTO ref_policy_epoch SELECT * FROM ref_policy_epoch", + "INSERT OR REPLACE INTO ref_policy_budget SELECT * FROM ref_policy_budget", + ] { + let before = state(&fixture.handle).await?; + assert!(edit(&fixture, sql).await.is_err(), "accepted {sql}"); + assert_eq!(state(&fixture.handle).await?, before, "{sql}"); + } + pending.ready(&graph.prepared).await?; + edit(&fixture, "UPDATE ref_policy_guards SET valid=0").await?; + let before = state(&fixture.handle).await?; + assert!( + edit(&fixture, "UPDATE ref_policy_guards SET valid=1") + .await + .is_err() + ); + assert_eq!(state(&fixture.handle).await?, before); + assert!(pending.ready(&graph.prepared).await.is_err()); + close(fixture, graph).await +} diff --git a/crates/canopy-server/src/packs/publication/tests/ref_policy/pages.rs b/crates/canopy-server/src/packs/publication/tests/ref_policy/pages.rs new file mode 100644 index 0000000..939d018 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/ref_policy/pages.rs @@ -0,0 +1,184 @@ +use super::*; + +async fn denied(fixture: &Fixture, page: RefPolicyPage, reason: PreparationDenial) -> Result { + let before = state(&fixture.handle).await?; + let result = fixture + .client() + .command::(&fixture.target, identity()?, page) + .await; + assert!( + matches!(result,Err(InvocationError::Rejected(ref value)) + if value.output == RefPolicyReply::Denied(reason)), + "{result:?}" + ); + assert_eq!(state(&fixture.handle).await?, before); + Ok(()) +} + +#[tokio::test] +async fn pages_bound_bytes_and_updates_remap_bits_and_replay_only_contiguous_intent() -> Result { + let (fixture, graph) = rooted(ObjectFormat::Sha256).await?; + let changes = plan( + (0..134) + .map(|i| { + let name = if i < 5 { + format!("refs/tags/{}-{i}", "x".repeat(65_520)) + } else { + format!("refs/tags/tag-{i:04}") + }; + update( + &name, + (i % 2 == 1).then_some((graph.blob, 1)), + Some(graph.tip), + ) + }) + .collect(), + ); + let pending = graph + .prepared + .ref_policy_preparation(changes, graph.root.path(), graph.budget.clone(), limits()) + .await?; + let first = pending.page(&graph.prepared, 0).await?; + assert_eq!(first.proof.plan.updates.len(), 3); + let second = pending.page(&graph.prepared, 3).await?; + assert_eq!(second.proof.plan.updates.len(), 128); + let third = pending.page(&graph.prepared, 131).await?; + assert_eq!(third.proof.plan.updates.len(), 3); + for page in [&first, &second, &third] { + let mut e = BoundedEncoder::new(REF_POLICY_PAGE_BYTES)?; + page.encode(&mut e)?; + let bytes = e.finish(); + let mut d = BoundedDecoder::new(&bytes, REF_POLICY_PAGE_BYTES)?; + assert_eq!(RefPolicyPage::decode(&mut d)?, *page); + d.finish()?; + for i in 0..page.proof.plan.updates.len() { + assert_eq!( + super::super::super::ref_proof::proven(&page.proof.ancestry, i), + (page.offset as usize + i).is_multiple_of(2) + ); + } + } + denied(&fixture, second.clone(), PreparationDenial::Missing).await?; + let mutation = identity()?; + let receipt = fixture + .client() + .command::(&fixture.target, mutation, first.clone()) + .await?; + assert_eq!( + receipt.output, + RefPolicyReply::Registered(RefPolicyProgress { + next: 3, + total: 134, + valid: true, + }) + ); + let same = fixture + .client() + .command::(&fixture.target, mutation, first.clone()) + .await?; + assert_eq!( + (receipt.receipt, receipt.output), + (same.receipt, same.output) + ); + assert!(pending.ready(&graph.prepared).await.is_err()); + denied( + &fixture, + pending.page(&graph.prepared, 2).await?, + PreparationDenial::Conflict, + ) + .await?; + let mut altered = second.clone(); + altered.offset = 4; + denied(&fixture, altered, PreparationDenial::Unauthorized).await?; + for page in [second, third] { + fixture + .client() + .command::(&fixture.target, identity()?, page) + .await?; + } + let before = state(&fixture.handle).await?; + let replay = fixture + .client() + .command::(&fixture.target, identity()?, first) + .await?; + assert!(matches!(replay.output,RefPolicyReply::Registered(progress) if progress.ready())); + assert_eq!(state(&fixture.handle).await?, before); + pending.ready(&graph.prepared).await?; + close(fixture, graph).await +} + +#[tokio::test] +async fn late_cursor_failure_rolls_back_guard_watches_budget_and_then_exact_retry_succeeds() +-> Result { + let (fixture, graph) = rooted(ObjectFormat::Sha1).await?; + protect(&fixture, &graph).await?; + let pending = graph + .prepared + .ref_policy_preparation( + plan(vec![update("refs/heads/main", None, Some(graph.tip))]), + graph.root.path(), + graph.budget.clone(), + limits(), + ) + .await?; + let page = pending.page(&graph.prepared, 0).await?; + edit(&fixture,"CREATE TRIGGER fail_guard_cursor BEFORE UPDATE OF next ON ref_policy_guards BEGIN SELECT RAISE(ABORT,'late guard cursor fault'); END").await?; + let before = state(&fixture.handle).await?; + let mutation = identity()?; + assert!( + fixture + .client() + .command::(&fixture.target, mutation, page.clone()) + .await + .is_err() + ); + assert_eq!(state(&fixture.handle).await?, before); + edit(&fixture, "DROP TRIGGER fail_guard_cursor").await?; + let result = fixture + .client() + .command::(&fixture.target, mutation, page.clone()) + .await?; + assert!(matches!(result.output,RefPolicyReply::Registered(progress) if progress.ready())); + let before = state(&fixture.handle).await?; + fixture + .client() + .command::(&fixture.target, identity()?, page) + .await?; + assert_eq!(state(&fixture.handle).await?, before); + close(fixture, graph).await +} + +#[tokio::test] +async fn watch_and_guard_capacity_refusals_precede_every_write() -> Result { + let (fixture, graph) = rooted(ObjectFormat::Sha256).await?; + protect(&fixture, &graph).await?; + let pending = graph + .prepared + .ref_policy_preparation( + plan(vec![update("refs/heads/main", None, Some(graph.tip))]), + graph.root.path(), + graph.budget.clone(), + limits(), + ) + .await?; + let page = pending.page(&graph.prepared, 0).await?; + edit( + &fixture, + &format!("UPDATE ref_policy_budget SET watches={MAX_REF_POLICY_WATCHES}"), + ) + .await?; + denied(&fixture, page.clone(), PreparationDenial::Capacity).await?; + edit(&fixture, "UPDATE ref_policy_budget SET watches=0").await?; + let token = page.proof.certificate.data()?.token; + let mut e = BoundedEncoder::new(256)?; + token.encode(&mut e)?; + edit(&fixture,&format!("WITH RECURSIVE seq(x) AS (VALUES(1) UNION ALL SELECT x+1 FROM seq WHERE x<{MAX_REF_POLICY_GUARDS}) INSERT INTO ref_policy_guards SELECT CAST(printf('%016d',x) AS BLOB),zeroblob(32),X'{}',0,1,0,0 FROM seq",hex::encode(e.finish()))).await?; + denied(&fixture, page.clone(), PreparationDenial::Capacity).await?; + edit(&fixture, "DELETE FROM ref_policy_guards").await?; + fixture + .client() + .command::(&fixture.target, identity()?, page) + .await?; + pending.ready(&graph.prepared).await?; + close(fixture, graph).await +} diff --git a/crates/canopy-server/src/packs/publication/tests/ref_snapshot.rs b/crates/canopy-server/src/packs/publication/tests/ref_snapshot.rs new file mode 100644 index 0000000..7ef315b --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/ref_snapshot.rs @@ -0,0 +1,275 @@ +use super::publishing::{assembled, plan, state, update}; +use super::*; +use crate::RefExpectation; +use crate::packs::ref_state::{RefStateIndex, RefStateSnapshot, RefStateSnapshotRoot}; + +#[tokio::test] +async fn query_derived_ref_preparation_retains_exact_base_and_canonical_plan() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let graph = assembled(&fixture, [201; 16], 3).await?; + assert!(matches!( + graph + .prepared + .prepare_ref_snapshot(&plan(vec![update( + "refs/heads/main", + None, + Some(graph.initial) + )])) + .await, + Err(RefSnapshotPreparationError::Unavailable) + )); + let index = RefStateIndex::new(Arc::clone(&graph.store), format); + let initial = index + .prepare( + None, + graph.prepared.token().artifact_operation, + &plan(vec![ + update("refs/heads/main", None, Some(graph.initial)), + update("refs/tags/old", None, Some(graph.initial)), + ]), + ) + .await?; + let old = RefStateSnapshotRoot::upload( + &graph.store, + graph.prepared.token().artifact_operation, + RefStateSnapshot { + repository: fixture.repository, + format, + generation: 1, + default_branch: "refs/heads/main".into(), + root: Some(initial.root()), + }, + ) + .await?; + fixture + .install_generation(1, graph.prepared.catalog(), Some(old)) + .await?; + let prepared = graph.prepared.reconcile().await?; + assert_eq!(prepared.base().refs, Some(old)); + let changes = plan(vec![ + update("refs/tags/old", Some((graph.initial, 1)), None), + update("refs/heads/main", Some((graph.initial, 1)), Some(graph.tip)), + update("refs/heads/new", None, Some(graph.other)), + ]); + fn send(value: T) -> T { + value + } + let transition = send(prepared.prepare_ref_snapshot(&changes)).await?; + assert_eq!(transition.base(), prepared.base()); + assert_eq!( + transition.plan_digest(), + super::super::ref_proof::plan_digest(&changes)? + ); + let saved = transition.snapshot().read(&graph.store).await?; + assert_eq!( + (saved.generation, saved.format, saved.repository), + (2, format, fixture.repository) + ); + assert_eq!(saved.default_branch, "refs/heads/main"); + assert_eq!( + index.read(saved.root.clone(), "refs/heads/main").await?, + Some(RefExpectation { + oid: Some(graph.tip), + version: 2 + }) + ); + assert_eq!( + index.read(saved.root.clone(), "refs/tags/old").await?, + Some(RefExpectation { + oid: None, + version: 2 + }) + ); + assert_eq!( + index + .read(old.read(&graph.store).await?.root, "refs/heads/main") + .await?, + Some(RefExpectation { + oid: Some(graph.initial), + version: 1 + }) + ); + let retry = prepared.prepare_ref_snapshot(&changes).await?; + assert_eq!(transition.snapshot(), retry.snapshot()); + let before = state(&fixture.handle).await?; + let inline = Box::pin(prepared.ref_proof( + plan(vec![update("refs/heads/untracked", None, Some(graph.tip))]), + graph.root.path(), + graph.budget.clone(), + crate::packs::metadata::tests::limits(), + )) + .await?; + assert_eq!(inline.certificate.data()?.base.refs, Some(old)); + let reply = fixture + .client() + .command::(&fixture.target, identity()?, inline) + .await; + assert!( + matches!(reply, Err(InvocationError::Rejected(ref value)) if value.output==PublicationReply::Denied(PreparationDenial::Conflict)) + ); + assert_eq!(state(&fixture.handle).await?, before); + + fixture + .install_generation(2, prepared.catalog(), Some(transition.snapshot())) + .await?; + let current = prepared.reconcile().await?; + assert_eq!(current.base().refs, Some(transition.snapshot())); + assert!(matches!( + current.prepare_ref_snapshot(&changes).await, + Err(RefSnapshotPreparationError::State( + crate::packs::ref_state::RefStateError::Changed + )) + )); + let mut denied = plan(vec![update("refs/heads/actor", None, Some(graph.tip))]); + denied.actor = "outsider".into(); + assert!(matches!( + current.prepare_ref_snapshot(&denied).await, + Err(RefSnapshotPreparationError::Context) + )); + } + Ok(()) +} + +#[tokio::test] +async fn selected_ref_metadata_rejects_cross_format_future_generation_and_absent_artifacts() +-> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let graph = assembled(&fixture, [202; 16], 0).await?; + for (n, format, ref_generation, missing) in [ + (1, ObjectFormat::Sha256, 1, false), + (2, ObjectFormat::Sha1, 3, false), + (3, ObjectFormat::Sha1, 1, true), + ] { + let root = RefStateSnapshotRoot::upload( + &graph.store, + graph.prepared.token().artifact_operation, + RefStateSnapshot { + repository: fixture.repository, + format, + generation: ref_generation, + default_branch: "refs/heads/main".into(), + root: None, + }, + ) + .await?; + let root = if missing { + // A syntactically valid public descriptor is still not proof that + // its selected manifest/body exists in the trusted store. + let mut e = BoundedEncoder::new(128)?; + root.encode(&mut e)?; + let mut bytes = e.finish(); + let last = bytes.len() - 1; + bytes[last] ^= 1; + let mut d = BoundedDecoder::new(&bytes, 128)?; + let root = RefStateSnapshotRoot::decode(&mut d)?; + d.finish()?; + root + } else { + root + }; + fixture + .install_generation(n, graph.prepared.catalog(), Some(root)) + .await?; + let prepared = graph.prepared.reconcile().await?; + let result = prepared + .prepare_ref_snapshot(&plan(vec![update( + "refs/heads/main", + None, + Some(graph.initial), + )])) + .await; + if missing { + assert!(matches!( + result, + Err(RefSnapshotPreparationError::Snapshot(_)) + )); + } else { + assert!(matches!(result, Err(RefSnapshotPreparationError::Context))); + } + } + Ok(()) +} + +#[tokio::test] +async fn ref_root_is_in_joint_fact_codec_certificate_and_retained_generation() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let graph = assembled(&fixture, [203; 16], 0).await?; + let root = RefStateSnapshotRoot::upload( + &graph.store, + graph.prepared.token().artifact_operation, + RefStateSnapshot { + repository: fixture.repository, + format: fixture.format, + generation: 0, + default_branch: "refs/heads/main".into(), + root: None, + }, + ) + .await?; + fixture + .install_generation(1, graph.prepared.catalog(), Some(root)) + .await?; + let prepared = graph.prepared.reconcile().await?; + let base = prepared.base(); + let mut e = BoundedEncoder::new(512)?; + base.encode(&mut e)?; + let bytes = e.finish(); + let mut d = BoundedDecoder::new(&bytes, 512)?; + assert_eq!(GenerationFact::decode(&mut d)?, base); + d.finish()?; + for at in 0..bytes.len() { + let mut d = BoundedDecoder::new(&bytes[..at], 512)?; + assert!(GenerationFact::decode(&mut d).is_err()); + } + let certificate = prepared.issue_certificate(None, None).await?; + assert!(certificate.bytes()?.len() <= CERTIFICATE_BYTES as usize); + assert_eq!(certificate.data()?.base.refs, Some(root)); + let mut maximal = certificate.data()?; + maximal.actor = "a".repeat(64); + maximal.retention_floor = maximal.base.generation; + maximal.retention_certificate = maximal.base.certificate; + maximal.object_count = i64::MAX as u64; + maximal.edge_count = i64::MAX as u64; + maximal.input_count = i64::MAX as u64; + maximal.input_checkpoint_digest = Some([51; 32]); + maximal.refs_digest = Some([52; 32]); + maximal.completion_digest = Some([53; 32]); + maximal.token.owner.epoch = u64::MAX; + maximal.token.attempt = i64::MAX as u64; + assert!( + CatalogCertificate::seal(&maximal, &[16; 32])? + .bytes()? + .len() + <= CERTIFICATE_BYTES as usize + ); + let mut altered = certificate.clone(); + let mut data = certificate.data()?; + data.base.refs = None; + let mut e = BoundedEncoder::new(960)?; + data.encode(&mut e)?; + altered.0.body = e.finish(); + assert!(!altered.authenticated(&[16; 32])); + fixture + .install_generation(2, prepared.catalog(), Some(root)) + .await?; + let reply = fixture + .client() + .command::( + &fixture.target, + identity()?, + MaintenanceRequest { + repository: fixture.repository, + actor: "owner".into(), + owner: graph.prepared.token().owner, + }, + ) + .await?; + assert_eq!(reply.output, 0); // original floor zero retains both joint roots + let mut invalid = base; + invalid.generation = 0; + invalid.catalog = None; + invalid.certificate = None; + assert!(invalid.validate().is_err()); // empty genesis cannot carry a root + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/refs.rs b/crates/canopy-server/src/packs/publication/tests/refs.rs new file mode 100644 index 0000000..3fc68c4 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/refs.rs @@ -0,0 +1,225 @@ +//! Fresh-schema checks of the shared ref boundary. Apply is trusted fixture +//! injection only; the real publisher must also verify catalog and policy facts. +use super::*; +use crate::{PushPlan, RefExpectation, RefUpdate}; + +pub(super) struct FixtureRefs; +pub(super) struct FixturePlan { + plan: PushPlan, + apply: bool, +} +impl WireValue for FixturePlan { + fn encode(&self, e: &mut BoundedEncoder) -> std::result::Result<(), CodecError> { + self.plan.encode(e)?; + e.write_bool(self.apply) + } + fn decode(d: &mut BoundedDecoder<'_>) -> std::result::Result { + Ok(Self { + plan: PushPlan::decode(d)?, + apply: d.read_bool()?, + }) + } +} +impl Command for FixtureRefs { + const MODULE: &'static str = "repository"; + const ID: u32 = 90; + const CODEC_VERSION: u32 = 1; + type Input = FixturePlan; + type Output = bool; + fn execute( + context: &mut CommandContext<'_, '_>, + FixturePlan { plan, apply }: Self::Input, + ) -> cellule_runtime::Result> { + let Some(validated) = crate::refs::validate_refs(context, &plan)? else { + return Ok(CommandResult::Rejected(false)); + }; + if apply { + validated.apply(context)?; + } + Ok(CommandResult::Success(true)) + } +} +fn oid(byte: u8) -> crate::ObjectId { + crate::ObjectId::Sha1([byte; 20]) +} +fn update(name: &str, expected: Option<(Option, i64)>, new: Option) -> RefUpdate { + RefUpdate { + name: name.into(), + expected: expected.map(|(id, version)| RefExpectation { + oid: id.map(oid), + version, + }), + new_oid: new.map(oid), + } +} +fn plan(updates: Vec) -> PushPlan { + PushPlan { + actor: "owner".into(), + updates, + } +} +async fn apply(fixture: &Fixture, plan: PushPlan) -> Result { + match fixture + .client() + .command::( + &fixture.target, + identity()?, + FixturePlan { plan, apply: true }, + ) + .await + { + Ok(committed) => Ok(committed.output), + Err(InvocationError::Rejected(value)) => Ok(value.output), + Err(error) => Err(error.into()), + } +} +async fn state(handle: &CellHandle) -> Result> { + Ok(handle + .query(0, 64 << 10, |connection| { + let generation: i64 = connection.query_row( + "SELECT generation FROM ref_generation WHERE singleton=1", + [], + |r| r.get(0), + )?; + let mut stmt = connection.prepare("SELECT name,oid,version FROM refs ORDER BY name")?; + let refs = stmt + .query_map([], |r| { + Ok(( + r.get::<_, String>(0)?, + r.get::<_, Option>>(1)?, + r.get::<_, i64>(2)?, + )) + })? + .collect::>>()?; + serde_json::to_vec(&(generation, refs)) + .map_err(|_| cellule_runtime::Error::Command("fixture refs encoding")) + }) + .await?) +} + +#[tokio::test] +async fn fresh_ref_validation_reads_no_graph_tables_and_denies_before_writes() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let before = state(&fixture.handle).await?; + assert!( + fixture + .client() + .command::( + &fixture.target, + identity()?, + FixturePlan { + plan: plan(vec![update("refs/heads/main", None, Some(7))]), + apply: false + } + ) + .await? + .output + ); + assert_eq!(state(&fixture.handle).await?, before); + for updates in [ + vec![ + update("refs/heads/team", None, Some(7)), + update("refs/heads/team/topic", None, Some(7)), + ], + vec![ + update("refs/heads/main", None, Some(7)), + update("refs/heads/main", None, Some(8)), + ], + vec![update("refs/canopy/candidate", None, Some(7))], + vec![update("refs/heads/../main", None, Some(7))], + vec![update("refs/heads/main", None, None)], + ] { + assert!(!apply(&fixture, plan(updates)).await?); + assert_eq!(state(&fixture.handle).await?, before); + } + let mut foreign = plan(vec![update("refs/heads/main", None, Some(7))]); + foreign.updates[0].new_oid = Some(crate::ObjectId::Sha256([7; 32])); + assert!(!apply(&fixture, foreign).await?); + let mut unauthorized = plan(vec![update("refs/heads/main", None, Some(7))]); + unauthorized.actor = "outsider".into(); + assert!(!apply(&fixture, unauthorized).await?); + assert_eq!(state(&fixture.handle).await?, before); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn fresh_ref_boundary_retains_tombstones_and_atomic_namespace_changes() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + assert!( + apply( + &fixture, + plan(vec![update("refs/heads/team", None, Some(7))]) + ) + .await? + ); + assert!( + apply( + &fixture, + plan(vec![ + update("refs/heads/team", Some((Some(7), 1)), None), + update("refs/heads/team/topic", None, Some(8)) + ]) + ) + .await? + ); + let before = state(&fixture.handle).await?; + // A deleted ancestor can only be recreated with its retained version, and + // its currently live descendant must be deleted in that same transaction. + assert!( + !apply( + &fixture, + plan(vec![update("refs/heads/team", None, Some(7))]) + ) + .await? + ); + assert!( + !apply( + &fixture, + plan(vec![update("refs/heads/team", Some((None, 2)), Some(7))]) + ) + .await? + ); + assert_eq!(state(&fixture.handle).await?, before); + assert!( + apply( + &fixture, + plan(vec![ + update("refs/heads/team", Some((None, 2)), Some(7)), + update("refs/heads/team/topic", Some((Some(8), 1)), None) + ]) + ) + .await? + ); + let after = state(&fixture.handle).await?; + assert!( + !apply( + &fixture, + plan(vec![update("refs/heads/team", Some((Some(7), 1)), Some(9))]) + ) + .await? + ); + assert_eq!(state(&fixture.handle).await?, after); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn fresh_namespace_validation_checks_descendants_beyond_one_page() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let updates: Vec<_> = (0..300) + .map(|n| update(&format!("refs/heads/team/{n:04}"), None, Some(7))) + .collect(); + assert!(apply(&fixture, plan(updates)).await?); + let before = state(&fixture.handle).await?; + let mut deletes: Vec<_> = (0..299) + .map(|n| update(&format!("refs/heads/team/{n:04}"), Some((Some(7), 1)), None)) + .collect(); + deletes.push(update("refs/heads/team", None, Some(8))); + assert!(!apply(&fixture, plan(deletes.clone())).await?); + assert_eq!(state(&fixture.handle).await?, before); + deletes.push(update("refs/heads/team/0299", Some((Some(7), 1)), None)); + assert!(apply(&fixture, plan(deletes)).await?); + fixture.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/root_completion.rs b/crates/canopy-server/src/packs/publication/tests/root_completion.rs new file mode 100644 index 0000000..8d88c2d --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/root_completion.rs @@ -0,0 +1,136 @@ +use super::super::root_completion::tests::{audit_native, change_namespace, response}; +use super::publishing::{edit, state}; +use super::*; +use crate::packs::metadata::tests::limits; +use canopy_object_storage::artifact::ArtifactStore; +use cellule_ltx::DiskBudget; +use std::path::Path; + +pub(super) async fn qualify( + fixture: &Fixture, + prepared: &PreparedCatalog, + store: &Arc, + request: PushCompletionRequest, + directory: &Path, + budget: DiskBudget, +) -> Result { + let plan = request.plan.unwrap(); + assert!(prepared.base().refs.is_some(), "prepared rooted base"); + let (checkpoint, _, _, _) = prepared.base.session.push_checkpoint().await?; + let mut encoded = BoundedEncoder::new(CERTIFICATE_BYTES)?; + checkpoint.encode(&mut encoded)?; + assert_eq!( + prepared.input_checkpoint_digest, + Some(*blake3::hash(&encoded.finish()).as_bytes()), + "prepared/checkpoint custody digest" + ); + let pending = prepared + .ref_policy_preparation(plan.clone(), directory, budget.clone(), limits()) + .await?; + let mut start = 0; + while start < pending.plan().updates.len() { + let page = pending.page(prepared, start).await?; + start += page.proof.plan.updates.len(); + fixture + .client() + .command::(&fixture.target, identity()?, page) + .await?; + } + let guard = pending.ready(prepared).await?; + let before = state(&fixture.handle).await?; + let completion = prepared + .root_push_completion(&guard, directory, budget.clone(), limits(), None) + .await + .map_err(|error| format!("root completion preparation: {error:?}"))?; + assert_eq!(completion.outcomes.ref_generation, 1); + assert_eq!( + response(completion.outcomes.native, store).await?, + request.response + ); + assert_eq!( + response(completion.outcomes.rejected, store).await?, + crate::push::report::rejected_report(&request.response, crate::push::report::REJECTED)? + ); + assert_eq!( + response(completion.outcomes.replayed, store).await?, + crate::push::report::rejected_report( + &request.response, + "Canopy signed push certificate was already used" + )? + ); + for root in [ + completion.outcomes.native, + completion.outcomes.rejected, + completion.outcomes.replayed, + ] { + assert_eq!(root.operation(), prepared.token().artifact_operation); + assert!(root.artifact().size < 1024); + let native = audit_native(root, store).await?; + let (proof, _, _, _) = prepared.base.session.push_checkpoint().await?; + assert_eq!(Some(native), proof.native_result()?); + } + let mut e = BoundedEncoder::new(ROOT_COMPLETION_BYTES)?; + completion.encode(&mut e)?; + let bytes = e.finish(); + assert!(bytes.len() < 2048); + let mut d = BoundedDecoder::new(&bytes, ROOT_COMPLETION_BYTES)?; + assert_eq!(RootPushCompletion::decode(&mut d)?, completion); + d.finish()?; + for choice in 0..8 { + let mut changed = completion.clone(); + match choice { + 0 => changed.outcomes.response_id = *uuid::Uuid::new_v4().as_bytes(), + 1 => changed.outcomes.ref_generation += 1, + 2 => std::mem::swap(&mut changed.outcomes.native, &mut changed.outcomes.rejected), + 3 => std::mem::swap( + &mut changed.outcomes.rejected, + &mut changed.outcomes.replayed, + ), + 4 => { + changed.outcomes.signed = Some(RootSignedPushFact { + digest: [1; 32], + key: "key".into(), + size: 1, + }) + } + 5 => changed.proof.guard.plan_digest[0] ^= 1, + 6 => changed.proof.snapshot = prepared.base().refs.unwrap(), + _ => changed.outcomes.native = change_namespace(changed.outcomes.native), + } + assert!( + changed + .encode(&mut BoundedEncoder::new(ROOT_COMPLETION_BYTES)?) + .is_err(), + "choice {choice}" + ); + } + // The existing inline publisher must not accept a root-completion purpose. + let ancestry = vec![0; plan.updates.len().div_ceil(8)]; + let denied = fixture + .client() + .command::( + &fixture.target, + identity()?, + RefPublicationProof { + certificate: completion.proof.certificate, + plan, + ancestry, + }, + ) + .await; + assert!(matches!(denied, Err(InvocationError::Rejected(value)) + if value.output == PublicationReply::Denied(PreparationDenial::Unauthorized))); + assert_eq!(state(&fixture.handle).await?, before); + edit( + fixture, + "INSERT INTO branch_rules VALUES('refs/heads/main',1,1,0,1,0,0)", + ) + .await?; + assert!( + prepared + .root_push_completion(&guard, directory, budget, limits(), None) + .await + .is_err() + ); + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/staging.rs b/crates/canopy-server/src/packs/publication/tests/staging.rs new file mode 100644 index 0000000..1bca2e2 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/staging.rs @@ -0,0 +1,825 @@ +use super::*; + +fn staging(reply: StagingReply) -> Result { + match reply { + StagingReply::Granted(lease) => Ok(*lease), + StagingReply::Denied(reason) => Err(format!("denied: {reason:?}").into()), + } +} +fn denied( + result: std::result::Result< + cellule_runtime::Committed, + InvocationError, + >, + reason: PreparationDenial, +) { + assert!( + matches!(result, Err(InvocationError::Rejected(ref value)) if value.output == StagingReply::Denied(reason)) + ); +} +async fn start(fixture: &Fixture, operation: [u8; 16]) -> Result { + staging( + fixture + .client() + .command::(&fixture.target, identity()?, fixture.begin(operation)) + .await? + .output, + ) +} + +#[test] +fn staged_pin_normalized_binding_has_no_nullable_foreign_key_escape() -> Result { + let mut connection = rusqlite::Connection::open_in_memory()?; + connection.execute_batch("PRAGMA foreign_keys=ON")?; + connection.execute_batch(SCHEMA)?; + connection.execute( + "INSERT INTO catalog_generations(generation,catalog,certificate) VALUES(1,x'01',zeroblob(32))", + [], + )?; + connection.execute("INSERT INTO catalog_leases(incarnation,admission_sequence,operation,owner_epoch,artifact_operation,generation,expires_at_ms) VALUES(zeroblob(16),1,zeroblob(16),x'0000000000000001',?1,NULL,100)", [artifact_number(1).as_slice()])?; + for (logical, epoch, sequence, namespace, floor, expires) in [ + ([1u8; 16], 1u64, 1, artifact_number(1), None, 100), + ([0u8; 16], 2u64, 1, artifact_number(1), None, 100), + ([0u8; 16], 1u64, 2, artifact_number(1), None, 100), + ([0u8; 16], 1u64, 1, artifact_number(2), None, 100), + ([0u8; 16], 1u64, 1, artifact_number(1), Some(0), 100), + ([0u8; 16], 1u64, 1, artifact_number(1), None, 101), + ] { + let tx = connection.transaction()?; + tx.execute("INSERT INTO catalog_operations(id,actor,request_digest,incarnation,owner_epoch,admission_sequence,artifact_operation,generation,expires_at_ms) VALUES(?1,'owner',zeroblob(32),zeroblob(16),?2,?3,?4,?5,?6)", rusqlite::params![logical.as_slice(),epoch.to_be_bytes().as_slice(),sequence,namespace.as_slice(),floor,expires])?; + assert!(tx.commit().is_err()); + } + connection.execute("INSERT INTO catalog_operations(id,actor,request_digest,incarnation,owner_epoch,admission_sequence,artifact_operation,generation,expires_at_ms) VALUES(zeroblob(16),'owner',zeroblob(32),zeroblob(16),x'0000000000000001',1,?1,NULL,100)", [artifact_number(1).as_slice()])?; + assert!( + connection + .execute( + "UPDATE catalog_leases SET attestation=x'01',attestation_digest=zeroblob(32)", + [] + ) + .is_err() + ); + assert!( + connection + .execute( + "UPDATE catalog_operations SET attestation=x'01',attestation_digest=zeroblob(32)", + [] + ) + .is_err() + ); + let tx = connection.transaction()?; + tx.execute("UPDATE catalog_leases SET generation=1", [])?; + assert!(tx.commit().is_err()); // both sides must bind in one transaction + assert_eq!( + connection.query_row("SELECT binding_generation FROM catalog_leases", [], |r| r + .get::<_, i64>( + 0 + ))?, + -1 + ); + let tx = connection.transaction()?; + tx.execute("UPDATE catalog_leases SET generation=1", [])?; + tx.execute("UPDATE catalog_operations SET generation=1", [])?; + tx.commit()?; + for sql in [ + "UPDATE catalog_leases SET generation=NULL", + "UPDATE catalog_leases SET generation=0", + "UPDATE catalog_leases SET artifact_operation=randomblob(16)", + ] { + assert!(connection.execute(sql, []).is_err()); + } + assert!( + connection + .execute("DELETE FROM catalog_generations WHERE generation=1", []) + .is_err() + ); + Ok(()) +} + +#[tokio::test] +async fn staging_exact_replay_late_bind_and_phase_separation_preserve_one_namespace() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let client = fixture.client(); + let mutation = identity()?; + let input = fixture.begin([201; 16]); + let first = client + .command::(&fixture.target, mutation, input.clone()) + .await?; + let staged = staging(first.output.clone())?; + let replay = client + .command::(&fixture.target, mutation, input.clone()) + .await?; + assert_eq!(replay.receipt, first.receipt); + assert_eq!(replay.output, first.output); + assert_eq!( + staging( + client + .command::(&fixture.target, identity()?, input.clone()) + .await? + .output + )? + .token, + staged.token + ); + assert_eq!(fixture.counts().await?, (1, 1)); + assert!( + client + .query::(&fixture.target, None, check(staged.token)) + .await? + .output + .is_none() + ); + assert!( + client + .query::(&fixture.target, None, check(staged.token)) + .await? + .output + .is_none() + ); + for reply in [ + client + .command::(&fixture.target, identity()?, input.clone()) + .await, + client + .command::(&fixture.target, identity()?, request(staged.token)) + .await, + client + .command::(&fixture.target, identity()?, request(staged.token)) + .await, + ] { + rejected(reply, PreparationDenial::Conflict); + } + let renewed = staging( + client + .command::(&fixture.target, identity()?, request(staged.token)) + .await? + .output, + )?; + assert_eq!(renewed.token, staged.token); + assert!(renewed.expires_at_ms >= staged.expires_at_ms); + fixture.install_empty_root(1).await?; + fixture.install_empty_root(2).await?; + assert_eq!(super::frontier::reap(&fixture).await?, 1); + assert_eq!( + super::frontier::facts(&fixture).await?, + super::frontier::expected(&[0, 2]) + ); + let bind_identity = identity()?; + let bound = client + .command::(&fixture.target, bind_identity, check(staged.token)) + .await?; + let active = lease(bound.output.clone())?; + assert_eq!(active.token, staged.token); + assert_eq!(active.base.generation, 2); + assert_eq!(active.expires_at_ms, renewed.expires_at_ms); + fixture.install_empty_root(3).await?; + let replay = client + .command::(&fixture.target, bind_identity, check(staged.token)) + .await?; + assert_eq!(replay.receipt, bound.receipt); + assert_eq!(replay.output, bound.output); + let again = lease( + client + .command::(&fixture.target, identity()?, check(staged.token)) + .await? + .output, + )?; + assert_eq!(again.base, active.base); + assert_eq!(again.expires_at_ms, active.expires_at_ms); + assert!( + client + .query::(&fixture.target, None, check(staged.token)) + .await? + .output + .is_none() + ); + for reply in [ + client + .command::(&fixture.target, identity()?, input) + .await, + client + .command::(&fixture.target, identity()?, request(staged.token)) + .await, + client + .command::(&fixture.target, identity()?, request(staged.token)) + .await, + ] { + denied(reply, PreparationDenial::Conflict); + } + assert_eq!(super::frontier::reap(&fixture).await?, 0); + assert_eq!(fixture.counts().await?, (1, 1)); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn staged_input_does_not_accumulate_generation_history_beyond_fact_capacity() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let staged = start(&fixture, [202; 16]).await?; + let catalog = fixture.install_empty_root(1).await?; + let mut e = BoundedEncoder::new(256)?; + catalog.encode(&mut e)?; + let bytes = e.finish(); + // Trusted catalog injection exercises retention, not publication throughput. + // Use bounded batches; the production reaper runs between every batch. + for first in (2..=10_241).step_by(512) { + let bytes = bytes.clone(); + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([202; 32]), + sql::now(0)?, + 1, + 0, + move |tx| { + let mut insert = + tx.prepare("INSERT INTO catalog_generations(generation,catalog,certificate) VALUES(?1,?2,?3)")?; + for n in first..first + 512 { + insert.execute(rusqlite::params![n, bytes, [42u8; 32].as_slice()])?; + } + drop(insert); + tx.execute("UPDATE catalog_state SET generation=?1", [first + 511])?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + assert_eq!(super::frontier::reap(&fixture).await?, 512); + assert_eq!( + super::frontier::facts(&fixture).await?, + super::frontier::expected(&[0, first as u64 + 511]) + ); + let renewed = staging( + fixture + .client() + .command::(&fixture.target, identity()?, request(staged.token)) + .await? + .output, + )?; + assert_eq!(renewed.token, staged.token); + } + let bound = lease( + fixture + .client() + .command::(&fixture.target, identity()?, check(staged.token)) + .await? + .output, + )?; + assert_eq!(bound.base.generation, 10_241); + assert_eq!(fixture.counts().await?, (1, 1)); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn staging_claim_abort_and_stale_owner_do_not_reuse_or_drop_old_input_custody() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let old = start(&fixture, [203; 16]).await?; + let client = fixture.client(); + let mut wrong = check(old.token); + wrong.token.owner.epoch += 1; + rejected( + client + .command::(&fixture.target, identity()?, wrong.clone()) + .await, + PreparationDenial::Stale, + ); + denied( + client + .command::( + &fixture.target, + identity()?, + LeaseRequest { + check: wrong, + lease_ms: DEFAULT_LEASE_MS, + }, + ) + .await, + PreparationDenial::Stale, + ); + let next = staging( + client + .command::(&fixture.target, identity()?, request(old.token)) + .await? + .output, + )?; + assert_ne!(next.token.artifact_operation, old.token.artifact_operation); + assert_eq!(next.token.artifact_operation, artifact_number(2)); + assert_eq!(fixture.counts().await?, (1, 2)); + rejected( + client + .command::(&fixture.target, identity()?, check(old.token)) + .await, + PreparationDenial::Stale, + ); + assert!( + client + .query::(&fixture.target, None, check(old.token)) + .await? + .output + .is_none() + ); + client + .command::(&fixture.target, identity()?, check(next.token)) + .await?; + assert_eq!(fixture.counts().await?, (0, 2)); + assert_eq!(super::frontier::reap(&fixture).await?, 0); + let recreated = start(&fixture, [203; 16]).await?; + assert_eq!(recreated.token.artifact_operation, artifact_number(3)); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn physical_inputs_verified_before_binding_feed_the_existing_catalog_proof() -> Result { + use crate::packs::{ + catalog::{CatalogFileLimits, CatalogFiles, CatalogIndexes}, + metadata::tests::limits, + verification::physical::tests::prepared_for_store, + }; + use canopy_object_storage::artifact::ArtifactStore; + use cellule_ltx::DiskBudget; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let staged = start(&fixture, [204; 16]).await?; + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new( + Arc::clone(&provider), + fixture.repository, + )); + let native = prepared_for_store( + format, + 32, + staged.token.artifact_operation, + provider, + Arc::clone(&store), + ) + .await?; + let indexes = Arc::new(CatalogIndexes::new(Arc::clone(&store), format)); + let files = Arc::new(CatalogFiles::new( + fixture.root.path(), + DiskBudget::new(64 << 20), + store, + format, + CatalogFileLimits::default(), + )?); + assert!(matches!( + PreparationBaseResolver::open( + fixture.client(), + fixture.target.clone(), + check(staged.token), + Arc::clone(&indexes), + Arc::clone(&files), + None + ) + .await, + Err(PreparationBaseError::Inactive) + )); + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(256 << 20); + let (witness, segments) = + super::prepare::physical(&native, root.path(), budget.clone()).await?; + assert_eq!(fixture.counts().await?, (1, 1)); + let bound = fixture + .client() + .command::(&fixture.target, identity()?, check(staged.token)) + .await?; + let active = lease(bound.output)?; + assert_eq!(active.token, staged.token); + let base = Arc::new( + PreparationBaseResolver::open( + fixture.client(), + fixture.target.clone(), + check(active.token), + indexes, + files, + Some(bound.receipt), + ) + .await?, + ); + let mut assembler = + CatalogPreparation::new(root.path(), budget.clone(), base, limits()).await?; + assembler.begin_pack(witness)?; + for segment in segments { + assembler.add_segment(segment).await?; + } + assembler.finish_pack().await?; + let proof = assembler.finish().await?; + assert_eq!(proof.token(), staged.token); + assert_eq!(proof.object_count(), native.fixture.objects.len() as u64); + let certificate = proof.certificate().await?; + assert_eq!(certificate.data()?.retention_floor, active.base.generation); + assert!(matches!( + proof.attest(identity()?).await?.output, + AttestationOutcome::Registered(_) + )); + drop(proof); + super::prepare::cleaned(root.path(), &budget).await?; + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn staging_takeover_restores_exact_outcomes_and_requires_a_new_creating_attempt() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let mutation = identity()?; + let input = fixture.begin([205; 16]); + let first = fixture + .client() + .command::(&fixture.target, mutation, input.clone()) + .await?; + let old = staging(first.output.clone())?; + fixture.handle.drain().await?; + fixture.runtime.shutdown().await?; + let session = SessionId::from_bytes([205; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, session)?; + let authority = CellAuthority::new(fixture.layout.clone()); + let idle = authority + .load(fixture.target.cell_id()) + .await? + .ok_or("idle")?; + let proof = CellCatalog::new(fixture.layout.clone(), fixture.target.tenant()) + .lookup(fixture.target.cell_id()) + .await? + .ok_or("proof")?; + let handle = runtime + .acquire_idle_restored( + proof, + fixture.replica.clone(), + authority, + idle, + fixture.root.path().join("staging-restored.sqlite"), + Owner { + session, + endpoint: "https://staging-successor.invalid".into(), + }, + ) + .await?; + let client = CellClient::local(Arc::clone(&fixture.registry), handle.clone()); + let replay = client + .command::(&fixture.target, mutation, input) + .await?; + assert_eq!(replay.receipt, first.receipt); + assert_eq!(replay.output, first.output); + denied( + client + .command::(&fixture.target, identity()?, request(old.token)) + .await, + PreparationDenial::Stale, + ); + rejected( + client + .command::(&fixture.target, identity()?, check(old.token)) + .await, + PreparationDenial::Stale, + ); + let claim_identity = identity()?; + let claimed = client + .command::(&fixture.target, claim_identity, request(old.token)) + .await?; + let next = staging(claimed.output.clone())?; + assert_eq!(next.token.owner, handle.owner_fence()); + assert_eq!(next.token.artifact_operation, artifact_number(2)); + assert!(next.token.attempt > old.token.attempt); + let replay = client + .command::(&fixture.target, claim_identity, request(old.token)) + .await?; + assert_eq!(replay.receipt, claimed.receipt); + assert_eq!(replay.output, claimed.output); + assert_eq!(counts(&handle).await?, (1, 2)); + assert!( + client + .query::(&fixture.target, None, check(old.token)) + .await? + .output + .is_none() + ); + let active = lease( + client + .command::(&fixture.target, identity()?, check(next.token)) + .await? + .output, + )?; + assert_eq!(active.token, next.token); + runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn staging_checks_current_access_exact_identity_and_expiry_before_late_binding() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([206; 32]), + sql::now(0)?, + 1, + 0, + |tx| { + tx.execute( + "INSERT INTO repository_members(account,role) VALUES('writer','write')", + [], + )?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + let client = fixture.client(); + let mut input = fixture.begin([206; 16]); + input.actor = "writer".into(); + let staged = staging( + client + .command::(&fixture.target, identity()?, input.clone()) + .await? + .output, + )?; + let valid = LeaseCheck { + token: staged.token, + actor: "writer".into(), + }; + let mut wrong_checks = Vec::new(); + for field in 0..4 { + let mut wrong = valid.clone(); + match field { + 0 => wrong.token.request_digest[0] ^= 1, + 1 => wrong.token.artifact_operation = artifact_number(2), + 2 => wrong.token.attempt += 1, + _ => wrong.actor = "owner".into(), + } + wrong_checks.push(wrong); + } + for wrong in wrong_checks { + assert!( + client + .query::(&fixture.target, None, wrong.clone()) + .await? + .output + .is_none() + ); + rejected( + client + .command::(&fixture.target, identity()?, wrong) + .await, + PreparationDenial::Stale, + ); + } + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([207; 32]), + sql::now(0)?, + 1, + 0, + |tx| { + tx.execute( + "UPDATE repository_members SET role='read' WHERE account='writer'", + [], + )?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + assert!( + client + .query::(&fixture.target, None, valid.clone()) + .await? + .output + .is_none() + ); + rejected( + client + .command::(&fixture.target, identity()?, valid.clone()) + .await, + PreparationDenial::Unauthorized, + ); + denied( + client + .command::( + &fixture.target, + identity()?, + LeaseRequest { + check: valid.clone(), + lease_ms: DEFAULT_LEASE_MS, + }, + ) + .await, + PreparationDenial::Unauthorized, + ); + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([208; 32]), + sql::now(0)?, + 1, + 0, + |tx| { + tx.execute( + "UPDATE repository_members SET role='write' WHERE account='writer'", + [], + )?; + tx.execute("UPDATE catalog_operations SET expires_at_ms=0", [])?; + tx.execute("UPDATE catalog_leases SET expires_at_ms=0", [])?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + denied( + client + .command::(&fixture.target, identity()?, input) + .await, + PreparationDenial::Expired, + ); + denied( + client + .command::( + &fixture.target, + identity()?, + LeaseRequest { + check: valid.clone(), + lease_ms: DEFAULT_LEASE_MS, + }, + ) + .await, + PreparationDenial::Expired, + ); + rejected( + client + .command::(&fixture.target, identity()?, valid.clone()) + .await, + PreparationDenial::Expired, + ); + assert!( + client + .query::(&fixture.target, None, valid) + .await? + .output + .is_none() + ); + assert_eq!(super::frontier::reap(&fixture).await?, 2); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn staging_shares_operation_and_pin_quotas_and_binds_without_allocating_another_pin() -> Result +{ + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let staged = start(&fixture, [209; 16]).await?; + let owner = fixture.handle.owner_fence(); + for first in (1..MAX_OPERATIONS).step_by(REAP_ROWS as usize) { + let last = (first + REAP_ROWS).min(MAX_OPERATIONS); + fixture.handle.execute(identity()?,Digest::from_bytes([209;32]),sql::now(0)?,1,0,move |tx| { + let mut pin = tx.prepare("INSERT INTO catalog_leases(incarnation,admission_sequence,operation,owner_epoch,artifact_operation,generation,expires_at_ms) VALUES(?1,?2,?3,?4,?5,?6,0)")?; + let mut operation = tx.prepare("INSERT INTO catalog_operations(id,actor,request_digest,incarnation,owner_epoch,admission_sequence,artifact_operation,generation,expires_at_ms) VALUES(?1,'owner',zeroblob(32),?2,?3,?4,?5,?6,0)")?; + for n in first..last { + let mut id = [0u8;16]; id[..8].copy_from_slice(&n.to_be_bytes()); + let seq = 1_000_000+n as i64; + let floor = if n%2==0 { Some(0) } else { None }; // both phases consume the same quotas + pin.execute(rusqlite::params![owner.incarnation.as_bytes().as_slice(),seq,id.as_slice(),owner.epoch.to_be_bytes().as_slice(),artifact_number(seq as u64).as_slice(),floor])?; + operation.execute(rusqlite::params![id.as_slice(),owner.incarnation.as_bytes().as_slice(),owner.epoch.to_be_bytes().as_slice(),seq,artifact_number(seq as u64).as_slice(),floor])?; + } + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success(Vec::new())) + }).await?; + } + let client = fixture.client(); + denied( + client + .command::(&fixture.target, identity()?, fixture.begin([210; 16])) + .await, + PreparationDenial::Capacity, + ); + rejected( + client + .command::(&fixture.target, identity()?, fixture.begin([211; 16])) + .await, + PreparationDenial::Capacity, + ); + fixture + .handle + .execute( + identity()?, + Digest::from_bytes([210; 32]), + sql::now(0)?, + 1, + 0, + move |tx| { + tx.execute( + "DELETE FROM catalog_operations WHERE id!=?1", + [staged.token.operation.as_slice()], + )?; + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success( + Vec::new(), + )) + }, + ) + .await?; + for first in (MAX_OPERATIONS..MAX_GENERATION_LEASES).step_by(REAP_ROWS as usize) { + let last = (first + REAP_ROWS).min(MAX_GENERATION_LEASES); + fixture.handle.execute(identity()?,Digest::from_bytes([211;32]),sql::now(0)?,1,0,move |tx| { + let mut pin = tx.prepare("INSERT INTO catalog_leases(incarnation,admission_sequence,operation,owner_epoch,artifact_operation,generation,expires_at_ms) VALUES(?1,?2,?3,?4,?5,NULL,0)")?; + for n in first..last { + let mut id = [0u8;16]; id[..8].copy_from_slice(&n.to_be_bytes()); + let seq = 1_000_000+n as i64; + pin.execute(rusqlite::params![owner.incarnation.as_bytes().as_slice(),seq,id.as_slice(),owner.epoch.to_be_bytes().as_slice(),artifact_number(seq as u64).as_slice()])?; + } + Ok(cellule_runtime::cell::executor::HandlerOutcome::Success(Vec::new())) + }).await?; + } + assert_eq!(fixture.counts().await?, (1, MAX_GENERATION_LEASES)); + denied( + client + .command::(&fixture.target, identity()?, fixture.begin([210; 16])) + .await, + PreparationDenial::Capacity, + ); + denied( + client + .command::(&fixture.target, identity()?, request(staged.token)) + .await, + PreparationDenial::Capacity, + ); + let bound = lease( + client + .command::(&fixture.target, identity()?, check(staged.token)) + .await? + .output, + )?; + assert_eq!(bound.token, staged.token); + assert_eq!(fixture.counts().await?, (1, MAX_GENERATION_LEASES)); + assert_eq!(super::frontier::reap(&fixture).await?, REAP_ROWS); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[test] +fn staging_codecs_are_bounded_and_reject_truncated_or_invalid_leases() -> Result { + let original = StagingLease { + token: PreparationToken { + repository: uuid::Uuid::new_v4().into_bytes(), + operation: [212; 16], + artifact_operation: artifact_number(1), + request_digest: [212; 32], + owner: OwnerFence { + incarnation: IncarnationId::from_bytes([212; 16]), + epoch: u64::MAX, + }, + attempt: 1, + }, + format: ObjectFormat::Sha256, + observed_at_ms: 10, + expires_at_ms: 100, + }; + let reply = StagingReply::Granted(Box::new(original)); + let mut e = BoundedEncoder::new(256)?; + reply.encode(&mut e)?; + let bytes = e.finish(); + let mut d = BoundedDecoder::new(&bytes, 256)?; + assert_eq!(StagingReply::decode(&mut d)?, reply); + d.finish()?; + for length in 0..bytes.len() { + let mut d = BoundedDecoder::new(&bytes[..length], 256)?; + assert!(StagingReply::decode(&mut d).is_err()); + } + for invalid in [ + StagingLease { + observed_at_ms: -1, + ..original + }, + StagingLease { + expires_at_ms: 10, + ..original + }, + ] { + assert!(invalid.encode(&mut BoundedEncoder::new(256)?).is_err()); + } + for reason in [ + PreparationDenial::Unauthorized, + PreparationDenial::Conflict, + PreparationDenial::Stale, + PreparationDenial::Expired, + PreparationDenial::Capacity, + PreparationDenial::Missing, + ] { + let reply = StagingReply::Denied(reason); + let mut e = BoundedEncoder::new(1)?; + reply.encode(&mut e)?; + let bytes = e.finish(); + let mut d = BoundedDecoder::new(&bytes, 1)?; + assert_eq!(StagingReply::decode(&mut d)?, reply); + d.finish()?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/staging_service.rs b/crates/canopy-server/src/packs/publication/tests/staging_service.rs new file mode 100644 index 0000000..8005246 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/staging_service.rs @@ -0,0 +1,580 @@ +mod bound; +mod publication; +use super::*; +use tokio::{ + sync::oneshot, + time::{Duration, timeout}, +}; + +async fn submit( + fixture: &Fixture, + coordinator: &StagingCoordinator, + operation: [u8; 16], + actor: &str, +) -> Result { + let mut input = fixture.begin(operation); + input.actor = actor.into(); + let ready = + ReadyStaging::new(fixture.client(), fixture.target.clone(), input, identity()?).await?; + coordinator.submit(ready).map_err(|(e, _)| e.into()) +} +async fn active(ticket: &StagingTicket) -> Result { + match timeout(Duration::from_secs(10), ticket.wait()).await? { + StagingState::Active(lease) => Ok(lease), + other => Err(format!("unexpected {other:?}").into()), + } +} +async fn terminal(ticket: &StagingTicket) -> Result { + Ok(timeout(Duration::from_secs(10), ticket.wait_terminal()).await?) +} +async fn changed_lease(ticket: &StagingTicket, old: i64) -> Result { + timeout(Duration::from_secs(10), async { + loop { + match ticket.state() { + StagingState::Active(l) | StagingState::Draining(l) if l.expires_at_ms > old => { + return Ok(l); + } + StagingState::Fenced(e) | StagingState::Uncertain(e) => { + return Err(format!("unexpected {e:?}").into()); + } + _ => tokio::task::yield_now().await, + } + } + }) + .await? +} + +#[tokio::test] +async fn staged_service_canceled_observers_keep_workers_and_results_until_single_handoff() -> Result +{ + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let coordinator = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let ticket = submit(&fixture, &coordinator, [220; 16], "owner").await?; + let initial = active(&ticket).await?; + let (release, wait) = oneshot::channel(); + let (entered, started) = oneshot::channel(); + let work = ticket.spawn(move |ctx| async move { + let token = ctx.token()?; + let _ = entered.send(()); + wait.await.map_err(|_| StagingError::Worker)?; + ctx.ensure_live()?; + Ok(token) + })?; + let work_id = work.id(); + timeout(Duration::from_secs(10), started).await??; + drop(work); + drop(ticket); + let retained = coordinator.pending([220; 16]).ok_or("lost job")?; + assert_eq!(coordinator.stats().workers, 1); + retained.renew_for_test(); + let renewed = changed_lease(&retained, initial.expires_at_ms).await?; + assert_eq!(renewed.token, initial.token); + retained.seal()?; + assert!(matches!( + retained.spawn(|_| async { Ok(()) }), + Err(StagingError::Inactive) + )); + release.send(()).map_err(|_| "worker disappeared")?; + let result = retained + .pending_task::(work_id) + .ok_or("lost result")?; + assert!(retained.pending_task::(work_id).is_none()); + let token = result.wait().await.map_err(|e| format!("work {e}"))?; + assert_eq!(token, initial.token); + assert!(matches!(result.wait().await,Err(e) if matches!(*e,StagingError::NotReady))); + let StagingState::Bound(bound) = terminal(&retained).await? else { + return Err("not bound".into()); + }; + assert_eq!(bound.lease.token, initial.token); + assert_eq!(bound.lease.expires_at_ms, renewed.expires_at_ms); + assert_eq!(coordinator.stats().workers, 0); + assert!(coordinator.close_and_drain().await.is_empty()); + assert_eq!(coordinator.stats().admitted, 0); + assert_eq!(fixture.counts().await?, (1, 1)); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn staged_service_resolves_begin_renew_and_bind_exactly_after_absence_lost_ack_or_panic() +-> Result { + for fault in [1, 2, 3] { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let coordinator = + StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + coordinator.fault_for_test(fault); + let ticket = submit(&fixture, &coordinator, [221; 16], "owner").await?; + assert!(matches!( + terminal(&ticket).await?, + StagingState::Uncertain(_) + )); + assert_eq!(coordinator.stats().admitted, 1); + assert_eq!(coordinator.stats().command_bytes, 8192); + coordinator.recover(&ticket)?; + let original = active(&ticket).await?; + assert_eq!(original.token.artifact_operation, artifact_number(1)); + assert_eq!(fixture.counts().await?, (1, 1)); + coordinator.fault_for_test(fault); + ticket.renew_for_test(); + assert!(matches!( + terminal(&ticket).await?, + StagingState::Uncertain(_) + )); + coordinator.recover(&ticket)?; + // wait observes Resolving and then a queried live Active result. + let renewed = active(&ticket).await?; + assert_eq!(renewed.token, original.token); + coordinator.fault_for_test(fault); + ticket.seal()?; + let StagingState::Uncertain(error) = terminal(&ticket).await? else { + return Err("not uncertain bind".into()); + }; + let StagingError::Bind(ref error) = *error else { + return Err("wrong operation".into()); + }; + let InvocationError::Pending(evidence) = error.as_ref() else { + return Err("missing evidence".into()); + }; + let evidence = (**evidence).clone(); + let pending = timeout(Duration::from_secs(10), coordinator.close_and_drain()).await?; + assert_eq!(pending.len(), 1); + assert_eq!(coordinator.stats().command_bytes, 8192); + coordinator.recover(&pending[0])?; + let StagingState::Bound(bound) = terminal(&pending[0]).await? else { + return Err("bind did not resolve".into()); + }; + let cellule_runtime::Resolution::Committed(outcome) = + fixture.client().resolve(&evidence).await? + else { + return Err("missing committed outcome".into()); + }; + assert_eq!(bound.receipt.commit_sequence, outcome.commit_sequence()); + assert_eq!(bound.lease.token, original.token); + assert_eq!(fixture.counts().await?, (1, 1)); + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn staged_service_replayed_renewal_is_not_a_new_clock_or_permission_after_revocation() +-> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + super::publishing::edit( + &fixture, + "INSERT INTO repository_members VALUES('writer','write')", + ) + .await?; + let coordinator = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let ticket = submit(&fixture, &coordinator, [222; 16], "writer").await?; + active(&ticket).await?; + let (entered, started) = oneshot::channel(); + let work = ticket.spawn(move |_| async move { + let _ = entered.send(()); + std::future::pending::>().await + })?; + timeout(Duration::from_secs(10), started).await??; + coordinator.fault_for_test(2); + ticket.renew_for_test(); + assert!(matches!( + terminal(&ticket).await?, + StagingState::Uncertain(_) + )); + super::publishing::edit( + &fixture, + "UPDATE repository_members SET role='read' WHERE account='writer'", + ) + .await?; + coordinator.recover(&ticket)?; + assert!(matches!(terminal(&ticket).await?, StagingState::Fenced(_))); + assert!(work.wait().await.is_err()); + assert!( + timeout(Duration::from_secs(10), coordinator.close_and_drain()) + .await? + .is_empty() + ); + assert_eq!(coordinator.stats().workers, 0); + assert_eq!(coordinator.stats().admitted, 0); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn staged_service_account_operation_and_worker_bounds_preserve_rejected_ready_requests() +-> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + super::publishing::edit( + &fixture, + "INSERT INTO repository_members VALUES('writer','write')", + ) + .await?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits { + operations: 3, + per_actor: 1, + workers: 2, + workers_per_actor: 1, + ..StagingLimits::default() + }, + )?; + let first = submit(&fixture, &coordinator, [223; 16], "owner").await?; + active(&first).await?; + let ready = ReadyStaging::new( + fixture.client(), + fixture.target.clone(), + fixture.begin([224; 16]), + identity()?, + ) + .await?; + let (error, ready) = coordinator + .submit(ready) + .err() + .ok_or("actor over-admitted")?; + assert!(matches!(error, StagingError::Capacity)); + let second = submit(&fixture, &coordinator, [225; 16], "writer").await?; + active(&second).await?; + let (release, wait) = oneshot::channel(); + let work = first.spawn(move |_| async move { + wait.await.map_err(|_| StagingError::Worker)?; + Ok(7u64) + })?; + assert!(matches!( + first.spawn(|_| async { Ok(()) }), + Err(StagingError::Capacity) + )); + let writer = second.spawn(|_| async { Ok(8u64) })?; + assert_eq!(coordinator.stats().workers, 2); + assert!(matches!( + second.spawn(|_| async { Ok(()) }), + Err(StagingError::Capacity) + )); + assert_eq!(writer.wait().await.map_err(|e| format!("writer {e}"))?, 8); + first.stop(); + release.send(()).map_err(|_| "worker vanished")?; + assert_eq!(work.wait().await.map_err(|e| format!("work {e}"))?, 7); + assert!(matches!(terminal(&first).await?, StagingState::Stopped)); + let admitted = coordinator + .submit(ready) + .map_err(|(e, _)| format!("retry {e}"))?; + active(&admitted).await?; + assert_eq!(coordinator.stats().admitted, 2); + assert_eq!(coordinator.stats().accounts, 2); + assert!( + timeout(Duration::from_secs(10), coordinator.close_and_drain()) + .await? + .is_empty() + ); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn staged_service_worker_failure_and_panic_fence_before_binding_and_release_admission() +-> Result { + for panic in [false, true] { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + let coordinator = + StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let ticket = submit(&fixture, &coordinator, [226; 16], "owner").await?; + let lease = active(&ticket).await?; + let work = ticket.spawn(move |_| async move { + assert!(!panic, "injected producer panic"); + Err::<(), _>(StagingError::Context) + })?; + assert!(work.wait().await.is_err()); + assert!(matches!(terminal(&ticket).await?, StagingState::Fenced(_))); + assert!(coordinator.close_and_drain().await.is_empty()); + assert!( + fixture + .client() + .query::(&fixture.target, None, check(lease.token)) + .await? + .output + .is_none() + ); + assert_eq!(coordinator.stats().workers, 0); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn staged_service_owned_native_verification_hands_off_to_the_existing_private_catalog_pipeline() +-> Result { + use crate::packs::{ + catalog::{CatalogFileLimits, CatalogFiles, CatalogIndexes}, + metadata::tests::limits, + verification::physical::tests::prepared_for_store, + }; + use canopy_object_storage::artifact::ArtifactStore; + use cellule_ltx::DiskBudget; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = Fixture::new(format).await?; + let coordinator = + StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let ticket = submit(&fixture, &coordinator, [227; 16], "owner").await?; + let initial = active(&ticket).await?; + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new( + Arc::clone(&provider), + fixture.repository, + )); + let native = Arc::new( + prepared_for_store( + format, + 32, + initial.token.artifact_operation, + provider, + Arc::clone(&store), + ) + .await?, + ); + let root = Arc::new(tempfile::TempDir::new()?); + let budget = DiskBudget::new(256 << 20); + let work = { + let root = Arc::clone(&root); + let budget = budget.clone(); + let native = Arc::clone(&native); + ticket.spawn(move |ctx| async move { + ctx.ensure_live()?; + let result = super::prepare::physical(&native, root.path(), budget) + .await + .map_err(|e| StagingError::Input(e.to_string().into()))?; + ctx.ensure_live()?; + Ok(result) + })? + }; + let (witness, segments) = work.wait().await.map_err(|e| format!("native {e}"))?; + ticket.seal()?; + assert!(matches!(terminal(&ticket).await?, StagingState::Bound(_))); + let indexes = Arc::new(CatalogIndexes::new(Arc::clone(&store), format)); + let files = Arc::new(CatalogFiles::new( + fixture.root.path(), + DiskBudget::new(64 << 20), + store, + format, + CatalogFileLimits::default(), + )?); + let base = Arc::new(ticket.open_base(indexes, files).await?); + let bound_work = { + let root = root.clone(); + let budget = budget.clone(); + ticket.spawn_bound(move |_| async move { + async { + let mut assembler = + CatalogPreparation::new(root.path(), budget.clone(), base, limits()) + .await?; + assembler.begin_pack(witness)?; + for segment in segments { + assembler.add_segment(segment).await?; + } + assembler.finish_pack().await?; + let proof = assembler.finish().await?; + Ok(proof) + } + .await + .map_err(|e: CatalogPreparationError| StagingError::Input(Box::new(e))) + })? + }; + let proof = bound_work.wait().await.map_err(|e| e.to_string())?; + assert_eq!(proof.token(), initial.token); + assert_eq!(proof.object_count(), native.fixture.objects.len() as u64); + assert!(matches!( + proof.attest(identity()?).await?.output, + AttestationOutcome::Registered(_) + )); + drop(proof); + super::prepare::cleaned(root.path(), &budget).await?; + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + } + Ok(()) +} + +#[tokio::test] +async fn staged_service_automatic_renewal_runs_without_an_observer_or_manual_tick() -> Result { + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + let coordinator = StagingCoordinator::new( + fixture.target.clone(), + StagingLimits { + renew_before_ms: DEFAULT_LEASE_MS - 1000, + ..StagingLimits::default() + }, + )?; + let ticket = submit(&fixture, &coordinator, [228; 16], "owner").await?; + let lease = active(&ticket).await?; + drop(ticket); + let retained = coordinator.pending([228; 16]).ok_or("lost automatic job")?; + let renewed = changed_lease(&retained, lease.expires_at_ms).await?; + assert_eq!(renewed.token, lease.token); + assert!( + fixture + .client() + .query::(&fixture.target, None, check(lease.token)) + .await? + .output + .is_none() + ); + assert_eq!(fixture.counts().await?, (1, 1)); + retained.stop(); + assert!(matches!(terminal(&retained).await?, StagingState::Stopped)); + assert!(coordinator.close_and_drain().await.is_empty()); + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn staged_service_rejects_invalid_profiles_foreign_targets_and_duplicate_logical_requests() +-> Result { + let fixture = Fixture::new(ObjectFormat::Sha1).await?; + for limits in [ + StagingLimits { + operations: 0, + ..StagingLimits::default() + }, + StagingLimits { + per_actor: 32, + ..StagingLimits::default() + }, + StagingLimits { + workers: 0, + ..StagingLimits::default() + }, + StagingLimits { + workers_per_actor: 0, + ..StagingLimits::default() + }, + StagingLimits { + workers_per_actor: 65, + ..StagingLimits::default() + }, + StagingLimits { + workers: MAX_GENERATION_LEASES as usize + 1, + ..StagingLimits::default() + }, + StagingLimits { + lease_ms: 0, + ..StagingLimits::default() + }, + StagingLimits { + lease_ms: MAX_LEASE_MS + 1, + ..StagingLimits::default() + }, + StagingLimits { + renew_before_ms: DEFAULT_LEASE_MS, + ..StagingLimits::default() + }, + StagingLimits { + lifetime_ms: 1, + ..StagingLimits::default() + }, + StagingLimits { + lifetime_ms: u64::MAX, + ..StagingLimits::default() + }, + ] { + assert!(matches!( + StagingCoordinator::new(fixture.target.clone(), limits), + Err(StagingError::InvalidLimits) + )); + } + let coordinator = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let ticket = submit(&fixture, &coordinator, [229; 16], "owner").await?; + active(&ticket).await?; + let ready = ReadyStaging::new( + fixture.client(), + fixture.target.clone(), + fixture.begin([229; 16]), + identity()?, + ) + .await?; + let (error, ready) = coordinator + .submit(ready) + .err() + .ok_or("duplicate accepted")?; + assert!(matches!(error, StagingError::Duplicate)); + let foreign = Fixture::new(ObjectFormat::Sha256).await?; + let other = StagingCoordinator::new(foreign.target.clone(), StagingLimits::default())?; + let (error, _) = other.submit(ready).err().ok_or("foreign accepted")?; + assert!(matches!(error, StagingError::Foreign)); + assert!(matches!(other.recover(&ticket), Err(StagingError::Foreign))); + assert!(coordinator.close_and_drain().await.is_empty()); + let ready = ReadyStaging::new( + fixture.client(), + fixture.target.clone(), + fixture.begin([230; 16]), + identity()?, + ) + .await?; + assert!(matches!( + coordinator.submit(ready), + Err((StagingError::Closed, _)) + )); + foreign.runtime.shutdown().await?; + fixture.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn staged_service_revocation_drops_completed_owned_results_before_releasing_worker_credit() +-> Result { + use std::sync::atomic::{AtomicBool, Ordering}; + struct OwnedInput { + coordinator: StagingCoordinator, + dropped: Arc, + wrong_order: Arc, + } + impl Drop for OwnedInput { + fn drop(&mut self) { + if self.coordinator.stats().workers != 1 { + self.wrong_order.store(true, Ordering::Release); + } + self.dropped.store(true, Ordering::Release); + } + } + let fixture = Fixture::new(ObjectFormat::Sha256).await?; + super::publishing::edit( + &fixture, + "INSERT INTO repository_members VALUES('writer','write')", + ) + .await?; + let coordinator = StagingCoordinator::new(fixture.target.clone(), StagingLimits::default())?; + let ticket = submit(&fixture, &coordinator, [231; 16], "writer").await?; + active(&ticket).await?; + let dropped = Arc::new(AtomicBool::new(false)); + let wrong_order = Arc::new(AtomicBool::new(false)); + let resource = OwnedInput { + coordinator: coordinator.clone(), + dropped: Arc::clone(&dropped), + wrong_order: Arc::clone(&wrong_order), + }; + let (done, finished) = oneshot::channel(); + let work = ticket.spawn(move |_| async move { + let _ = done.send(()); + Ok(resource) + })?; + timeout(Duration::from_secs(10), finished).await??; + // Retain the observer deliberately. Neither completed input ownership nor + // its credit may depend on dropping an external ticket after revocation. + assert_eq!(coordinator.stats().workers, 1); + super::publishing::edit( + &fixture, + "UPDATE repository_members SET role='read' WHERE account='writer'", + ) + .await?; + ticket.renew_for_test(); + assert!(matches!(terminal(&ticket).await?, StagingState::Fenced(_))); + assert!( + timeout(Duration::from_secs(10), coordinator.close_and_drain()) + .await? + .is_empty() + ); + assert!(dropped.load(Ordering::Acquire)); + assert!(!wrong_order.load(Ordering::Acquire)); + assert_eq!(coordinator.stats().workers, 0); + assert!(work.wait().await.is_err()); + fixture.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/staging_service/bound.rs b/crates/canopy-server/src/packs/publication/tests/staging_service/bound.rs new file mode 100644 index 0000000..9795148 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/staging_service/bound.rs @@ -0,0 +1,692 @@ +use super::*; +use crate::packs::catalog::{CatalogFileLimits, CatalogFiles}; +use crate::packs::closure::{BaseResolver, ClosureError}; +use canopy_object_storage::artifact::ArtifactStore; +use cellule_ltx::DiskBudget; + +async fn wait_for( + ticket: &StagingTicket, + predicate: impl Fn(&StagingState) -> bool, +) -> Result { + timeout(Duration::from_secs(10), async { + loop { + let state = ticket.state(); + if predicate(&state) { + return Ok(state); + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await? +} +pub(super) async fn bind( + f: &Fixture, + c: &StagingCoordinator, + op: [u8; 16], + actor: &str, +) -> Result { + let ticket = submit(f, c, op, actor).await?; + active(&ticket).await?; + ticket.seal()?; + assert!(matches!(terminal(&ticket).await?, StagingState::Bound(_))); + Ok(ticket) +} +pub(super) async fn claim( + f: &Fixture, + c: &StagingCoordinator, + token: PreparationToken, + mutation: MutationIdentity, +) -> Result { + let ready = ReadyStaging::claim_bound( + f.client(), + f.target.clone(), + LeaseRequest { + check: check(token), + lease_ms: DEFAULT_LEASE_MS, + }, + mutation, + ) + .await?; + Ok(c.submit(ready).map_err(|(e, _)| e)?) +} +async fn new_token(f: &Fixture, op: [u8; 16]) -> Result { + Ok(lease( + f.client() + .command::(&f.target, identity()?, f.begin(op)) + .await? + .output, + )? + .token) +} +#[tokio::test] +async fn bound_service_automatic_renewal_keeps_canceled_worker_and_result_owned_through_close() +-> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let c = StagingCoordinator::new( + f.target.clone(), + StagingLimits { + renew_before_ms: DEFAULT_LEASE_MS - 1000, + ..StagingLimits::default() + }, + )?; + let ticket = bind(&f, &c, [203; 16], "owner").await?; + let original = ticket.bound_result().ok_or("binding receipt")?; + let session = ticket.bound_session()?; + let weak = Arc::downgrade(&session); + let (release, wait) = oneshot::channel(); + let (entered, running) = oneshot::channel(); + let worker = ticket.spawn_bound(move |session| async move { + session.live_lease()?; + let _ = entered.send(()); + wait.await.map_err(|_| StagingError::Worker)?; + session.live_lease()?; + Ok(session) + })?; + let id = worker.id(); + timeout(Duration::from_secs(10), running).await??; + drop(worker); + drop(session); + drop(ticket); + let retained = c.pending([203; 16]).ok_or("bound job lost")?; + timeout(Duration::from_secs(10), async { + while retained.bound_renewal().is_none() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + let renewed = lease(retained.bound_renewal().ok_or("renewal outcome")?.output)?; + assert!(renewed.expires_at_ms > original.lease.expires_at_ms); + assert_eq!(renewed.base, original.lease.base); + assert_eq!(renewed.token, original.lease.token); + assert_eq!( + retained.bound_result().ok_or("original result")?.receipt, + original.receipt + ); + assert!(weak.upgrade().is_some()); + assert_eq!(c.stats().workers, 1); + assert_eq!(c.stats().admitted, 1); + let closing = c.clone(); + let close = tokio::spawn(async move { closing.close_and_drain().await }); + timeout(Duration::from_secs(10), async { + while !c.stats().closed { + tokio::task::yield_now().await; + } + }) + .await?; + assert!(!close.is_finished()); + assert!(retained.spawn_bound(|_| async { Ok(()) }).is_err()); + release.send(()).map_err(|_| "worker lost")?; + let result = retained + .pending_task::>(id) + .ok_or("bound result lost")?; + let transferred = result.wait().await.map_err(|e| e.to_string())?; + assert!(timeout(Duration::from_secs(10), close).await??.is_empty()); + assert!(transferred.live_lease().is_err()); + assert!(retained.bound_session().is_err()); + assert_eq!(c.stats().workers, 0); + assert_eq!(c.stats().admitted, 0); + assert_eq!( + retained + .bound_result() + .ok_or("stopped binding receipt")? + .receipt, + original.receipt + ); + f.runtime.shutdown().await?; + Ok(()) +} +#[tokio::test] +async fn bound_service_renewal_retains_exact_absent_lost_and_panicked_commands_through_closed_recovery() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in [1, 2, 3] { + let f = Fixture::new(format).await?; + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let ticket = bind(&f, &c, [204; 16], "owner").await?; + let original = ticket.bound_result().ok_or("binding")?; + let shared = ticket.bound_session()?; + f.install_empty_root(1).await?; + c.fault_for_test(fault); + ticket.renew_for_test(); + let StagingState::Uncertain(error) = wait_for(&ticket, |s| { + matches!(s, StagingState::Uncertain(_) | StagingState::Fenced(_)) + }) + .await? + else { + return Err("bound renewal uncertainty".into()); + }; + let StagingError::BoundRenew(error) = &*error else { + return Err("renewal evidence kind".into()); + }; + let InvocationError::Pending(evidence) = &**error else { + return Err("renewal evidence".into()); + }; + let evidence = (**evidence).clone(); + let sequence = match f.client().resolve(&evidence).await? { + cellule_runtime::Resolution::Absent => None, + cellule_runtime::Resolution::Committed(value) => Some(value.commit_sequence()), + other => return Err(format!("unexpected {other:?}").into()), + }; + assert_eq!(sequence.is_some(), fault != 1); + assert_eq!(c.stats().command_bytes, 8192); + drop(ticket); + let retained = c.pending([204; 16]).ok_or("renewal lost")?; + assert_eq!(c.close_and_drain().await.len(), 1); + c.recover(&retained)?; + wait_for(&retained, |s| { + matches!(s, StagingState::Bound(_) | StagingState::Fenced(_)) + }) + .await?; + assert!(c.close_and_drain().await.is_empty()); + let renewed = retained.bound_renewal().ok_or("known renewal")?; + let granted = lease(renewed.output)?; + assert_eq!(granted.base, original.lease.base); + assert_eq!(granted.token, original.lease.token); + if let Some(sequence) = sequence { + assert_eq!(renewed.receipt.commit_sequence, sequence); + } + let cellule_runtime::Resolution::Committed(resolved) = + f.client().resolve(&evidence).await? + else { + return Err("resolved renewal".into()); + }; + assert_eq!(renewed.receipt.commit_sequence, resolved.commit_sequence()); + assert_eq!( + retained.bound_result().ok_or("original binding")?.receipt, + original.receipt + ); + assert!(shared.live_lease().is_err()); + assert_eq!(c.stats().admitted, 0); + f.runtime.shutdown().await?; + } + } + Ok(()) +} +#[tokio::test] +async fn bound_service_claim_retains_exact_identity_and_new_namespace_through_closed_recovery() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in [1, 2, 3] { + let f = Fixture::new(format).await?; + let old = new_token(&f, [205; 16]).await?; + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let mutation = identity()?; + c.fault_for_test(fault); + let ticket = claim(&f, &c, old, mutation).await?; + let StagingState::Uncertain(error) = terminal(&ticket).await? else { + return Err("bound Claim uncertainty".into()); + }; + let StagingError::BoundClaim(error) = &*error else { + return Err("bound Claim evidence".into()); + }; + let InvocationError::Pending(evidence) = &**error else { + return Err("Claim evidence".into()); + }; + let evidence = (**evidence).clone(); + drop(ticket); + let retained = c.pending([205; 16]).ok_or("Claim lost")?; + assert_eq!(c.close_and_drain().await.len(), 1); + c.recover(&retained)?; + let StagingState::Bound(bound) = terminal(&retained).await? else { + return Err("Claim recovery".into()); + }; + assert_ne!(bound.lease.token, old); + assert_ne!(bound.lease.token.artifact_operation, old.artifact_operation); + let replay = f + .client() + .command::( + &f.target, + mutation, + LeaseRequest { + check: check(old), + lease_ms: DEFAULT_LEASE_MS, + }, + ) + .await?; + assert_eq!(bound.receipt, replay.receipt); + assert_eq!(bound.lease, lease(replay.output)?); + let cellule_runtime::Resolution::Committed(value) = + f.client().resolve(&evidence).await? + else { + return Err("Claim exact resolve".into()); + }; + assert_eq!(value.commit_sequence(), bound.receipt.commit_sequence); + assert!(c.close_and_drain().await.is_empty()); + assert!(retained.bound_session().is_err()); + assert_eq!(f.counts().await?, (1, 2)); + f.runtime.shutdown().await?; + } + } + Ok(()) +} +#[tokio::test] +async fn bound_service_known_renewal_receipts_survive_revocation_expiry_and_superseding_claim() +-> Result { + for committed in [false, true] { + for mode in [0, 1, 2] { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let ticket = bind(&f, &c, [206; 16], "owner").await?; + let session = ticket.bound_session()?; + let binding = ticket.bound_result().ok_or("binding")?; + c.fault_for_test(if committed { 2 } else { 1 }); + ticket.renew_for_test(); + let StagingState::Uncertain(error) = wait_for(&ticket, |s| { + matches!(s, StagingState::Uncertain(_) | StagingState::Fenced(_)) + }) + .await? + else { + return Err("renew uncertain".into()); + }; + let StagingError::BoundRenew(error) = &*error else { + return Err("renew error".into()); + }; + let InvocationError::Pending(evidence) = &**error else { + return Err("evidence".into()); + }; + let evidence = (**evidence).clone(); + let denial = match mode { + 0 => { + super::super::publishing::edit( + &f, + "UPDATE repository_identity SET owner='other' WHERE singleton=1", + ) + .await?; + PreparationDenial::Unauthorized + } + 1 => { + super::super::publishing::edit(&f, "UPDATE catalog_operations SET expires_at_ms=0; UPDATE catalog_leases SET expires_at_ms=0").await?; + PreparationDenial::Expired + } + _ => { + f.client() + .command::( + &f.target, + identity()?, + LeaseRequest { + check: session.check.clone(), + lease_ms: DEFAULT_LEASE_MS, + }, + ) + .await?; + PreparationDenial::Stale + } + }; + c.recover(&ticket)?; + let StagingState::Fenced(error) = + wait_for(&ticket, |s| matches!(s, StagingState::Fenced(_))).await? + else { + return Err("fresh custody not fenced".into()); + }; + if committed { + let renewal = ticket.bound_renewal().ok_or("known outcome lost")?; + assert!(matches!(renewal.output, PreparationReply::Granted(_))); + let cellule_runtime::Resolution::Committed(value) = + f.client().resolve(&evidence).await? + else { + return Err("original committed receipt".into()); + }; + assert_eq!(renewal.receipt.commit_sequence, value.commit_sequence()); + } else { + assert!(ticket.bound_renewal().is_none()); + assert!( + matches!(&*error, StagingError::BoundRenew(value) if matches!(&**value, InvocationError::Rejected(result) if result.output == PreparationReply::Denied(denial))) + ); + } + assert!(session.live_lease().is_err()); + assert!(ticket.bound_session().is_err()); + assert_eq!( + ticket.bound_result().ok_or("binding lost")?.receipt, + binding.receipt + ); + assert!(c.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + } + Ok(()) +} +#[tokio::test] +async fn bound_service_phase_handoff_and_residence_cap_fence_existing_bases_and_inflight_workers() +-> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let native = + crate::packs::catalog::tests::prepared_for_repository(f.format, f.repository).await?; + f.install_catalog(1, native.stored).await?; + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let ticket = submit(&f, &c, [207; 16], "owner").await?; + active(&ticket).await?; + let work = ticket.spawn(|ctx| async { Ok(ctx) })?; + let old = work.wait().await.map_err(|e| e.to_string())?; + ticket.seal()?; + assert!(matches!(terminal(&ticket).await?, StagingState::Bound(_))); + assert!(old.ensure_live().is_err()); + assert!(ticket.spawn(|_| async { Ok(()) }).is_err()); + let session = ticket.bound_session()?; + let store = native.store.clone(); + let root = tempfile::TempDir::new()?; + let base = ticket + .open_base( + native.indexes.clone(), + Arc::new(CatalogFiles::new( + root.path(), + DiskBudget::new(64 << 20), + store, + f.format, + CatalogFileLimits::default(), + )?), + ) + .await?; + let base_context = base.context().base.ok_or("native base")?; + let oid = *native + .fixture + .objects + .keys() + .next() + .ok_or("native object")?; + assert!(base.resolve(base_context, &[oid]).await?.objects[0].is_some()); + // Fixture a past ceiling without fencing: the base reader must reject even + // before the supervisor gets a chance to apply its permanent shared fence. + let mut limited = base.select_current().await?; + limited.session.ceiling = Some(tokio::time::Instant::now()); + assert!(!session.fenced.load(std::sync::atomic::Ordering::Acquire)); + assert!(matches!( + limited.resolve(base_context, &[oid]).await, + Err(ClosureError::LeaseExpired) + )); + assert!(!session.fenced.load(std::sync::atomic::Ordering::Acquire)); + let (entered, running) = oneshot::channel(); + let worker = ticket.spawn_bound(move |_| async move { + let _ = entered.send(()); + std::future::pending::>().await + })?; + timeout(Duration::from_secs(10), running).await??; + tokio::time::pause(); + tokio::time::advance(Duration::from_millis(DEFAULT_LEASE_MS + 1)).await; + tokio::time::resume(); + wait_for(&ticket, |s| matches!(s, StagingState::Fenced(_))).await?; + assert!(session.live_lease().is_err()); + assert!(base.live_lease().is_err()); + assert!(matches!( + base.resolve(base_context, &[oid]).await, + Err(ClosureError::LeaseExpired) + )); + assert!(worker.wait().await.is_err()); + assert!(c.close_and_drain().await.is_empty()); + assert_eq!(c.stats().workers, 0); + f.runtime.shutdown().await?; + Ok(()) +} +#[tokio::test] +async fn bound_service_worker_caps_results_and_failure_reuse_staging_admission() -> Result { + use std::sync::atomic::{AtomicBool, Ordering}; + struct Owned { + c: StagingCoordinator, + dropped: Arc, + wrong: Arc, + } + impl Drop for Owned { + fn drop(&mut self) { + if self.c.stats().workers != 1 { + self.wrong.store(true, Ordering::Release); + } + self.dropped.store(true, Ordering::Release); + } + } + let f = Fixture::new(ObjectFormat::Sha256).await?; + let c = StagingCoordinator::new( + f.target.clone(), + StagingLimits { + workers: 2, + workers_per_actor: 1, + ..StagingLimits::default() + }, + )?; + let a = bind(&f, &c, [208; 16], "owner").await?; + let b = bind(&f, &c, [209; 16], "owner").await?; + let session = a.bound_session()?; + let dropped = Arc::new(AtomicBool::new(false)); + let wrong = Arc::new(AtomicBool::new(false)); + let owned = Owned { + c: c.clone(), + dropped: dropped.clone(), + wrong: wrong.clone(), + }; + let (done, completed) = oneshot::channel(); + let work = a.spawn_bound(move |_| async move { + let _ = done.send(()); + Ok(owned) + })?; + timeout(Duration::from_secs(10), completed).await??; + assert_eq!(c.stats().workers, 1); + assert!(b.spawn_bound(|_| async { Ok(()) }).is_err()); + super::super::publishing::edit( + &f, + "UPDATE repository_identity SET owner='other' WHERE singleton=1", + ) + .await?; + a.renew_for_test(); + wait_for(&a, |s| matches!(s, StagingState::Fenced(_))).await?; + assert!(work.wait().await.is_err()); + assert!(dropped.load(Ordering::Acquire)); + assert!(!wrong.load(Ordering::Acquire)); + assert!(session.live_lease().is_err()); + assert_eq!(c.stats().workers, 0); + assert!(c.close_and_drain().await.is_empty()); + for invalid in [0, MAX_LEASE_MS + 1] { + assert!( + StagingCoordinator::new( + f.target.clone(), + StagingLimits { + bound_lifetime_ms: invalid, + ..StagingLimits::default() + } + ) + .is_err() + ); + } + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn bound_service_checkpoint_shares_renewal_order_exact_recovery_and_original_receipt_custody() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in [1, 2, 3] { + let f = Fixture::new(format).await?; + let (source, first) = super::super::inputs::active(&f, [210; 16]).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let prior = super::super::inputs::seal(&f, &first, store.clone(), 300).await?; + first + .register_inputs(prior.clone(), identity()?) + .map_err(|(e, _)| e)? + .wait() + .await + .map_err(|e| e.to_string())?; + first.seal()?; + let StagingState::Bound(old) = terminal(&first).await? else { + return Err("source bind".into()); + }; + assert!(source.close_and_drain().await.is_empty()); + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let ticket = claim(&f, &c, old.lease.token, identity()?).await?; + assert!(matches!(terminal(&ticket).await?, StagingState::Bound(_))); + let session = ticket.bound_session()?; + let parent = prior.clone(); + let provider = store.clone(); + let worker = ticket.spawn_bound(move |s| async move { + s.adopt_native_inputs(provider, &parent) + .await + .map_err(|e| StagingError::Input(Box::new(e))) + })?; + let adopted = worker.wait().await.map_err(|e| e.to_string())?; + assert_eq!(adopted.root()?, prior.root()?); + // Resolve a due renewal before the admitted checkpoint; its original + // output remains separate from the checkpoint's new exact command. + ticket.renew_for_test(); + timeout(Duration::from_secs(10), async { + while ticket.bound_renewal().is_none() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + let renewal = ticket.bound_renewal().ok_or("ordered renewal")?; + let mutation = identity()?; + c.fault_for_test(fault); + let observer = ticket + .register_inputs(adopted.clone(), mutation) + .map_err(|(e, _)| e)?; + drop(observer); + let StagingState::Uncertain(error) = wait_for(&ticket, |s| { + matches!(s, StagingState::Uncertain(_) | StagingState::Fenced(_)) + }) + .await? + else { + return Err("bound checkpoint uncertainty".into()); + }; + let StagingError::Checkpoint(error) = &*error else { + return Err("bound checkpoint evidence".into()); + }; + let InvocationError::Pending(evidence) = &**error else { + return Err("checkpoint evidence".into()); + }; + let evidence = (**evidence).clone(); + assert_eq!(c.stats().command_bytes, 12 << 10); + if fault == 2 { + super::super::publishing::edit( + &f, + "UPDATE repository_identity SET owner='other' WHERE singleton=1", + ) + .await?; + } + assert_eq!(c.close_and_drain().await.len(), 1); + c.recover(&ticket)?; + let recorded = ticket + .pending_inputs() + .ok_or("checkpoint observer lost")? + .wait() + .await + .map_err(|e| e.to_string())?; + let original = f + .client() + .command::(&f.target, mutation, adopted.clone()) + .await?; + assert_eq!(recorded, original.receipt); + assert!(recorded.commit_sequence > renewal.receipt.commit_sequence); + let cellule_runtime::Resolution::Committed(value) = + f.client().resolve(&evidence).await? + else { + return Err("checkpoint resolution".into()); + }; + assert_eq!(recorded.commit_sequence, value.commit_sequence()); + assert!(c.close_and_drain().await.is_empty()); + assert!(session.live_lease().is_err()); + assert_eq!( + ticket.bound_renewal().ok_or("renewal overwritten")?.receipt, + renewal.receipt + ); + if fault != 2 { + let current = f + .client() + .query::(&f.target, Some(recorded), session.check.clone()) + .await? + .output + .ok_or("destination checkpoint")?; + assert_eq!(current.root()?, prior.root()?); + } else { + assert!(matches!(ticket.state(), StagingState::Fenced(_))); + } + f.runtime.shutdown().await?; + } + } + Ok(()) +} +#[tokio::test] +async fn bound_service_restored_owner_claim_retains_old_pin_and_owns_new_session_workers() -> Result +{ + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let old = new_token(&f, [211; 16]).await?; + f.handle.drain().await?; + f.runtime.shutdown().await?; + let owner_session = SessionId::from_bytes([212; 16]); + let runtime = CellRuntime::new(SqlWorkerPool::new(1, 4)?, 64 << 20, owner_session)?; + let authority = CellAuthority::new(f.layout.clone()); + let idle = authority.load(f.target.cell_id()).await?.ok_or("idle")?; + let provision = CellCatalog::new(f.layout.clone(), f.target.tenant()) + .lookup(f.target.cell_id()) + .await? + .ok_or("provision")?; + let handle = runtime + .acquire_idle_restored( + provision, + f.replica.clone(), + authority, + idle, + f.root.path().join("bound-lifecycle-restored.sqlite"), + Owner { + session: owner_session, + endpoint: "https://bound-lifecycle-restored.invalid".into(), + }, + ) + .await?; + let client = CellClient::local(f.registry.clone(), handle.clone()); + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let mutation = identity()?; + c.fault_for_test(2); + let ready = ReadyStaging::claim_bound( + client.clone(), + f.target.clone(), + LeaseRequest { + check: check(old), + lease_ms: DEFAULT_LEASE_MS, + }, + mutation, + ) + .await?; + let ticket = c.submit(ready).map_err(|(e, _)| e)?; + assert!(matches!( + terminal(&ticket).await?, + StagingState::Uncertain(_) + )); + c.recover(&ticket)?; + let StagingState::Bound(bound) = terminal(&ticket).await? else { + return Err("restored bound Claim".into()); + }; + assert_ne!(bound.lease.token.owner, old.owner); + assert_ne!(bound.lease.token.artifact_operation, old.artifact_operation); + let replay = client + .command::( + &f.target, + mutation, + LeaseRequest { + check: check(old), + lease_ms: DEFAULT_LEASE_MS, + }, + ) + .await?; + assert_eq!(bound.receipt, replay.receipt); + let old_pin = handle.query(0, 32, move |conn| { Ok(conn.query_row("SELECT generation FROM catalog_leases WHERE incarnation=?1 AND admission_sequence=?2", rusqlite::params![old.owner.incarnation.as_bytes().as_slice(), old.attempt as i64], |row| row.get::<_, i64>(0))?.to_be_bytes().to_vec()) }).await?; + assert_eq!(old_pin.as_slice(), 0i64.to_be_bytes()); + let worker = + ticket.spawn_bound(|session| async move { Ok(session.live_lease()?.0.token) })?; + assert_eq!( + worker.wait().await.map_err(|e| e.to_string())?, + bound.lease.token + ); + ticket.renew_for_test(); + timeout(Duration::from_secs(10), async { + while ticket.bound_renewal().is_none() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + assert_eq!( + lease(ticket.bound_renewal().ok_or("restored renewal")?.output)?.token, + bound.lease.token + ); + assert!(c.close_and_drain().await.is_empty()); + runtime.shutdown().await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/publication/tests/staging_service/publication.rs b/crates/canopy-server/src/packs/publication/tests/staging_service/publication.rs new file mode 100644 index 0000000..4f49139 --- /dev/null +++ b/crates/canopy-server/src/packs/publication/tests/staging_service/publication.rs @@ -0,0 +1,513 @@ +use super::super::coordinator::{finished, refused, request}; +use super::super::publishing::edit; +use super::*; + +async fn ready(session: &Arc) -> Result { + Ok(session + .ready_outcome(identity()?, request(refused())) + .await?) +} + +#[tokio::test] +async fn bound_final_waits_for_exact_renewal_and_adopted_checkpoint_recovery_before_dispatch() +-> Result { + use canopy_object_storage::artifact::ArtifactStore; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in [1, 2, 3] { + for renewal in [false, true] { + let f = Fixture::new(format).await?; + let (source, first) = super::super::inputs::active(&f, [235; 16]).await?; + let store = Arc::new(ArtifactStore::new(Arc::new(InMemory::new()), f.repository)); + let prior = super::super::inputs::seal(&f, &first, store.clone(), 3).await?; + first + .register_inputs(prior.clone(), identity()?) + .map_err(|(e, _)| e)? + .wait() + .await + .map_err(|e| e.to_string())?; + first.seal()?; + let StagingState::Bound(original) = terminal(&first).await? else { + return Err("source binding lost".into()); + }; + assert!(source.close_and_drain().await.is_empty()); + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let p = + PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let ticket = super::bound::claim(&f, &c, original.lease.token, identity()?).await?; + assert!(matches!(terminal(&ticket).await?, StagingState::Bound(_))); + let session = ticket.bound_session()?; + let adopted = session.adopt_native_inputs(store, &prior).await?; + assert_eq!(adopted.root()?, prior.root()?); + let input = ready(&session).await?; + c.fault_for_test(fault); + if renewal { + ticket.renew_for_test(); + } + let checkpoint = ticket + .register_inputs(adopted, identity()?) + .map_err(|(e, _)| e)?; + let observer = ticket.publish(&p, input)?; + let StagingState::Uncertain(error) = wait_for(&ticket, |s| { + matches!( + s, + StagingState::Uncertain(_) + | StagingState::Fenced(_) + | StagingState::Published(_) + ) + }) + .await? + else { + return Err("pre-publication command uncertainty lost".into()); + }; + if renewal { + assert!(matches!(error.as_ref(), StagingError::BoundRenew(_))); + } else { + assert!(matches!(error.as_ref(), StagingError::Checkpoint(_))); + } + assert!(matches!(observer.state(), PublicationState::Held)); + assert_eq!(c.stats().command_bytes, 12 << 10); + assert_eq!(p.stats().await.held, 1); + assert_eq!(c.close_and_drain().await.len(), 1); + assert_eq!(p.close_and_drain().await.len(), 1); + c.recover(&ticket)?; + let registration = timeout(Duration::from_secs(10), checkpoint.wait()) + .await? + .map_err(|e| e.to_string())?; + let completed = finished(timeout(Duration::from_secs(10), observer.wait()).await?)?; + assert!(completed.receipt.commit_sequence > registration.commit_sequence); + if renewal { + assert!( + registration.commit_sequence + > ticket + .bound_renewal() + .ok_or("original renewal outcome lost")? + .receipt + .commit_sequence + ); + } + assert!(matches!( + terminal(&ticket).await?, + StagingState::Published(Ok(_)) + )); + assert_eq!(observer.response().await?, refused()); + assert_eq!(p.reservations_for_test().await, (0, 0, 0)); + assert!(c.close_and_drain().await.is_empty()); + assert!(p.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + } + } + Ok(()) +} +async fn wait_for( + ticket: &StagingTicket, + predicate: impl Fn(&StagingState) -> bool, +) -> Result { + timeout(Duration::from_secs(10), async { + loop { + let state = ticket.state(); + if predicate(&state) { + return state; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .map_err(Into::into) +} + +#[tokio::test] +async fn bound_final_publication_drains_retained_work_and_due_renewal_through_closed_admission() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let f = Fixture::new(format).await?; + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let ticket = super::bound::bind(&f, &c, [231; 16], "owner").await?; + let session = ticket.bound_session()?; + let input = ready(&session).await?; + let (release, blocked) = oneshot::channel(); + let (entered, running) = oneshot::channel(); + let work = ticket.spawn_bound(move |_| async move { + let _ = entered.send(()); + blocked.await.map_err(|_| StagingError::Worker)?; + Ok(42u64) + })?; + let work_id = work.id(); + timeout(Duration::from_secs(10), running).await??; + ticket.renew_for_test(); + let publication = ticket.publish(&p, input)?; + drop(publication); + drop(work); + assert!(matches!(ticket.state(), StagingState::Finishing)); + assert!(ticket.spawn_bound(|_| async { Ok(()) }).is_err()); + assert!(ticket.bound_session().is_err()); + assert_eq!(p.stats().await.held, 1); + assert_eq!(c.stats().workers, 1); + timeout(Duration::from_secs(10), async { + while ticket.bound_renewal().is_none() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + let renewed = ticket.bound_renewal().ok_or("serialized renewal")?; + assert!(matches!(ticket.state(), StagingState::Finishing)); + assert_eq!(p.close_and_drain().await.len(), 1); + let closing = c.clone(); + let closed = tokio::spawn(async move { closing.close_and_drain().await }); + timeout(Duration::from_secs(10), async { + while !c.stats().closed { + tokio::task::yield_now().await; + } + }) + .await?; + assert!(!closed.is_finished()); + release.send(()).map_err(|_| "producer disappeared")?; + let retained = ticket + .pending_task::(work_id) + .ok_or("retained result lost")?; + assert_eq!(retained.wait().await.map_err(|e| e.to_string())?, 42); + let publication = ticket.pending_publication().ok_or("final observer lost")?; + let value = finished(timeout(Duration::from_secs(10), publication.wait()).await?)?; + assert!(value.receipt.commit_sequence > renewed.receipt.commit_sequence); + assert!(matches!( + terminal(&ticket).await?, + StagingState::Published(Ok(PublicationOutcome::Push(_))) + )); + assert_eq!(publication.response().await?, refused()); + assert!(timeout(Duration::from_secs(10), closed).await??.is_empty()); + assert!(session.live_lease().is_err()); + assert!( + f.client() + .query::( + &f.target, + Some(value.receipt), + check(session.lease.token) + ) + .await? + .output + .is_none() + ); + assert_eq!(c.stats().admitted, 0); + assert_eq!(c.stats().workers, 0); + assert_eq!(p.reservations_for_test().await, (0, 0, 0)); + assert!(p.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + } + Ok(()) +} + +#[test] +fn bound_final_exact_recovery_preserves_commits_and_refuses_absence_after_custody_loss() -> Result { + // Virtual time belongs to one scenario: offsets otherwise accumulate and + // cross unrelated SDK deadlines expressed using the real monotonic clock. + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + for fault in [1, 2, 3] { + for expired in [false, true] { + tokio::runtime::Builder::new_current_thread() + .enable_all() + .build()? + .block_on(exact_case(format, fault, expired))?; + } + } + } + Ok(()) +} +async fn exact_case(format: ObjectFormat, fault: u8, expired: bool) -> Result { + let f = Fixture::new(format).await?; + let c = StagingCoordinator::new( + f.target.clone(), + StagingLimits { + bound_lifetime_ms: 1000, + ..StagingLimits::default() + }, + )?; + let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let ticket = super::bound::bind(&f, &c, [232; 16], "owner").await?; + let session = ticket.bound_session()?; + p.fault_for_test(fault); + let observer = ticket.publish(&p, ready(&session).await?)?; + drop(observer); + let StagingState::Uncertain(error) = wait_for(&ticket, |s| { + matches!( + s, + StagingState::Uncertain(_) | StagingState::Published(_) | StagingState::Fenced(_) + ) + }) + .await? + else { + return Err("final uncertainty lost".into()); + }; + let StagingError::Publication(error) = error.as_ref() else { + return Err("wrong lifecycle uncertainty".into()); + }; + let PublicationError::Push(InvocationError::Pending(evidence)) = error.as_ref() else { + return Err("wrong final evidence".into()); + }; + let original = match f.client().resolve(evidence).await? { + cellule_runtime::Resolution::Committed(value) => Some(value.commit_sequence()), + cellule_runtime::Resolution::Absent => None, + other => return Err(format!("unexpected resolution {other:?}").into()), + }; + assert_eq!(original.is_some(), fault != 1); + assert_eq!(c.stats().command_bytes, 8 << 10); + assert_eq!(p.stats().await.command_bytes, 8 << 20); + if expired { + tokio::time::pause(); + tokio::time::advance(Duration::from_millis(1001)).await; + tokio::time::resume(); + assert!(session.live_lease().is_err()); + } else { + edit(&f, "UPDATE repository_identity SET owner='replacement'; UPDATE ref_generation SET visibility='private'").await?; + } + assert_eq!(c.close_and_drain().await.len(), 1); + assert_eq!(p.close_and_drain().await.len(), 1); + let retained = c.pending([232; 16]).ok_or("lifecycle command lost")?; + c.recover(&retained)?; + let state = terminal(&retained).await?; + let StagingState::Published(outcome) = state else { + return Err(format!( + "final outcome {state:?}; format {format:?}, fault {fault}, expired {expired}" + ) + .into()); + }; + let observer = retained + .pending_publication() + .ok_or("final command observer lost")?; + match outcome { + Ok(PublicationOutcome::Push(value)) if fault != 1 => { + assert_eq!(Some(value.receipt.commit_sequence), original); + if !expired { + assert_eq!(observer.response().await?, refused()); + assert!(matches!( + replay_push_response(&f.client(), &f.target, f.begin([232; 16]), None).await, + Err(CatalogPushResponseError::Denied( + PreparationDenial::Unauthorized + )) + )); + edit(&f, "UPDATE repository_identity SET owner='owner'").await?; + } + assert_eq!(observer.response().await?, refused()); + } + Err(error) if fault == 1 && expired => assert!(matches!( + error.as_ref(), + PublicationError::Push(InvocationError::NotStarted(_)) + )), + Err(error) if fault == 1 => assert!( + matches!(error.as_ref(), PublicationError::Push(InvocationError::Rejected(value)) + if value.output == CatalogCompletionReply::Denied(PreparationDenial::Unauthorized)) + ), + other => return Err(format!("unexpected exact result {other:?}").into()), + } + assert!(session.live_lease().is_err()); + assert_eq!(c.stats().admitted, 0); + assert_eq!(p.reservations_for_test().await, (0, 0, 0)); + assert!(c.close_and_drain().await.is_empty()); + assert!(p.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn bound_final_ceiling_discards_held_proof_and_drops_result_before_worker_credit() -> Result { + use std::sync::atomic::{AtomicBool, Ordering}; + struct Owned { + c: StagingCoordinator, + dropped: Arc, + wrong: Arc, + } + impl Drop for Owned { + fn drop(&mut self) { + if self.c.stats().workers != 1 { + self.wrong.store(true, Ordering::Release); + } + self.dropped.store(true, Ordering::Release); + } + } + let f = Fixture::new(ObjectFormat::Sha256).await?; + let c = StagingCoordinator::new( + f.target.clone(), + StagingLimits { + bound_lifetime_ms: 1000, + ..StagingLimits::default() + }, + )?; + let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let ticket = super::bound::bind(&f, &c, [233; 16], "owner").await?; + let session = ticket.bound_session()?; + let input = ready(&session).await?; + let dropped = Arc::new(AtomicBool::new(false)); + let wrong = Arc::new(AtomicBool::new(false)); + let owned = Owned { + c: c.clone(), + dropped: dropped.clone(), + wrong: wrong.clone(), + }; + let (done, completed) = oneshot::channel(); + let work = ticket.spawn_bound(move |_| async move { + let _ = done.send(()); + Ok(owned) + })?; + timeout(Duration::from_secs(10), completed).await??; + let observer = ticket.publish(&p, input)?; + assert!(matches!(observer.state(), PublicationState::Held)); + tokio::time::pause(); + tokio::time::advance(Duration::from_millis(1001)).await; + tokio::time::resume(); + assert!(matches!(terminal(&ticket).await?, StagingState::Fenced(_))); + assert!(matches!(observer.wait().await, PublicationState::Discarded)); + assert!(work.wait().await.is_err()); + assert!(dropped.load(Ordering::Acquire)); + assert!(!wrong.load(Ordering::Acquire)); + assert!(session.live_lease().is_err()); + assert_eq!(super::super::completion::counts(&f.handle).await?[0], 0); + assert_eq!(p.reservations_for_test().await, (0, 0, 0)); + assert!(c.close_and_drain().await.is_empty()); + assert!(p.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn bound_final_refusals_keep_exact_ready_and_require_shared_session_and_final_kind() -> Result +{ + let f = Fixture::new(ObjectFormat::Sha256).await?; + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let ticket = super::bound::bind(&f, &c, [234; 16], "owner").await?; + let session = ticket.bound_session()?; + let unrelated = Arc::new( + PreparationSession::open( + f.client(), + f.target.clone(), + check(session.lease.token), + None, + ) + .await?, + ); + let failure = ticket + .publish(&p, ready(&unrelated).await?) + .err() + .ok_or("independent clock accepted")?; + assert!(matches!(failure.reason, StagingError::Context)); + drop(failure); + let renewal = session.ready_renew(identity()?, DEFAULT_LEASE_MS).await?; + let failure = ticket + .publish(&p, renewal) + .err() + .ok_or("nonfinal renewal accepted")?; + assert!(matches!(failure.reason, StagingError::Context)); + drop(failure); + assert_eq!(p.stats().await.admitted, 0); + let foreign = PublicationCoordinator::new( + crate::repository_target( + TenantId::from_bytes([99; 16]), + f.target.application(), + f.repository, + )?, + PublicationLimits::default(), + )?; + let failure = ticket + .publish(&foreign, ready(&session).await?) + .err() + .ok_or("foreign final coordinator accepted")?; + assert!(matches!( + failure.reason, + StagingError::PublicationAdmission(PublicationScheduleError::Foreign) + )); + assert!(matches!(ticket.state(), StagingState::Bound(_))); + let second = ready(&session).await?; + let (release, entered) = p.pause_for_test().await; + let observer = ticket.publish(&p, failure.ready)?; + let failure = ticket + .publish(&p, second) + .err() + .ok_or("duplicate final accepted")?; + assert!(matches!(failure.reason, StagingError::Duplicate)); + drop(failure); + timeout(Duration::from_secs(10), entered).await??; + release.send(()).map_err(|_| "final dispatch disappeared")?; + finished(timeout(Duration::from_secs(10), observer.wait()).await?)?; + assert!(matches!( + terminal(&ticket).await?, + StagingState::Published(Ok(_)) + )); + assert_eq!(observer.response().await?, refused()); + assert!(c.close_and_drain().await.is_empty()); + assert!(p.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn bound_final_queued_transport_rechecks_ceiling_before_initial_execution() -> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let c = StagingCoordinator::new( + f.target.clone(), + StagingLimits { + bound_lifetime_ms: 1000, + ..StagingLimits::default() + }, + )?; + let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let ticket = super::bound::bind(&f, &c, [236; 16], "owner").await?; + let session = ticket.bound_session()?; + let (release, entered) = p.pause_for_test().await; + let observer = ticket.publish(&p, ready(&session).await?)?; + timeout(Duration::from_secs(10), entered).await??; + assert!(matches!(observer.state(), PublicationState::Running)); + tokio::time::pause(); + tokio::time::advance(Duration::from_millis(1001)).await; + tokio::time::resume(); + release.send(()).map_err(|_| "transport disappeared")?; + assert!( + matches!(terminal(&ticket).await?, StagingState::Published(Err(error)) + if matches!(error.as_ref(), PublicationError::Push(InvocationError::NotStarted(_)))) + ); + assert!(observer.response().await.is_err()); + assert_eq!(super::super::completion::counts(&f.handle).await?[0], 0); + assert_eq!(p.reservations_for_test().await, (0, 0, 0)); + assert!(session.live_lease().is_err()); + assert!(c.close_and_drain().await.is_empty()); + assert!(p.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + Ok(()) +} + +#[tokio::test] +async fn bound_final_observes_shared_coordinator_recovery_without_losing_lifecycle_admission() +-> Result { + let f = Fixture::new(ObjectFormat::Sha256).await?; + let c = StagingCoordinator::new(f.target.clone(), StagingLimits::default())?; + let p = PublicationCoordinator::new(f.target.clone(), PublicationLimits::default())?; + let ticket = super::bound::bind(&f, &c, [237; 16], "owner").await?; + let session = ticket.bound_session()?; + p.fault_for_test(2); + let observer = ticket.publish(&p, ready(&session).await?)?; + assert!(matches!( + terminal(&ticket).await?, + StagingState::Uncertain(_) + )); + assert_eq!(c.stats().admitted, 1); + assert_eq!(c.close_and_drain().await.len(), 1); + assert_eq!(p.close_and_drain().await.len(), 1); + let exact = p.pending([237; 16]).await.ok_or("shared command lost")?; + exact.recover().await?; + // The lifecycle observes the shared ticket transition itself. No second + // recovery call, replacement command or caller-owned result is necessary. + assert!(matches!( + timeout( + Duration::from_secs(10), + wait_for(&ticket, |s| matches!(s, StagingState::Published(_))) + ) + .await??, + StagingState::Published(Ok(_)) + )); + assert_eq!(observer.response().await?, refused()); + assert_eq!(c.stats().admitted, 0); + assert_eq!(p.reservations_for_test().await, (0, 0, 0)); + assert!(c.close_and_drain().await.is_empty()); + assert!(p.close_and_drain().await.is_empty()); + f.runtime.shutdown().await?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/ref_state/mod.rs b/crates/canopy-server/src/packs/ref_state/mod.rs new file mode 100644 index 0000000..d3ad6c6 --- /dev/null +++ b/crates/canopy-server/src/packs/ref_state/mod.rs @@ -0,0 +1,103 @@ +//! Conditional immutable ref state; raw roots do not confer publication rights. +//! Final owner/ACL/policy/root-CAS and durable outcome publication remain Cell +//! responsibilities. This data plane is not selected by the serving path yet. +use super::directory::index::{IndexError, NodeRef, RangeCursor, RangeIndex, ReadStats}; +use crate::{ObjectFormat, PushPlan, RefExpectation}; +use canopy_object_storage::artifact::ArtifactStore; +use cellule_runtime::codec::{BoundedDecoder, BoundedEncoder, CodecError}; +use std::sync::Arc; + +mod record; +pub use record::{RefNameKey, RefStateRecord}; +mod snapshot; +mod transition; +pub use snapshot::{RefSnapshotError, RefStateSnapshot, RefStateSnapshotRoot}; +#[cfg(test)] +mod tests; + +pub type RefStateRoot = NodeRef; +pub type RefStateTree = RangeIndex; +pub const MAX_NAME_BYTES: usize = 65_535; + +#[derive(Debug, thiserror::Error)] +pub enum RefStateError { + #[error("ref state index failed")] + Index(#[from] IndexError), + #[error("ref state codec failed")] + Codec(#[from] CodecError), + #[error("ref plan shape failed")] + Shape(#[from] super::publication::RefProofError), + #[error("ref expectation changed")] + Changed, + #[error("live ref namespace conflicts")] + Namespace, +} + +pub struct RefStateIndex { + tree: RefStateTree, +} +/// Conditional path-copy result. Its plan digest is the existing canonical +/// PushPlan digest, not a substitute for membership, ancestry or policy proof. +pub struct RefTransition { + base: Option, + root: RefStateRoot, + plan_digest: [u8; 32], +} +impl RefTransition { + pub fn base(&self) -> Option { + self.base.clone() + } + pub fn root(&self) -> RefStateRoot { + self.root.clone() + } + pub fn plan_digest(&self) -> [u8; 32] { + self.plan_digest + } +} +impl RefStateIndex { + pub fn new(store: Arc, format: ObjectFormat) -> Self { + Self { + tree: RefStateTree::new(store, format), + } + } + pub fn repository(&self) -> [u8; 16] { + self.tree.repository() + } + pub fn format(&self) -> ObjectFormat { + self.tree.format() + } + pub fn stats(&self) -> ReadStats { + self.tree.stats() + } + pub fn clear_cache(&self) -> Result<(), IndexError> { + self.tree.clear_cache() + } + pub async fn read( + &self, + root: Option, + name: &str, + ) -> Result, RefStateError> { + if !crate::refs::valid_ref_name(name) { + return Err(CodecError::Invalid("ref name").into()); + } + Ok(self + .tree + .find(root, RefNameKey::new(name)?) + .await? + .map(|record| record.state)) + } + /// Seek a bounded-height ordered cursor. Live cursors skip authenticated + /// zero-weight subtrees; ordinary cursors retain deletion versions. + pub fn cursor( + &self, + root: Option, + after: Option, + live_only: bool, + ) -> Result, IndexError> { + if live_only { + self.tree.positive_cursor(root, after) + } else { + self.tree.cursor(root, after) + } + } +} diff --git a/crates/canopy-server/src/packs/ref_state/record.rs b/crates/canopy-server/src/packs/ref_state/record.rs new file mode 100644 index 0000000..60fa85d --- /dev/null +++ b/crates/canopy-server/src/packs/ref_state/record.rs @@ -0,0 +1,121 @@ +use super::*; +use crate::packs::directory::index::{IndexKey, IndexRecord, record::sealed}; + +/// A byte-ordered UTF-8 seek coordinate. Leaves separately enforce Git ref +/// syntax, so prefix bounds such as `refs/heads/topic/` are valid coordinates. +#[derive(Clone, Debug, PartialEq, Eq, PartialOrd, Ord)] +pub struct RefNameKey(Arc); +impl RefNameKey { + pub fn new(name: &str) -> Result { + if name.len() > MAX_NAME_BYTES || name.contains('\0') { + return Err(CodecError::Invalid("ref key size or NUL")); + } + Ok(Self(Arc::from(name))) + } + pub fn as_str(&self) -> &str { + &self.0 + } +} +impl sealed::Key for RefNameKey {} +impl IndexKey for RefNameKey { + fn valid(&self, _: ObjectFormat) -> bool { + self.0.len() <= MAX_NAME_BYTES && !self.0.contains('\0') + } + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + e.write_text(&self.0) + } + fn decode(d: &mut BoundedDecoder<'_>, _: ObjectFormat) -> Result { + Self::new(d.read_text()?) + } +} +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct RefStateRecord { + pub(super) name: RefNameKey, + pub(super) state: RefExpectation, +} +impl RefStateRecord { + pub fn new( + name: &str, + state: RefExpectation, + format: ObjectFormat, + ) -> Result { + let record = Self { + name: RefNameKey::new(name)?, + state, + }; + record.validate_record([0; 16], format)?; + Ok(record) + } + pub fn name(&self) -> &str { + self.name.as_str() + } + pub fn state(&self) -> &RefExpectation { + &self.state + } +} +impl sealed::Record for RefStateRecord {} +impl IndexRecord for RefStateRecord { + type Key = RefNameKey; + const FANOUT: usize = 128; + const DOMAIN: &'static [u8] = b"canopy.ref-state-index.v1\0"; + const NODE_BYTES: u32 = 512 << 10; + const MAX_HEIGHT: u8 = 16; + fn valid_counts(records: u64, live: u64) -> bool { + live <= records && records <= i64::MAX as u64 + } + fn first_key(&self) -> RefNameKey { + self.name.clone() + } + fn last_key(&self) -> RefNameKey { + self.name.clone() + } + fn object_count(&self) -> u64 { + u64::from(self.state.oid.is_some()) + } + fn validate_record(&self, _: [u8; 16], format: ObjectFormat) -> Result<(), IndexError> { + if !self.name.valid(format) + || !crate::refs::valid_ref_name(self.name()) + || self.state.version <= 0 + || self + .state + .oid + .is_some_and(|oid| oid.is_zero() || oid.format() != format) + { + return Err(IndexError::Integrity); + } + Ok(()) + } + fn encode_record(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.name.encode(e)?; + e.write_i64(self.state.version)?; + e.write_bool(self.state.oid.is_some())?; + if let Some(oid) = self.state.oid { + e.write_bytes(&oid)?; + } + Ok(()) + } + fn decode_record( + d: &mut BoundedDecoder<'_>, + repository: [u8; 16], + format: ObjectFormat, + ) -> Result { + let name = RefNameKey::decode(d, format)?; + let version = d.read_i64()?; + let oid = if d.read_bool()? { + Some( + crate::ObjectId::try_from(d.read_bytes()?) + .map_err(|_| CodecError::Invalid("ref object ID"))?, + ) + } else { + None + }; + let record = Self { + name, + state: RefExpectation { oid, version }, + }; + record + .validate_record(repository, format) + .map_err(|_| CodecError::Invalid("ref state"))?; + Ok(record) + } +} diff --git a/crates/canopy-server/src/packs/ref_state/snapshot.rs b/crates/canopy-server/src/packs/ref_state/snapshot.rs new file mode 100644 index 0000000..5f76ab7 --- /dev/null +++ b/crates/canopy-server/src/packs/ref_state/snapshot.rs @@ -0,0 +1,150 @@ +use super::*; +use crate::packs::{ + InputRootError, + directory::index::codec::{fixed, read_reference, reference}, + input_artifact::StoredInputRoot, +}; +use cellule_runtime::codec::WireValue; + +const DOMAIN: &[u8] = b"canopy.ref-state-snapshot.v1\0"; +const ROOT_BYTES: u32 = 256 << 10; +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct RefStateSnapshot { + pub repository: [u8; 16], + pub format: ObjectFormat, + pub generation: u64, + pub default_branch: String, + pub root: Option, +} +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct RefStateSnapshotRoot(StoredInputRoot); +#[derive(Debug, thiserror::Error)] +pub enum RefSnapshotError { + #[error("ref snapshot transport failed")] + Root(#[from] InputRootError), + #[error("ref snapshot codec failed")] + Codec(#[from] CodecError), + #[error("ref snapshot context differs")] + Context, +} +struct Record { + operation: [u8; 16], + snapshot: RefStateSnapshot, +} +impl RefStateSnapshotRoot { + pub fn operation(self) -> [u8; 16] { + self.0.operation + } + pub fn artifact(self) -> canopy_object_storage::artifact::ArtifactDescriptor { + self.0.artifact + } + /// Stores a descriptor, not a publication certificate or traversal proof. + pub async fn upload( + store: &ArtifactStore, + operation: [u8; 16], + snapshot: RefStateSnapshot, + ) -> Result { + if snapshot.repository != store.repository() { + return Err(RefSnapshotError::Context); + } + Ok(Self( + StoredInputRoot::upload( + store, + operation, + &Record { + operation, + snapshot, + }, + ROOT_BYTES, + ) + .await?, + )) + } + pub async fn read(self, store: &ArtifactStore) -> Result { + let record: Record = self.0.read(store, ROOT_BYTES).await?; + if record.operation != self.operation() || record.snapshot.repository != store.repository() + { + return Err(RefSnapshotError::Context); + } + Ok(record.snapshot) + } +} +impl WireValue for RefStateSnapshotRoot { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.0.validate(ROOT_BYTES)?; + self.0.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let root = StoredInputRoot::decode(d)?; + root.validate(ROOT_BYTES)?; + Ok(Self(root)) + } +} +impl Record { + fn validate(&self) -> Result<(), CodecError> { + super::super::publication::codec::artifact_valid(self.operation)?; + let snapshot = &self.snapshot; + if crate::validate_repository_id(snapshot.repository).is_err() + || snapshot.generation > i64::MAX as u64 + || snapshot.generation == 0 && snapshot.root.is_some() + || snapshot.default_branch.len() > MAX_NAME_BYTES + || !snapshot.default_branch.starts_with("refs/heads/") + || !crate::refs::valid_ref_name(&snapshot.default_branch) + { + return Err(CodecError::Invalid("ref snapshot")); + } + if let Some(root) = &snapshot.root { + root.validate(snapshot.format) + .map_err(|_| CodecError::Invalid("ref snapshot index"))?; + super::super::publication::codec::artifact_valid(root.operation)?; + } + Ok(()) + } +} +impl WireValue for Record { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.validate()?; + e.write_bytes(DOMAIN)?; + e.write_bytes(&self.operation)?; + e.write_bytes(&self.snapshot.repository)?; + e.write_u8(self.snapshot.format.bytes() as u8)?; + e.write_u64(self.snapshot.generation)?; + e.write_text(&self.snapshot.default_branch)?; + e.write_bool(self.snapshot.root.is_some())?; + if let Some(root) = &self.snapshot.root { + reference(e, root.clone())?; + } + Ok(()) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + if d.read_bytes()? != DOMAIN { + return Err(CodecError::Invalid("ref snapshot domain")); + } + let operation = fixed(d)?; + let repository = fixed(d)?; + let format = match d.read_u8()? { + 20 => ObjectFormat::Sha1, + 32 => ObjectFormat::Sha256, + _ => return Err(CodecError::Invalid("ref snapshot format")), + }; + let generation = d.read_u64()?; + let default_branch = d.read_text()?.to_owned(); + let root = if d.read_bool()? { + Some(read_reference(d, format)?) + } else { + None + }; + let record = Self { + operation, + snapshot: RefStateSnapshot { + repository, + format, + generation, + default_branch, + root, + }, + }; + record.validate()?; + Ok(record) + } +} diff --git a/crates/canopy-server/src/packs/ref_state/tests.rs b/crates/canopy-server/src/packs/ref_state/tests.rs new file mode 100644 index 0000000..18e2dae --- /dev/null +++ b/crates/canopy-server/src/packs/ref_state/tests.rs @@ -0,0 +1,1030 @@ +use super::*; +use crate::{ObjectId, RefUpdate}; +use cellule_runtime::codec::WireValue; +use futures_core::Stream; +use object_store::{ObjectStore, memory::InMemory}; +use std::{future::poll_fn, pin::Pin}; + +type Result = std::result::Result>; +fn send(value: T) -> T { + value +} +fn operation(n: u64) -> [u8; 16] { + let mut op = *b"CANOPY0100000000"; + op[8..].copy_from_slice(&n.to_be_bytes()); + op +} +fn repository() -> [u8; 16] { + let mut repo = [1; 16]; + repo[6] = 0x41; + repo[8] = 0x81; + repo +} +fn oid(n: u64, format: ObjectFormat) -> ObjectId { + let mut bytes = vec![0; format.bytes()]; + bytes[..8].copy_from_slice(&n.to_be_bytes()); + bytes.try_into().unwrap() +} +fn state(n: Option, version: i64, format: ObjectFormat) -> RefExpectation { + RefExpectation { + oid: n.map(|n| oid(n, format)), + version, + } +} +fn index(format: ObjectFormat) -> (RefStateIndex, Arc, Arc) { + let objects: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new(objects.clone(), repository())); + (RefStateIndex::new(store.clone(), format), store, objects) +} +fn plan(updates: Vec) -> PushPlan { + PushPlan { + actor: "alice".into(), + updates, + } +} +fn update(name: &str, expected: Option, new_oid: Option) -> RefUpdate { + RefUpdate { + name: name.into(), + expected, + new_oid, + } +} +async fn listed(objects: &dyn ObjectStore) -> Result> { + let mut stream = objects.list(None); + let mut metas = Vec::new(); + while let Some(meta) = poll_fn(|cx| Stream::poll_next(Pin::new(&mut stream), cx)).await { + metas.push(meta?); + } + Ok(metas) +} +async fn object_count(objects: &dyn ObjectStore) -> Result { + Ok(listed(objects).await?.len()) +} +#[tokio::test] +async fn immutable_versions_tombstones_and_expectations_for_both_formats() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (index, _, objects) = index(format); + let name = "refs/heads/main"; + let first = index + .prepare( + None, + operation(1), + &plan(vec![update(name, None, Some(oid(1, format)))]), + ) + .await? + .root(); + let same = index + .prepare( + Some(first.clone()), + operation(2), + &plan(vec![update( + name, + Some(state(Some(1), 1, format)), + Some(oid(1, format)), + )]), + ) + .await? + .root(); + assert_eq!( + index.read(Some(same.clone()), name).await?, + Some(state(Some(1), 2, format)) + ); + let deleted = index + .prepare( + Some(same), + operation(3), + &plan(vec![update(name, Some(state(Some(1), 2, format)), None)]), + ) + .await? + .root(); + assert_eq!((deleted.record_count, deleted.object_count), (1, 0)); + assert_eq!( + index.read(Some(deleted.clone()), name).await?, + Some(state(None, 3, format)) + ); + let before = object_count(&*objects).await?; + for expected in [ + None, + Some(state(Some(1), 1, format)), + Some(state(Some(1), 3, format)), + ] { + assert!(matches!( + index + .prepare( + Some(deleted.clone()), + operation(4), + &plan(vec![update(name, expected, Some(oid(2, format)))]) + ) + .await, + Err(RefStateError::Changed) + )); + } + assert_eq!(object_count(&*objects).await?, before); + let recreated = index + .prepare( + Some(deleted.clone()), + operation(4), + &plan(vec![update( + name, + Some(state(None, 3, format)), + Some(oid(2, format)), + )]), + ) + .await?; + assert_eq!(recreated.base(), Some(deleted)); + assert_eq!( + index.read(Some(recreated.root()), name).await?, + Some(state(Some(2), 4, format)) + ); + index.clear_cache()?; + assert_eq!( + index.read(Some(first), name).await?, + Some(state(Some(1), 1, format)) + ); + let terminal = index + .tree + .build_sorted( + operation(5), + [RefStateRecord::new( + name, + state(Some(1), i64::MAX, format), + format, + )], + ) + .await? + .ok_or("terminal")?; + assert!(matches!( + index + .prepare( + Some(terminal), + operation(6), + &plan(vec![update( + name, + Some(state(Some(1), i64::MAX, format)), + None + )]) + ) + .await, + Err(RefStateError::Shape(_)) + )); + } + Ok(()) +} +#[tokio::test] +async fn atomic_namespace_swaps_and_unicode_are_validated_before_uploads() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (index, _, objects) = index(format); + let parent = "refs/heads/équipe"; + let child = "refs/heads/équipe/東京"; + let root = index + .prepare( + None, + operation(1), + &plan(vec![update(parent, None, Some(oid(1, format)))]), + ) + .await? + .root(); + let before = object_count(&*objects).await?; + assert!(matches!( + index + .prepare( + Some(root.clone()), + operation(2), + &plan(vec![update(child, None, Some(oid(2, format)))]) + ) + .await, + Err(RefStateError::Namespace) + )); + assert!(matches!( + index + .prepare( + None, + operation(2), + &plan(vec![ + update(parent, None, Some(oid(1, format))), + update(child, None, Some(oid(2, format))) + ]) + ) + .await, + Err(RefStateError::Namespace) + )); + assert_eq!(object_count(&*objects).await?, before); + let swapped = index + .prepare( + Some(root), + operation(2), + &plan(vec![ + update(child, None, Some(oid(2, format))), + update(parent, Some(state(Some(1), 1, format)), None), + ]), + ) + .await? + .root(); + assert_eq!((swapped.record_count, swapped.object_count), (2, 1)); + let before = object_count(&*objects).await?; + assert!(matches!( + index + .prepare( + Some(swapped.clone()), + operation(3), + &plan(vec![update( + parent, + Some(state(None, 2, format)), + Some(oid(3, format)) + )]) + ) + .await, + Err(RefStateError::Namespace) + )); + assert_eq!(object_count(&*objects).await?, before); + let restored = index + .prepare( + Some(swapped), + operation(3), + &plan(vec![ + update(parent, Some(state(None, 2, format)), Some(oid(3, format))), + update(child, Some(state(Some(2), 1, format)), None), + ]), + ) + .await? + .root(); + let mut live = index.cursor(Some(restored), None, true)?; + assert_eq!(live.next().await?.ok_or("live")?.name(), parent); + assert!(live.next().await?.is_none()); + } + Ok(()) +} +#[tokio::test] +async fn initial_twenty_thousand_ref_plan_builds_by_nodes_and_seeks_one_path() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (index, _, objects) = index(format); + let names: Vec<_> = (0..20_000) + .map(|n| format!("refs/heads/{n:05}/{}", "x".repeat(200))) + .collect(); + let initial = plan( + names + .iter() + .map(|name| update(name, None, Some(oid(1, format)))) + .collect(), + ); + let mut encoded_bytes = 0; + for start in (0..initial.updates.len()).step_by(32) { + let mut encoded = BoundedEncoder::new(64 << 10)?; + initial.encode_range(start..(start + 32).min(initial.updates.len()), &mut encoded)?; + encoded_bytes += encoded.finish().len(); + } + assert!(encoded_bytes > 4 << 20); + let transition = index.prepare(None, operation(1), &initial).await?; + assert_eq!( + transition.plan_digest(), + crate::packs::publication::ref_proof::plan_digest(&initial)? + ); + let root = transition.root(); + assert_eq!( + (root.record_count, root.object_count, root.height), + (20_000, 20_000, 2) + ); + assert!( + object_count(&*objects).await? <= 500, + "construction must scale with nodes, not members" + ); + index.clear_cache()?; + let before = index.stats(); + assert_eq!( + index.read(Some(root.clone()), &names[17_123]).await?, + Some(state(Some(1), 1, format)) + ); + assert_eq!( + index.stats().loaded_nodes - before.loaded_nodes, + u64::from(root.height) + 1 + ); + index.clear_cache()?; + let before = index.stats(); + let mut cursor = index.cursor( + Some(root.clone()), + Some(RefNameKey::new(&names[17_122])?), + true, + )?; + assert_eq!(cursor.next().await?.ok_or("seek")?.name(), names[17_123]); + assert_eq!( + index.stats().loaded_nodes - before.loaded_nodes, + u64::from(root.height) + 1 + ); + drop(cursor); + let before = object_count(&*objects).await?; + let changed = index + .prepare( + Some(root.clone()), + operation(2), + &plan(vec![update( + &names[17_123], + Some(state(Some(1), 1, format)), + Some(oid(2, format)), + )]), + ) + .await? + .root(); + assert!(object_count(&*objects).await? - before <= 2 * (usize::from(root.height) + 1)); + index.clear_cache()?; + assert_eq!( + index.read(Some(root), &names[17_123]).await?, + Some(state(Some(1), 1, format)) + ); + assert_eq!( + index.read(Some(changed.clone()), &names[17_123]).await?, + Some(state(Some(2), 2, format)) + ); + let bulk = plan( + names + .iter() + .enumerate() + .map(|(n, name)| { + update( + name, + Some(state( + Some(if n == 17_123 { 2 } else { 1 }), + if n == 17_123 { 2 } else { 1 }, + format, + )), + Some(oid(3, format)), + ) + }) + .collect(), + ); + index.clear_cache()?; + let reads = index.stats().loaded_nodes; + let writes = object_count(&*objects).await?; + let bulk_root = index + .prepare(Some(changed), operation(3), &bulk) + .await? + .root(); + assert_eq!( + (bulk_root.record_count, bulk_root.object_count), + (20_000, 20_000) + ); + assert!( + index.stats().loaded_nodes - reads <= 600, + "validate and rewrite by nodes, not by update paths" + ); + assert!( + object_count(&*objects).await? - writes <= 500, + "rewrite the complete existing large plan in bounded groups" + ); + let mut cursor = index.cursor(Some(bulk_root), None, false)?; + for (n, name) in names.iter().enumerate() { + let actual = cursor.next().await?.ok_or("bulk ref")?; + assert_eq!(actual.name(), name); + assert_eq!( + actual.state(), + &state(Some(3), if n == 17_123 { 3 } else { 2 }, format) + ); + } + assert!(cursor.next().await?.is_none()); + } + Ok(()) +} +#[tokio::test] +async fn live_seek_skips_dead_subtrees_without_losing_later_siblings() -> Result { + let format = ObjectFormat::Sha256; + let (index, _, _) = index(format); + let name = |n| format!("refs/heads/{n:05}"); + let root = index + .tree + .build_sorted( + operation(1), + (0..40_000).map(|n| { + RefStateRecord::new( + &name(n), + state( + if n == 0 || n == 39_999 { Some(1) } else { None }, + 1, + format, + ), + format, + ) + }), + ) + .await? + .ok_or("root")?; + assert_eq!((root.height, root.object_count), (2, 2)); + for seek in [1, 200, 16_000, 16_384, 20_000, 39_998] { + index.clear_cache()?; + let before = index.stats(); + let mut cursor = index.cursor( + Some(root.clone()), + Some(RefNameKey::new(&name(seek))?), + true, + )?; + assert_eq!( + cursor + .next() + .await? + .ok_or("later live sibling lost")? + .name(), + name(39_999) + ); + assert!(cursor.next().await?.is_none()); + assert!( + index.stats().loaded_nodes - before.loaded_nodes <= 2 * (u64::from(root.height) + 1) + ); + } + let mut forged = root.clone(); + forged.object_count = 0; + index.clear_cache()?; + assert!( + index + .cursor(Some(forged), None, true)? + .next() + .await + .is_err() + ); + let mut cursor = index.cursor(Some(root), Some(RefNameKey::new(&name(39_999))?), true)?; + assert!(cursor.next().await?.is_none()); + Ok(()) +} +#[tokio::test] +async fn long_names_split_by_bytes_and_snapshot_roundtrips_large_fences() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (index, store, objects) = index(format); + let names: Vec<_> = (0..40) + .map(|n| { + let prefix = format!("refs/heads/{n:03}/"); + format!("{prefix}{}", "x".repeat(MAX_NAME_BYTES - prefix.len())) + }) + .collect(); + let mut root = None; + for name in &names { + root = Some( + index + .prepare( + root, + operation(1), + &plan(vec![update(name, None, Some(oid(1, format)))]), + ) + .await? + .root(), + ); + } + let root = root.ok_or("root")?; + assert!(root.height >= 2); + let mut cursor = index.cursor(Some(root.clone()), None, false)?; + for name in &names { + assert_eq!(cursor.next().await?.ok_or("name")?.name(), name); + } + assert!(cursor.next().await?.is_none()); + for object in listed(&*objects).await? { + assert!(object.size <= 512 << 10); + } + let snapshot = RefStateSnapshot { + repository: repository(), + format, + generation: 1, + default_branch: names[0].clone(), + root: Some(root), + }; + let saved = RefStateSnapshotRoot::upload(&store, operation(2), snapshot.clone()).await?; + assert!(saved.artifact().size > 128 << 10); + assert!(saved.artifact().size <= 256 << 10); + let mut e = BoundedEncoder::new(128)?; + saved.encode(&mut e)?; + let bytes = e.finish(); + let mut d = BoundedDecoder::new(&bytes, 128)?; + let decoded = RefStateSnapshotRoot::decode(&mut d)?; + d.finish()?; + let mut wire = BoundedDecoder::new(&bytes, 128)?; + assert!(crate::packs::wire_request::WireRequestRoot::decode(&mut wire).is_err()); + let mut native = BoundedDecoder::new(&bytes, 128)?; + assert!(crate::packs::publication::NativeResultRoot::decode(&mut native).is_err()); + assert_eq!(decoded.read(&store).await?, snapshot); + let wrong = ArtifactStore::new(objects.clone(), [2; 16]); + assert!(decoded.read(&wrong).await.is_err()); + let mut forged_bytes = bytes.clone(); + forged_bytes[32] ^= 1; + let mut d = BoundedDecoder::new(&forged_bytes, 128)?; + let forged = RefStateSnapshotRoot::decode(&mut d)?; + d.finish()?; + assert!(forged.read(&store).await.is_err()); + let too_long = format!("{}x", names[0]); + assert!(RefNameKey::new(&too_long).is_err()); + let mut wrong_generation = snapshot.clone(); + wrong_generation.generation = 0; + assert!( + RefStateSnapshotRoot::upload(&store, operation(2), wrong_generation) + .await + .is_err() + ); + } + Ok(()) +} +#[tokio::test] +async fn sorted_builder_rejects_duplicates_disorder_and_zero_live_cursor_authenticates() -> Result { + let format = ObjectFormat::Sha1; + let (index, _, _) = index(format); + let a = RefStateRecord::new("refs/heads/a", state(None, 1, format), format)?; + let b = RefStateRecord::new("refs/heads/b", state(None, 2, format), format)?; + for records in [vec![a.clone(), a.clone()], vec![b.clone(), a.clone()]] { + assert!(matches!( + index + .tree + .build_sorted(operation(1), records.into_iter().map(Ok)) + .await, + Err(IndexError::RangeOverlap) + )); + } + assert!( + index + .tree + .build_sorted(operation(1), std::iter::empty()) + .await? + .is_none() + ); + let root = index + .tree + .build_sorted(operation(1), [Ok(a.clone()), Ok(b)]) + .await? + .ok_or("root")?; + assert_eq!((root.record_count, root.object_count), (2, 0)); + index.clear_cache()?; + let before = index.stats(); + assert!( + index + .cursor(Some(root.clone()), None, true)? + .next() + .await? + .is_none() + ); + assert_eq!(index.stats().loaded_nodes - before.loaded_nodes, 1); + assert_eq!( + index.cursor(Some(root), None, false)?.next().await?, + Some(a) + ); + Ok(()) +} + +#[tokio::test] +async fn namespace_check_finds_conflict_after_exhausted_positive_branch() -> Result { + let format = ObjectFormat::Sha256; + let (index, _, objects) = index(format); + let records = (0..40_000).map(|n| { + let name = format!("refs/heads/{}/{n:05}", if n < 200 { "aa" } else { "zz" }); + RefStateRecord::new( + &name, + state( + if n == 0 || n == 39_999 { Some(1) } else { None }, + 1, + format, + ), + format, + ) + }); + let root = index + .tree + .build_sorted(operation(1), records) + .await? + .ok_or("root")?; + index.clear_cache()?; + let before = object_count(&*objects).await?; + assert!(matches!( + index + .prepare( + Some(root), + operation(2), + &plan(vec![update("refs/heads/zz", None, Some(oid(1, format)))]) + ) + .await, + Err(RefStateError::Namespace) + )); + assert_eq!(object_count(&*objects).await?, before); + Ok(()) +} +#[tokio::test] +async fn snapshot_purpose_and_tree_format_are_authenticated() -> Result { + let format = ObjectFormat::Sha256; + let (index, store, _) = index(format); + let root = index + .prepare( + None, + operation(1), + &plan(vec![update("refs/heads/main", None, Some(oid(1, format)))]), + ) + .await? + .root(); + let foreign_format = RefStateIndex::new(store.clone(), ObjectFormat::Sha1); + assert!( + foreign_format + .read(Some(root.clone()), "refs/heads/main") + .await + .is_err() + ); + let snapshot = RefStateSnapshot { + repository: repository(), + format, + generation: 1, + default_branch: "refs/heads/main".into(), + root: Some(root), + }; + let saved = RefStateSnapshotRoot::upload(&store, operation(2), snapshot).await?; + let mut e = BoundedEncoder::new(128)?; + saved.encode(&mut e)?; + let bytes = e.finish(); + let mut d = BoundedDecoder::new(&bytes, 128)?; + let wrong_purpose = crate::packs::wire_request::WireRequestRoot::decode(&mut d)?; + d.finish()?; + assert!(wrong_purpose.read(&store).await.is_err()); + Ok(()) +} + +#[tokio::test] +async fn existing_batch_copies_changed_subtrees_once_and_preserves_all_versions() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (index, _, objects) = index(format); + let name = |n| format!("refs/heads/{n:05}"); + let original = index + .tree + .build_sorted( + operation(1), + (0..20_000) + .map(|n| RefStateRecord::new(&name(n), state(Some(1), 1, format), format)), + ) + .await? + .ok_or("original")?; + index.clear_cache()?; + let reads = index.stats().loaded_nodes; + let writes = object_count(&*objects).await?; + // Reversed intent exercises sorting without changing the canonical digest. + let changes = plan( + (10_000..10_512) + .rev() + .map(|n| { + update( + &name(n), + Some(state(Some(1), 1, format)), + if n % 2 == 0 { + None + } else { + Some(oid(2, format)) + }, + ) + }) + .collect(), + ); + let transition = index + .prepare(Some(original.clone()), operation(2), &changes) + .await?; + assert_eq!(transition.base(), Some(original.clone())); + assert_eq!( + transition.plan_digest(), + crate::packs::publication::ref_proof::plan_digest(&changes)? + ); + let current = transition.root(); + assert_eq!( + (current.record_count, current.object_count), + (20_000, 19_744) + ); + assert!( + object_count(&*objects).await? - writes <= 40, + "one node group per changed subtree, not one path per update" + ); + assert!( + index.stats().loaded_nodes - reads <= 40, + "sorted validation and rewriting must not reload history" + ); + for root in [original, current] { + let current = root.operation == operation(2); + let mut cursor = index.cursor(Some(root), None, false)?; + for n in 0..20_000 { + let actual = cursor.next().await?.ok_or("missing ref")?; + assert_eq!(actual.name(), name(n)); + let changed = current && (10_000..10_512).contains(&n); + assert_eq!( + actual.state(), + &state( + if changed && n % 2 == 0 { + None + } else if changed { + Some(2) + } else { + Some(1) + }, + if changed { 2 } else { 1 }, + format + ) + ); + } + assert!(cursor.next().await?.is_none()); + } + } + Ok(()) +} + +#[tokio::test] +async fn sparse_batch_reuses_higher_subtrees_and_inserts_before_between_and_after() -> Result { + let format = ObjectFormat::Sha256; + let (index, _, objects) = index(format); + let name = |n| format!("refs/heads/m/{n:06}"); + let old = index + .tree + .build_sorted( + operation(1), + (0..100_000).map(|n| RefStateRecord::new(&name(n), state(Some(1), 1, format), format)), + ) + .await? + .ok_or("old")?; + index.clear_cache()?; + let reads = index.stats().loaded_nodes; + let writes = object_count(&*objects).await?; + let changes = plan(vec![ + update("refs/heads/z", None, Some(oid(3, format))), + update(&name(40_000), Some(state(Some(1), 1, format)), None), + update("refs/heads/m/040000/topic", None, Some(oid(4, format))), + update( + &name(50_000), + Some(state(Some(1), 1, format)), + Some(oid(2, format)), + ), + update("refs/heads/m/050000x", None, Some(oid(5, format))), + update("refs/heads/a", None, Some(oid(6, format))), + ]); + let current = send(index.prepare(Some(old.clone()), operation(2), &changes)) + .await? + .root(); + assert_eq!( + (current.record_count, current.object_count), + (100_004, 100_003) + ); + assert!( + index.stats().loaded_nodes - reads <= 64, + "sparse preparation must not read 100,000 records" + ); + assert!( + object_count(&*objects).await? - writes <= 100, + "reuse unchanged higher subtrees" + ); + let mut expected: std::collections::BTreeMap<_, _> = (0..100_000) + .map(|n| (name(n), state(Some(1), 1, format))) + .collect(); + for update in &changes.updates { + expected.insert( + update.name.clone(), + RefExpectation { + oid: update.new_oid, + version: if update.expected.is_some() { 2 } else { 1 }, + }, + ); + } + let mut cursor = index.cursor(Some(current), None, false)?; + for (name, expected) in expected { + let actual = cursor.next().await?.ok_or("actual inventory exhausted")?; + assert_eq!(actual.name(), name); + assert_eq!(actual.state(), &expected); + } + assert!(cursor.next().await?.is_none()); + assert_eq!( + index.read(Some(old), &name(40_000)).await?, + Some(state(Some(1), 1, format)) + ); + Ok(()) +} + +#[tokio::test] +async fn sorted_rewrite_rejects_late_input_errors_and_retries_immutable_artifacts() -> Result { + let format = ObjectFormat::Sha1; + let (index, _, objects) = index(format); + let name = |n| format!("refs/heads/{n:05}"); + let old = index + .tree + .build_sorted( + operation(1), + (0..1_000) + .map(|n| RefStateRecord::new(&name(2 * n), state(Some(1), 1, format), format)), + ) + .await? + .ok_or("old")?; + let record = |n| RefStateRecord::new(&name(2 * n + 1), state(Some(2), 1, format), format); + let before = object_count(&*objects).await?; + let broken = (0..500) + .map(record) + .chain(std::iter::once(Err(IndexError::Integrity))); + assert!(matches!( + index + .tree + .upsert_sorted(Some(old.clone()), operation(2), broken) + .await, + Err(IndexError::Integrity) + )); + assert!( + object_count(&*objects).await? > before, + "late failure exercises already emitted immutable nodes" + ); + let root = index + .tree + .upsert_sorted(Some(old.clone()), operation(2), (0..1_000).map(record)) + .await? + .ok_or("merged")?; + let complete = object_count(&*objects).await?; + assert_eq!( + index + .tree + .upsert_sorted(Some(old.clone()), operation(2), (0..1_000).map(record)) + .await?, + Some(root.clone()) + ); + assert_eq!(object_count(&*objects).await?, complete); + let mut cursor = index.cursor(Some(root), None, false)?; + for n in 0..2_000 { + let actual = cursor.next().await?.ok_or("merged ref")?; + assert_eq!(actual.name(), name(n)); + assert_eq!( + actual.state(), + &state(Some(if n % 2 == 0 { 1 } else { 2 }), 1, format) + ); + } + assert!(cursor.next().await?.is_none()); + assert_eq!( + index + .tree + .upsert_sorted(Some(old.clone()), operation(3), std::iter::empty()) + .await?, + Some(old.clone()) + ); + let unsorted = [record(900), record(1)]; + assert!(matches!( + index + .tree + .upsert_sorted(Some(old), operation(4), unsorted) + .await, + Err(IndexError::RangeOverlap) + )); + Ok(()) +} + +#[tokio::test] +async fn long_name_batch_preserves_byte_bounds_for_changed_and_reused_levels() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (index, _, objects) = index(format); + let name = |n| { + let prefix = format!("refs/heads/{n:03}/"); + format!("{prefix}{}", "x".repeat(MAX_NAME_BYTES - prefix.len())) + }; + let old = index + .tree + .build_sorted( + operation(1), + (0..64) + .map(|n| RefStateRecord::new(&name(2 * n), state(Some(1), 1, format), format)), + ) + .await? + .ok_or("old")?; + let mut updates: Vec<_> = (0..64) + .step_by(3) + .map(|n| { + update( + &name(2 * n), + Some(state(Some(1), 1, format)), + if n % 2 == 0 { + None + } else { + Some(oid(2, format)) + }, + ) + }) + .collect(); + updates.extend( + (0..64) + .step_by(5) + .map(|n| update(&name(2 * n + 1), None, Some(oid(3, format)))), + ); + updates.reverse(); + let changed = plan(updates); + let current = index + .prepare(Some(old.clone()), operation(2), &changed) + .await? + .root(); + for meta in listed(&*objects).await? { + assert!(meta.size <= 512 << 10); + } + let mut expected: std::collections::BTreeMap<_, _> = (0..64) + .map(|n| (name(2 * n), state(Some(1), 1, format))) + .collect(); + for update in &changed.updates { + expected.insert( + update.name.clone(), + RefExpectation { + oid: update.new_oid, + version: if update.expected.is_some() { 2 } else { 1 }, + }, + ); + } + assert_eq!(expected.len(), 77); + let mut cursor = index.cursor(Some(current), None, false)?; + for (name, expected) in expected { + let actual = cursor.next().await?.ok_or("long-name inventory")?; + assert_eq!(actual.name(), name); + assert_eq!(actual.state(), &expected); + } + assert!(cursor.next().await?.is_none()); + assert_eq!( + index.read(Some(old), &name(0)).await?, + Some(state(Some(1), 1, format)) + ); + } + Ok(()) +} + +#[tokio::test] +async fn repeated_left_edge_inserts_keep_the_tree_dense() -> Result { + let format = ObjectFormat::Sha1; + let (index, _, _) = index(format); + let mut root = index + .tree + .build_sorted( + operation(1), + (0..512).map(|n| { + RefStateRecord::new( + &format!("refs/heads/m/{n:05}"), + state(Some(1), 1, format), + format, + ) + }), + ) + .await? + .ok_or("base")?; + for (step, n) in (0..32).rev().enumerate() { + root = index + .prepare( + Some(root), + operation(2 + step as u64), + &plan(vec![update( + &format!("refs/heads/a/{n:05}"), + None, + Some(oid(2, format)), + )]), + ) + .await? + .root(); + } + assert_eq!((root.record_count, root.object_count), (544, 544)); + index.clear_cache()?; + let before = index.stats().loaded_nodes; + let mut cursor = index.cursor(Some(root), None, false)?; + let mut count = 0; + while cursor.next().await?.is_some() { + count += 1; + } + assert_eq!(count, 544); + let reads = index.stats().loaded_nodes - before; + assert!( + reads <= 11, + "repeated prefix insertions left too many underfilled nodes: {reads}" + ); + Ok(()) +} + +#[tokio::test] +async fn repeated_prefix_inserts_balance_internal_tail_groups() -> Result { + let format = ObjectFormat::Sha1; + let (index, _, _) = index(format); + let mut root = index + .tree + .build_sorted( + operation(1), + (0..16_384).map(|n| { + RefStateRecord::new( + &format!("refs/heads/m/{n:05}"), + state(Some(1), 1, format), + format, + ) + }), + ) + .await? + .ok_or("base")?; + for (step, n) in (0..640).rev().enumerate() { + root = index + .prepare( + Some(root), + operation(2 + step as u64), + &plan(vec![update( + &format!("refs/heads/a/{n:05}"), + None, + Some(oid(2, format)), + )]), + ) + .await? + .root(); + } + assert_eq!((root.record_count, root.height), (17_024, 2)); + index.clear_cache()?; + let before = index.stats().loaded_nodes; + let mut cursor = index.cursor(Some(root), None, false)?; + let mut count = 0; + while cursor.next().await?.is_some() { + count += 1; + } + assert_eq!(count, 17_024); + let reads = index.stats().loaded_nodes - before; + assert!( + reads <= 145, + "internal split tails fragmented the tree: {reads}" + ); + Ok(()) +} diff --git a/crates/canopy-server/src/packs/ref_state/transition.rs b/crates/canopy-server/src/packs/ref_state/transition.rs new file mode 100644 index 0000000..bace6e8 --- /dev/null +++ b/crates/canopy-server/src/packs/ref_state/transition.rs @@ -0,0 +1,110 @@ +use super::*; +use std::collections::{BTreeMap, btree_map::IntoValues}; + +// A named streaming adapter keeps borrowed update lifetimes explicit across +// recursive await boundaries; no second plan/record inventory is allocated. +struct Records<'a> { + updates: IntoValues<&'a str, &'a crate::RefUpdate>, + format: ObjectFormat, +} +impl Iterator for Records<'_> { + type Item = Result; + fn next(&mut self) -> Option { + let update = self.updates.next()?; + Some(RefStateRecord::new( + &update.name, + RefExpectation { + oid: update.new_oid, + version: update.expected.as_ref().map_or(1, |old| old.version + 1), + }, + self.format, + )) + } +} + +impl RefStateIndex { + /// Validate the whole plan before writing any tree nodes. The selected + /// root must come from certified current state before final publication. + pub async fn prepare( + &self, + base: Option, + operation: [u8; 16], + plan: &PushPlan, + ) -> Result { + super::super::publication::codec::artifact_valid(operation)?; + super::super::publication::ref_proof::shape(plan, self.format())?; + if let Some(root) = &base { + self.tree.validate_root(root.clone()).await?; + } + let updates: BTreeMap<&str, _> = plan + .updates + .iter() + .map(|update| (update.name.as_str(), update)) + .collect(); + for update in updates.values() { + RefNameKey::new(&update.name)?; + if self.read(base.clone(), &update.name).await? != update.expected { + return Err(RefStateError::Changed); + } + } + for update in updates.values().filter(|update| update.new_oid.is_some()) { + let name = update.name.as_str(); + for (at, _) in name.match_indices('/').filter(|(at, _)| *at > 4) { + let ancestor = &name[..at]; + let live = if let Some(update) = updates.get(ancestor) { + update.new_oid.is_some() + } else { + self.read(base.clone(), ancestor) + .await? + .is_some_and(|state| state.oid.is_some()) + }; + if live { + return Err(RefStateError::Namespace); + } + } + let prefix = format!("{name}/"); + let end = format!("{name}0"); + if updates + .range(prefix.as_str()..end.as_str()) + .any(|(_, update)| update.new_oid.is_some()) + { + return Err(RefStateError::Namespace); + } + if name.len() == MAX_NAME_BYTES { + continue; + } + // The slash prefix cannot itself be a leaf. Seek directly to its + // interval, avoiding unrelated names between `name` and `name/`. + let mut cursor = self.cursor(base.clone(), Some(RefNameKey::new(&prefix)?), true)?; + while let Some(existing) = cursor.next().await? { + if existing.name() >= end.as_str() { + break; + } + if existing.name() < prefix.as_str() { + continue; + } + if updates + .get(existing.name()) + .is_none_or(|update| update.new_oid.is_some()) + { + return Err(RefStateError::Namespace); + } + } + } + let plan_digest = super::super::publication::ref_proof::plan_digest(plan)?; + let records = Records { + updates: updates.into_values(), + format: self.format(), + }; + let root = self + .tree + .upsert_sorted(base.clone(), operation, records) + .await? + .ok_or(RefStateError::Changed)?; + Ok(RefTransition { + base, + root, + plan_digest, + }) + } +} diff --git a/crates/canopy-server/src/packs/sources/codec.rs b/crates/canopy-server/src/packs/sources/codec.rs new file mode 100644 index 0000000..2f9bf70 --- /dev/null +++ b/crates/canopy-server/src/packs/sources/codec.rs @@ -0,0 +1,110 @@ +use super::*; +use cellule_runtime::codec::{BoundedDecoder, BoundedEncoder, CodecError}; +use index::{ + codec::{artifact, fixed, read_artifact}, + record::sealed, +}; + +impl sealed::Key for SegmentKey {} +impl IndexKey for SegmentKey { + fn valid(&self, _format: ObjectFormat) -> bool { + true + } + fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { + let mut bytes = [0; 48]; + bytes[..16].copy_from_slice(&self.operation); + bytes[16..].copy_from_slice(&self.digest); + encoder.write_bytes(&bytes) + } + fn decode(decoder: &mut BoundedDecoder<'_>, _format: ObjectFormat) -> Result { + let bytes: [u8; 48] = fixed(decoder)?; + Ok(Self { + operation: bytes[..16] + .try_into() + .map_err(|_| CodecError::Invalid("source operation"))?, + digest: bytes[16..] + .try_into() + .map_err(|_| CodecError::Invalid("source digest"))?, + }) + } +} +impl sealed::Record for SourceRecord {} +impl IndexRecord for SourceRecord { + type Key = SegmentKey; + const FANOUT: usize = SOURCE_FANOUT; + const DOMAIN: &'static [u8] = b"canopy.source-index.v1\0"; + fn first_key(&self) -> SegmentKey { + self.key() + } + fn last_key(&self) -> SegmentKey { + self.key() + } + fn object_count(&self) -> u64 { + u64::from(self.metadata.segment.identity.object_count) + } + fn validate_record( + &self, + repository: [u8; 16], + format: ObjectFormat, + ) -> Result<(), IndexError> { + self.validate(repository, format) + } + fn encode_record(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { + let segment = self.metadata.segment; + let identity = segment.identity; + encoder.write_bytes(&identity.operation)?; + encoder.write_bytes(&identity.git_checksum)?; + encoder.write_u32(identity.first_ordinal)?; + encoder.write_u32(identity.object_count)?; + encoder.write_u64(segment.edge_count)?; + encoder.write_bytes(&segment.inventory_digest)?; + segment.first_oid.encode(encoder)?; + segment.last_oid.encode(encoder)?; + artifact(encoder, self.metadata.artifact)?; + artifact(encoder, self.pack)?; + artifact(encoder, self.index)?; + encoder.write_u32(self.pack_object_count) + } + fn decode_record( + decoder: &mut BoundedDecoder<'_>, + repository: [u8; 16], + format: ObjectFormat, + ) -> Result { + let operation = fixed(decoder)?; + let git_checksum = ObjectId::decode(decoder, format)?; + let first_ordinal = decoder.read_u32()?; + let object_count = decoder.read_u32()?; + let edge_count = decoder.read_u64()?; + let inventory_digest = fixed(decoder)?; + let first_oid = ObjectId::decode(decoder, format)?; + let last_oid = ObjectId::decode(decoder, format)?; + let metadata = read_artifact(decoder)?; + let pack = read_artifact(decoder)?; + let index = read_artifact(decoder)?; + Ok(Self { + metadata: StoredSegment { + segment: SegmentDescriptor { + identity: SegmentIdentity { + repository, + operation, + format, + pack_digest: pack.digest, + git_checksum, + first_ordinal, + object_count, + }, + edge_count, + inventory_digest, + first_oid, + last_oid, + size: metadata.size, + digest: metadata.digest, + }, + artifact: metadata, + }, + pack, + index, + pack_object_count: decoder.read_u32()?, + }) + } +} diff --git a/crates/canopy-server/src/packs/sources/inputs.rs b/crates/canopy-server/src/packs/sources/inputs.rs new file mode 100644 index 0000000..52b8c2f --- /dev/null +++ b/crates/canopy-server/src/packs/sources/inputs.rs @@ -0,0 +1,60 @@ +//! Creating input inventory reuses the persistent range index, independently +//! of canonical source shards. Membership is custody, never publication proof. +use super::*; +use cellule_runtime::codec::{BoundedDecoder, BoundedEncoder, CodecError}; +use index::{ + codec::{artifact, fixed, read_artifact}, + record::sealed, +}; + +impl sealed::Record for NativePackDescriptor {} +impl IndexRecord for NativePackDescriptor { + type Key = SegmentKey; + const FANOUT: usize = SOURCE_FANOUT; + const DOMAIN: &'static [u8] = b"canopy.native-input-index.v1\0"; + fn first_key(&self) -> SegmentKey { + SegmentKey { + operation: self.operation, + digest: self.pack.digest, + } + } + fn last_key(&self) -> SegmentKey { + self.first_key() + } + fn object_count(&self) -> u64 { + u64::from(self.object_count) + } + fn validate_record( + &self, + repository: [u8; 16], + format: ObjectFormat, + ) -> Result<(), IndexError> { + self.validate(repository, format)?; + if self.pack.manifest_digest == [0; 32] || self.index.manifest_digest == [0; 32] { + return Err(IndexError::Integrity); + } + Ok(()) + } + fn encode_record(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + e.write_bytes(&self.operation)?; + self.git_checksum.encode(e)?; + e.write_u32(self.object_count)?; + artifact(e, self.pack)?; + artifact(e, self.index) + } + fn decode_record( + d: &mut BoundedDecoder<'_>, + repository: [u8; 16], + format: ObjectFormat, + ) -> Result { + Ok(Self { + repository, + operation: fixed(d)?, + format, + git_checksum: ObjectId::decode(d, format)?, + object_count: d.read_u32()?, + pack: read_artifact(d)?, + index: read_artifact(d)?, + }) + } +} diff --git a/crates/canopy-server/src/packs/sources/mod.rs b/crates/canopy-server/src/packs/sources/mod.rs new file mode 100644 index 0000000..82d096a --- /dev/null +++ b/crates/canopy-server/src/packs/sources/mod.rs @@ -0,0 +1,110 @@ +//! Immutable source bindings share the directory's bounded path-copy tree. +//! Membership binds artifact incarnations; canonical bodies, typed closure, +//! authorization, retained-root pins and Cell publication remain separate proofs. + +use super::{ + directory::{ + SegmentKey, + index::{self, IndexError, IndexKey, IndexRecord, NodeRef, RangeIndex}, + }, + metadata::{SegmentDescriptor, SegmentIdentity, StoredSegment}, +}; +use crate::{ObjectFormat, ObjectId}; +use canopy_object_storage::artifact::{ArtifactDescriptor, ArtifactKey, ArtifactKind}; + +mod codec; +mod verification; +pub use verification::{PackCoverage, VerifiedPackBinding}; +mod native; +pub use native::NativePackDescriptor; +mod inputs; +pub type NativeInputIndex = RangeIndex; +pub type NativeInputRoot = NodeRef; +mod resolve; +pub use resolve::{ResolvedSource, SourceLoader}; + +pub type SourceIndex = RangeIndex; +pub type SourceRoot = NodeRef; +pub const SOURCE_FANOUT: usize = 128; + +/// One metadata shard and its exact native pack/index artifact incarnations. +/// Several shards may bind the same pack; coverage requires an exact partition. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct SourceRecord { + pub metadata: StoredSegment, + pub pack: ArtifactDescriptor, + pub index: ArtifactDescriptor, + pub pack_object_count: u32, +} +impl SourceRecord { + /// Shared pack/index context, available before metadata shards are built. + /// This is a runtime value; the source/catalog wire contract is unchanged. + pub fn native(self) -> NativePackDescriptor { + let identity = self.metadata.segment.identity; + NativePackDescriptor { + repository: identity.repository, + operation: identity.operation, + format: identity.format, + git_checksum: identity.git_checksum, + object_count: self.pack_object_count, + pack: self.pack, + index: self.index, + } + } + pub fn key(self) -> SegmentKey { + SegmentKey { + operation: self.metadata.segment.identity.operation, + digest: self.metadata.artifact.digest, + } + } + pub fn pack_key(self) -> ArtifactKey { + self.artifact_key(ArtifactKind::Pack) + } + pub fn index_key(self) -> ArtifactKey { + self.artifact_key(ArtifactKind::Index) + } + pub fn metadata_key(self) -> ArtifactKey { + self.artifact_key(ArtifactKind::Metadata) + } + fn artifact_key(self, kind: ArtifactKind) -> ArtifactKey { + ArtifactKey { + operation: self.metadata.segment.identity.operation, + binding_digest: self.pack.digest, + kind, + } + } + pub fn validate(self, repository: [u8; 16], format: ObjectFormat) -> Result<(), IndexError> { + let segment = self.metadata.segment; + let identity = segment.identity; + let end = identity + .first_ordinal + .checked_add(identity.object_count) + .ok_or(IndexError::Integrity)?; + self.native().validate(repository, format)?; + if identity.repository != repository + || identity.format != format + || identity.git_checksum.format() != format + || identity.git_checksum.is_zero() + || identity.object_count == 0 + || end > self.pack_object_count + || identity.pack_digest != self.pack.digest + || segment.size != self.metadata.artifact.size + || segment.digest != self.metadata.artifact.digest + || segment.size < 4096 + || !segment.size.is_multiple_of(4096) + || segment.edge_count > i64::MAX as u64 + || segment.first_oid.format() != format + || segment.last_oid.format() != format + || segment.first_oid.is_zero() + || segment.first_oid > segment.last_oid + || (identity.object_count == 1) != (segment.first_oid == segment.last_oid) + || self.metadata.artifact.size > canopy_object_storage::external::MAX_ARTIFACT_BYTES + { + return Err(IndexError::Integrity); + } + Ok(()) + } +} + +#[cfg(test)] +pub(in crate::packs) mod tests; diff --git a/crates/canopy-server/src/packs/sources/native.rs b/crates/canopy-server/src/packs/sources/native.rs new file mode 100644 index 0000000..9698647 --- /dev/null +++ b/crates/canopy-server/src/packs/sources/native.rs @@ -0,0 +1,169 @@ +use super::super::metadata::{MetadataError, file_digest}; +use super::*; +use crate::git_format::{ObjectHasher, pack_index::PackIndex}; +use std::{fs::File, io::Read, path::Path}; + +/// The shared artifact context of every shard in one physical pack. This is +/// available before canonical inspection creates metadata. Descriptors alone +/// prove neither native validity, decoded bodies nor graph closure. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct NativePackDescriptor { + pub repository: [u8; 16], + pub operation: [u8; 16], + pub format: ObjectFormat, + pub git_checksum: ObjectId, + pub object_count: u32, + pub pack: ArtifactDescriptor, + pub index: ArtifactDescriptor, +} +impl NativePackDescriptor { + pub fn validate(self, repository: [u8; 16], format: ObjectFormat) -> Result<(), IndexError> { + let width = format.bytes() as u64; + let index_min = 8 + 256 * 4 + u64::from(self.object_count) * (width + 8) + 2 * width; + let index_max = index_min + u64::from(self.object_count) * 8; + if self.repository != repository + || self.format != format + || self.git_checksum.format() != format + || self.git_checksum.is_zero() + || self.object_count == 0 + || self.pack.size < 12 + width + || self.index.size < index_min + || self.index.size > index_max + || !(self.index.size - index_min).is_multiple_of(8) + || [self.pack, self.index] + .iter() + .any(|artifact| artifact.size > canopy_object_storage::external::MAX_ARTIFACT_BYTES) + { + return Err(IndexError::Integrity); + } + Ok(()) + } + pub fn key(self, kind: ArtifactKind) -> Result { + if !matches!(kind, ArtifactKind::Pack | ArtifactKind::Index) { + return Err(IndexError::Integrity); + } + Ok(ArtifactKey { + operation: self.operation, + binding_digest: self.pack.digest, + kind, + }) + } + /// Bounded blocking artifact/header/checksum verification. Callers retain + /// immutable admitted files. Native decoding and delta independence remain + /// unproved until the isolated physical verifier completes. + pub fn verify_files( + self, + pack_path: &Path, + index_path: &Path, + ) -> Result { + self.validate(self.repository, self.format)?; + if file_digest(index_path, self.index.size)? != self.index.digest { + return Err(IndexError::Integrity); + } + let index = PackIndex::open(index_path, self.format).map_err(MetadataError::from)?; + if index.len() != self.object_count || index.pack_checksum() != self.git_checksum { + return Err(IndexError::Integrity); + } + if pack_digest( + pack_path, + self.pack.size, + self.git_checksum, + self.object_count, + )? != self.pack.digest + { + return Err(IndexError::Integrity); + } + Ok(VerifiedPackBinding { + binding: self, + index, + }) + } + /// Local precursor only: manifest digests are filled by authenticated + /// upload before this descriptor escapes the capture service. + pub(crate) fn inspect_files( + repository: [u8; 16], + operation: [u8; 16], + format: ObjectFormat, + pack_path: &Path, + index_path: &Path, + ) -> Result { + let pack_size = std::fs::metadata(pack_path) + .map_err(MetadataError::from)? + .len(); + let index_size = std::fs::metadata(index_path) + .map_err(MetadataError::from)? + .len(); + let index = PackIndex::open(index_path, format).map_err(MetadataError::from)?; + let mut descriptor = Self { + repository, + operation, + format, + git_checksum: index.pack_checksum(), + object_count: index.len(), + pack: ArtifactDescriptor { + size: pack_size, + digest: [0; 32], + manifest_digest: [0; 32], + }, + index: ArtifactDescriptor { + size: index_size, + digest: [0; 32], + manifest_digest: [0; 32], + }, + }; + descriptor.validate(repository, format)?; + descriptor.index.digest = file_digest(index_path, index_size)?; + descriptor.pack.digest = pack_digest( + pack_path, + pack_size, + descriptor.git_checksum, + descriptor.object_count, + )?; + Ok(descriptor) + } +} + +fn pack_digest( + pack_path: &Path, + size: u64, + git_checksum: ObjectId, + object_count: u32, +) -> Result<[u8; 32], IndexError> { + let format = git_checksum.format(); + let mut file = File::open(pack_path).map_err(MetadataError::from)?; + if file.metadata().map_err(MetadataError::from)?.len() != size { + return Err(IndexError::Integrity); + } + let mut header = [0; 12]; + file.read_exact(&mut header).map_err(MetadataError::from)?; + let version = u32::from_be_bytes(header[4..8].try_into().map_err(|_| IndexError::Integrity)?); + let count = u32::from_be_bytes(header[8..].try_into().map_err(|_| IndexError::Integrity)?); + if &header[..4] != b"PACK" || !matches!(version, 2 | 3) || count != object_count { + return Err(IndexError::Integrity); + } + let mut whole = blake3::Hasher::new(); + let mut native = ObjectHasher::raw(format); + whole.update(&header); + native.update(&header); + let mut remaining = size - 12 - format.bytes() as u64; + let mut buffer = [0; 64 << 10]; + while remaining > 0 { + let length = remaining.min(buffer.len() as u64) as usize; + file.read_exact(&mut buffer[..length]) + .map_err(MetadataError::from)?; + whole.update(&buffer[..length]); + native.update(&buffer[..length]); + remaining -= length as u64; + } + let mut trailer = [0; 32]; + let trailer = &mut trailer[..format.bytes()]; + file.read_exact(trailer).map_err(MetadataError::from)?; + whole.update(trailer); + if trailer != git_checksum.as_ref() + || native.finalize() != git_checksum + || file.read(&mut buffer[..1]).map_err(MetadataError::from)? != 0 + { + return Err(IndexError::Integrity); + } + Ok(*whole.finalize().as_bytes()) +} diff --git a/crates/canopy-server/src/packs/sources/resolve.rs b/crates/canopy-server/src/packs/sources/resolve.rs new file mode 100644 index 0000000..92d2504 --- /dev/null +++ b/crates/canopy-server/src/packs/sources/resolve.rs @@ -0,0 +1,47 @@ +use super::super::{ + directory::DirectoryEntry, + metadata::{MetadataError, MetadataSegment}, +}; +use super::*; +use std::{future::Future, sync::Arc}; + +/// Service-owned admitted loading/caching. Implementations must retain file +/// admission and verify whole authenticated metadata bytes before opening SQL. +pub trait SourceLoader: Sync { + fn load( + &self, + segment: StoredSegment, + ) -> impl Future, MetadataError>> + Send; +} + +/// Pins the exact preferred metadata file, not a publication/closure proof. +pub struct ResolvedSource { + pub record: SourceRecord, + pub metadata: Arc, +} +impl SourceIndex { + /// Resolve an already selected directory entry through its pinned source + /// root. The caller supplies authorization and retains the catalog reader + /// lease throughout this call and subsequent artifact reads. + pub async fn resolve( + &self, + root: Option, + entry: DirectoryEntry, + loader: &impl SourceLoader, + ) -> Result { + if entry.header.object.oid.format() != self.format() { + return Err(IndexError::Integrity); + } + let record = self + .find(root, entry.source) + .await? + .ok_or(IndexError::Integrity)?; + let metadata = loader.load(record.metadata).await?; + tokio::task::spawn_blocking(move || { + record.verify_directory_entry(&metadata, entry)?; + Ok(ResolvedSource { record, metadata }) + }) + .await + .map_err(MetadataError::from)? + } +} diff --git a/crates/canopy-server/src/packs/sources/tests.rs b/crates/canopy-server/src/packs/sources/tests.rs new file mode 100644 index 0000000..c7c1878 --- /dev/null +++ b/crates/canopy-server/src/packs/sources/tests.rs @@ -0,0 +1,436 @@ +use super::super::metadata::{ + MetadataSegment, + tests::{builder, fill, fixture}, +}; +use super::*; +use canopy_object_storage::artifact::ArtifactStore; +use canopy_object_storage::external::MAX_ARTIFACT_BYTES; +use cellule_ltx::DiskBudget; +use object_store::{ObjectStore, ObjectStoreExt, memory::InMemory}; +use std::{ + path::{Path, PathBuf}, + sync::Arc, +}; + +type Result = std::result::Result>; + +fn oid(n: u64, format: ObjectFormat) -> ObjectId { + let mut bytes = vec![0; format.bytes()]; + bytes[..8].copy_from_slice(&n.to_be_bytes()); + bytes.try_into().unwrap() +} +// Descriptor-tree fixtures do not claim native verification or publication. +pub(in crate::packs) fn source(n: u64, format: ObjectFormat) -> SourceRecord { + let mut operation = [0; 16]; + operation[..8].copy_from_slice(&n.to_be_bytes()); + let descriptor = |size, byte| ArtifactDescriptor { + size, + digest: [byte; 32], + manifest_digest: [byte + 1; 32], + }; + let metadata = descriptor(16 << 10, 3); + let pack = descriptor(128, 5); + SourceRecord { + metadata: StoredSegment { + segment: SegmentDescriptor { + identity: SegmentIdentity { + repository: [1; 16], + operation, + format, + pack_digest: pack.digest, + git_checksum: oid(1, format), + first_ordinal: 0, + object_count: 2, + }, + edge_count: 0, + inventory_digest: [9; 32], + first_oid: oid(1, format), + last_oid: oid(2, format), + size: metadata.size, + digest: metadata.digest, + }, + artifact: metadata, + }, + pack, + index: descriptor( + 8 + 256 * 4 + 2 * (format.bytes() as u64 + 8) + 2 * format.bytes() as u64, + 7, + ), + pack_object_count: 2, + } +} +fn store() -> (Arc, Arc) { + let provider: Arc = Arc::new(InMemory::new()); + ( + Arc::new(ArtifactStore::new(Arc::clone(&provider), [1; 16])), + provider, + ) +} + +#[tokio::test] +async fn source_catalog_split_cold_lookup_seek_and_retained_roots_for_both_formats() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let (store, _) = store(); + let index = SourceIndex::new(store, format); + let mut root = None; + for n in (0..SOURCE_FANOUT + 17).rev() { + root = Some( + index + .insert(root, [5; 16], source(n as u64, format)) + .await?, + ); + } + let old = root.ok_or("root")?; + assert_eq!(old.height, 1); + assert_eq!(old.record_count, (SOURCE_FANOUT + 17) as u64); + assert_eq!(old.object_count, old.record_count * 2); + assert_eq!(index.insert(root, [6; 16], source(0, format)).await?, old); + index.clear_cache()?; + let stats = index.stats(); + assert_eq!( + index.find(root, source(100, format).key()).await?, + Some(source(100, format)) + ); + assert_eq!( + index.stats().loaded_nodes - stats.loaded_nodes, + u64::from(old.height) + 1 + ); + let mut cursor = index.cursor(root, Some(source(99, format).key()))?; + for n in 100..SOURCE_FANOUT + 17 { + assert_eq!(cursor.next().await?, Some(source(n as u64, format))); + } + assert!(cursor.next().await?.is_none()); + let mut changed = source(100, format); + changed.index.manifest_digest[0] ^= 1; + assert!(matches!( + index.insert(root, [6; 16], changed).await, + Err(IndexError::RangeOverlap) + )); + assert!(matches!( + index.remove(root, [6; 16], changed).await, + Err(IndexError::Stale) + )); + for n in 0..SOURCE_FANOUT + 17 { + root = index + .remove(root, [7; 16], source(n as u64, format)) + .await?; + } + assert!(root.is_none()); + assert_eq!( + index.find(Some(old), source(100, format).key()).await?, + Some(source(100, format)) + ); + } + Ok(()) +} + +#[tokio::test] +async fn catalog_binding_context_summary_and_authenticated_corruption_fail_closed() -> Result { + let format = ObjectFormat::Sha256; + let (store, provider) = store(); + let index = SourceIndex::new(Arc::clone(&store), format); + let original = source(1, format); + let root = index.insert(None, [5; 16], original).await?; + let mut forged = root; + forged.record_count += 1; + assert!(index.find(Some(forged), original.key()).await.is_err()); + let mut forged = root; + forged.first_key.operation[0] ^= 1; + assert!(index.find(Some(forged), original.key()).await.is_err()); + let wrong_format = SourceIndex::new(Arc::clone(&store), ObjectFormat::Sha1); + assert!(wrong_format.find(Some(root), original.key()).await.is_err()); + let wrong_repo = SourceIndex::new( + Arc::new(ArtifactStore::new(Arc::clone(&provider), [2; 16])), + format, + ); + assert!(wrong_repo.find(Some(root), original.key()).await.is_err()); + let key = ArtifactKey { + operation: root.operation, + binding_digest: root.artifact.digest, + kind: ArtifactKind::CatalogNode, + }; + let path = store.path(key, root.artifact.digest)?; + provider + .put( + &canopy_object_storage::external::part(&path, 0), + bytes::Bytes::from(vec![0; root.artifact.size as usize]).into(), + ) + .await?; + index.clear_cache()?; + assert!(index.find(Some(root), original.key()).await.is_err()); + let mut cursor = index.cursor(Some(root), None)?; + assert!(cursor.next().await.is_err()); + assert!(matches!(cursor.next().await, Err(IndexError::Integrity))); + Ok(()) +} + +#[test] +fn source_descriptor_rejects_mismatched_bindings_bounds_and_ordinal_overflow() { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let original = source(1, format); + assert!(original.validate([1; 16], format).is_ok()); + let mut invalid = Vec::new(); + let mut bad = original; + bad.metadata.segment.identity.pack_digest[0] ^= 1; + invalid.push(bad); + let mut bad = original; + bad.metadata.artifact.digest[0] ^= 1; + invalid.push(bad); + let mut bad = original; + bad.metadata.artifact.size += 1; + invalid.push(bad); + let mut bad = original; + bad.metadata.segment.identity.first_ordinal = u32::MAX; + invalid.push(bad); + let mut bad = original; + bad.pack_object_count = 1; + invalid.push(bad); + let mut bad = original; + bad.metadata.segment.identity.object_count = 0; + invalid.push(bad); + let mut bad = original; + bad.metadata.segment.last_oid = bad.metadata.segment.first_oid; + invalid.push(bad); + let mut bad = original; + bad.index.size -= 1; + invalid.push(bad); + let mut bad = original; + bad.index.size += 1; + invalid.push(bad); + let mut bad = original; + bad.pack.size = 1; + invalid.push(bad); + let mut bad = original; + bad.pack.size = MAX_ARTIFACT_BYTES + 1; + invalid.push(bad); + let mut bad = original; + bad.metadata.segment.edge_count = u64::MAX; + invalid.push(bad); + for bad in invalid { + assert!(bad.validate([1; 16], format).is_err(), "accepted {bad:?}"); + } + assert!(original.validate([2; 16], format).is_err()); + } +} + +fn native_paths(root: &Path) -> Result<(PathBuf, PathBuf)> { + let index = std::fs::read_dir(root.join("objects/pack"))? + .filter_map(|e| e.ok().map(|e| e.path())) + .find(|p| p.extension().is_some_and(|ext| ext == "idx")) + .ok_or("index")?; + Ok((index.with_extension("pack"), index)) +} +async fn upload( + store: &ArtifactStore, + key: ArtifactKey, + path: &Path, +) -> Result { + let bytes = std::fs::read(path)?; + Ok(store + .put( + key, + bytes.len() as u64, + *blake3::hash(&bytes).as_bytes(), + &mut bytes.as_slice(), + ) + .await?) +} + +struct DownloadLoader<'a> { + store: &'a ArtifactStore, + root: &'a Path, + budget: DiskBudget, +} +impl SourceLoader for DownloadLoader<'_> { + async fn load( + &self, + stored: StoredSegment, + ) -> std::result::Result, super::super::metadata::MetadataError> { + MetadataSegment::download( + self.root, + self.budget.clone(), + self.store, + stored, + super::super::metadata::tests::limits(), + ) + .await + } +} + +#[tokio::test] +async fn native_artifacts_and_exact_shard_partition_bindings_for_both_formats() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = fixture(format, 16).await?; + let (store, _) = store(); + let (pack_path, index_path) = native_paths(fixture.root.path())?; + let key = |kind| ArtifactKey { + operation: fixture.identity.operation, + binding_digest: fixture.identity.pack_digest, + kind, + }; + let pack = upload(&store, key(ArtifactKind::Pack), &pack_path).await?; + let index_artifact = upload(&store, key(ArtifactKind::Index), &index_path).await?; + let mut sources = Vec::new(); + let mut local_segments: Vec> = Vec::new(); + let halfway = fixture.index.len() / 2; + for (first, count) in [(0, halfway), (halfway, fixture.index.len() - halfway)] { + let identity = SegmentIdentity { + first_ordinal: first, + object_count: count, + ..fixture.identity + }; + let mut writer = builder(&fixture, DiskBudget::new(128 << 20), identity)?; + let objects = fixture + .index + .ids_from(first)? + .take(count as usize) + .map(|oid| Ok(fixture.objects.get(&oid?).ok_or("object")?.clone())) + .collect::>>()?; + fill(&mut writer, &objects)?; + let segment = Arc::new(writer.seal(&fixture.index)?); + let metadata = Arc::clone(&segment).upload(&store).await?; + sources.push(SourceRecord { + metadata, + pack, + index: index_artifact, + pack_object_count: fixture.index.len(), + }); + local_segments.push(segment); + } + let checked = sources[0].verify_native_files(&pack_path, &index_path)?; + checked.verify_source(sources[1])?; + assert_eq!(checked.index().len(), fixture.index.len()); + for (source, segment) in sources.iter().zip(&local_segments) { + let header = segment + .header(source.metadata.segment.first_oid)? + .ok_or("header")?; + let entry = super::super::directory::DirectoryEntry { + header, + source: source.key(), + location_version: 1, + }; + source.verify_directory_entry(segment, entry)?; + let mut bad = entry; + bad.source.operation[0] ^= 1; + assert!(source.verify_directory_entry(segment, bad).is_err()); + let mut bad = entry; + bad.header.object.digest[0] ^= 1; + assert!(matches!( + source.verify_directory_entry(segment, bad), + Err(IndexError::Metadata( + super::super::metadata::MetadataError::IdentityConflict + )) + )); + let mut bad = entry; + bad.location_version = 0; + assert!(source.verify_directory_entry(segment, bad).is_err()); + assert!( + source + .verify_directory_entry( + &local_segments[if source == &sources[0] { 1 } else { 0 }], + entry + ) + .is_err() + ); + } + let mut coverage = PackCoverage::new(sources[0])?; + for source in &sources { + coverage.add(*source)?; + } + coverage.finish()?; + let mut missing = PackCoverage::new(sources[0])?; + missing.add(sources[0])?; + assert!(missing.finish().is_err()); + let mut reversed = PackCoverage::new(sources[0])?; + assert!(reversed.add(sources[1]).is_err()); + assert!(reversed.add(sources[0]).is_err()); + assert!(reversed.finish().is_err()); + let mut duplicate = PackCoverage::new(sources[0])?; + duplicate.add(sources[0])?; + assert!(duplicate.add(sources[0]).is_err()); + assert!(duplicate.finish().is_err()); + let mut altered = sources[1]; + altered.index.manifest_digest[0] ^= 1; + assert!(checked.verify_source(altered).is_err()); + let mut coverage = PackCoverage::new(sources[0])?; + coverage.add(sources[0])?; + assert!(coverage.add(altered).is_err()); + assert!(coverage.finish().is_err()); + let mut endpoint = sources[0]; + endpoint.metadata.segment.first_oid = oid(1, format); + assert!(checked.verify_source(endpoint).is_err()); + let catalog = SourceIndex::new(Arc::clone(&store), format); + let root = catalog.insert(None, [5; 16], sources[0]).await?; + let root = catalog.insert(Some(root), [5; 16], sources[1]).await?; + catalog.clear_cache()?; + for source in &sources { + assert_eq!(catalog.find(Some(root), source.key()).await?, Some(*source)); + } + let budget = DiskBudget::new(128 << 20); + let loader = DownloadLoader { + store: &store, + root: fixture.root.path(), + budget: budget.clone(), + }; + let mut directory = super::super::directory::DirectoryBuilder::new( + fixture.root.path(), + DiskBudget::new(128 << 20), + [1; 16], + [8; 16], + format, + super::super::metadata::tests::limits(), + )?; + for segment in &local_segments { + directory.add_segment(segment)?; + } + let directory = directory.seal()?; + for source in &sources { + let entry = directory + .find(source.metadata.segment.first_oid)? + .ok_or("entry")?; + let resolved = catalog.resolve(Some(root), entry, &loader).await?; + assert_eq!(resolved.record, *source); + assert_eq!( + resolved.metadata.header(entry.header.object.oid)?, + Some(entry.header) + ); + assert_eq!(budget.used(), source.metadata.artifact.size); + drop(resolved); + assert_eq!(budget.used(), 0); + let mut conflict = entry; + conflict.header.edge_digest[0] ^= 1; + assert!( + catalog + .resolve(Some(root), conflict, &loader) + .await + .is_err() + ); + assert_eq!(budget.used(), 0); + assert!(catalog.resolve(None, entry, &loader).await.is_err()); + } + let mut bytes = std::fs::read(&pack_path)?; + bytes[12] ^= 1; + let bad_pack = fixture.root.path().join("corrupted.pack"); + std::fs::write(&bad_pack, &bytes)?; + assert!( + sources[0] + .verify_native_files(&bad_pack, &index_path) + .is_err() + ); + let mut bytes = std::fs::read(&index_path)?; + bytes[12] ^= 1; + let bad_index = fixture.root.path().join("corrupted.idx"); + std::fs::write(&bad_index, &bytes)?; + assert!( + sources[0] + .verify_native_files(&pack_path, &bad_index) + .is_err() + ); + // Rewrite only the declared BLAKE3: the native checksum must still fail. + let mut forged = sources[0]; + forged.pack.digest = *blake3::hash(&std::fs::read(&bad_pack)?).as_bytes(); + forged.metadata.segment.identity.pack_digest = forged.pack.digest; + assert!(forged.verify_native_files(&bad_pack, &index_path).is_err()); + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/sources/verification.rs b/crates/canopy-server/src/packs/sources/verification.rs new file mode 100644 index 0000000..433713c --- /dev/null +++ b/crates/canopy-server/src/packs/sources/verification.rs @@ -0,0 +1,169 @@ +use super::super::metadata::MetadataError; +use super::*; +use crate::git_format::pack_index::PackIndex; +use std::path::Path; + +/// Exact native artifact binding checked once for all metadata shards of a +/// physical pack. The service retains the original admitted immutable files. +/// This handle carries no canonical-body, closure or authorization certificate. +pub struct VerifiedPackBinding { + pub(super) binding: NativePackDescriptor, + pub(super) index: PackIndex, +} +impl VerifiedPackBinding { + pub fn index(&self) -> &PackIndex { + &self.index + } + pub fn verify_source(&self, source: SourceRecord) -> Result<(), IndexError> { + let expected = self.binding; + source.validate(expected.repository, expected.format)?; + let identity = source.metadata.segment.identity; + if source.native() != expected { + return Err(IndexError::Integrity); + } + let first = self + .index + .ids_from(identity.first_ordinal) + .map_err(MetadataError::from)? + .next() + .transpose() + .map_err(MetadataError::from)?; + let last_ordinal = identity.first_ordinal + identity.object_count - 1; + let last = self + .index + .ids_from(last_ordinal) + .map_err(MetadataError::from)? + .next() + .transpose() + .map_err(MetadataError::from)?; + if first != Some(source.metadata.segment.first_oid) + || last != Some(source.metadata.segment.last_oid) + { + return Err(IndexError::Integrity); + } + Ok(()) + } +} +impl SourceRecord { + /// Validates a directory's preferred-source pointer against the exact + /// verified metadata shard. This is independent of authorization/closure. + pub fn verify_directory_entry( + self, + segment: &super::super::metadata::MetadataSegment, + entry: super::super::directory::DirectoryEntry, + ) -> Result<(), IndexError> { + self.verify_directory_entries(segment, &[entry]) + } + pub fn verify_directory_entries( + self, + segment: &super::super::metadata::MetadataSegment, + entries: &[super::super::directory::DirectoryEntry], + ) -> Result<(), IndexError> { + if entries.len() > super::super::metadata::PAGE_OBJECTS { + return Err(IndexError::Limit); + } + let descriptor = self.metadata.segment; + self.validate(descriptor.identity.repository, descriptor.identity.format)?; + if segment.descriptor() != descriptor { + return Err(IndexError::Integrity); + } + for entry in entries { + if entry.source != self.key() + || entry.location_version == 0 + || entry.location_version > i64::MAX as u64 + || entry.header.object.oid.format() != descriptor.identity.format + || entry.header.object.oid < descriptor.first_oid + || entry.header.object.oid > descriptor.last_oid + { + return Err(IndexError::Integrity); + } + } + let ids: Vec<_> = entries + .iter() + .map(|entry| entry.header.object.oid) + .collect(); + let headers = segment.headers(&ids)?; + if headers.len() != entries.len() { + return Err(IndexError::Integrity); + } + for (entry, actual) in entries.iter().zip(headers) { + if actual != Some(entry.header) { + return Err(MetadataError::IdentityConflict.into()); + } + } + Ok(()) + } + /// Bounded blocking validation of exact local artifact bytes, native pack + /// header/trailer, checked index and shard endpoints. The service must pin + /// admitted immutable files and run this on its blocking verification pool. + /// This does not decode objects or certify canonical body/graph closure. + pub fn verify_native_files( + self, + pack_path: &Path, + index_path: &Path, + ) -> Result { + let identity = self.metadata.segment.identity; + self.validate(identity.repository, identity.format)?; + let verified = self.native().verify_files(pack_path, index_path)?; + verified.verify_source(self)?; + Ok(verified) + } +} + +/// Constant-space exact ordinal partition check. Feed shards in native ordinal +/// order after independently verifying each immutable metadata file. Equal row +/// counts, pack membership and this partition check are not closure proofs. +pub struct PackCoverage { + first: SourceRecord, + next_ordinal: u32, + last_oid: Option, + poisoned: bool, +} +impl PackCoverage { + pub fn new(first: SourceRecord) -> Result { + let identity = first.metadata.segment.identity; + first.validate(identity.repository, identity.format)?; + Ok(Self { + first, + next_ordinal: 0, + last_oid: None, + poisoned: false, + }) + } + pub fn add(&mut self, source: SourceRecord) -> Result<(), IndexError> { + if self.poisoned { + return Err(IndexError::Integrity); + } + self.poisoned = true; + let expected = self.first.metadata.segment.identity; + source.validate(expected.repository, expected.format)?; + let segment = source.metadata.segment; + let identity = segment.identity; + if source.pack != self.first.pack + || source.index != self.first.index + || source.pack_object_count != self.first.pack_object_count + || identity.operation != expected.operation + || identity.git_checksum != expected.git_checksum + || identity.first_ordinal != self.next_ordinal + || self.last_oid.is_some_and(|last| last >= segment.first_oid) + { + return Err(IndexError::Integrity); + } + self.next_ordinal = identity + .first_ordinal + .checked_add(identity.object_count) + .ok_or(IndexError::Integrity)?; + self.last_oid = Some(segment.last_oid); + self.poisoned = false; + Ok(()) + } + pub fn finish(self) -> Result<(), IndexError> { + if self.poisoned + || self.last_oid.is_none() + || self.next_ordinal != self.first.pack_object_count + { + return Err(IndexError::Integrity); + } + Ok(()) + } +} diff --git a/crates/canopy-server/src/packs/verification/mod.rs b/crates/canopy-server/src/packs/verification/mod.rs new file mode 100644 index 0000000..781a554 --- /dev/null +++ b/crates/canopy-server/src/packs/verification/mod.rs @@ -0,0 +1,81 @@ +//! Canonical decoded-object inspection with bounded typed edge emission. +//! This is not a physical pack, graph closure or publication certificate. + +use super::metadata::CanonicalObject; +pub use crate::git_objects::{EdgeSink, ObjectReadError}; +use crate::{ObjectFormat, ObjectId, git_objects::GitObjects}; +use std::path::Path; + +mod spool; +pub use spool::VerifiedObject; +pub(super) mod physical; +pub use physical::{ + PhysicalError, PhysicalLimits, PhysicalPackWitness, PhysicalPartition, PhysicalVerifier, +}; + +/// The service retains its admitted private native workspace and process +/// resources. A physical-pack verifier must isolate the exact verified pack, +/// without alternates, before using these decoded-object witnesses. +pub struct CanonicalVerifier { + native: GitObjects, + format: ObjectFormat, +} +impl CanonicalVerifier { + pub fn new( + git_dir: &Path, + format: ObjectFormat, + native: &crate::native_resources::NativeScope, + ) -> Result { + Ok(Self { + native: GitObjects::batch(git_dir, native)?, + format, + }) + } + /// Sink output remains private until this completes. Discard the whole + /// affected preparation after an error/cancellation. Once native inspection + /// starts, either event poisons the actor, including a canceled sink write. + pub async fn inspect( + &mut self, + oid: ObjectId, + sink: &mut impl EdgeSink, + ) -> Result { + if oid.format() != self.format || oid.is_zero() { + return Err(ObjectReadError::Malformed); + } + self.native.inspect_graph(oid, sink).await + } + pub async fn finish(self) -> Result<(), ObjectReadError> { + self.native.finish().await + } + + /// Retain dependency occurrences on admitted disk until the complete native + /// frame and canonical hashes have been verified. Blobs create no spool. + /// This witness still requires physical-pack isolation and closure checks. + pub async fn inspect_to_disk( + &mut self, + oid: ObjectId, + root: &Path, + budget: cellule_ltx::DiskBudget, + max_edge_bytes: u64, + ) -> Result { + let mut sink = spool::DiskSink::new(oid, root, budget, max_edge_bytes); + let object = self.inspect(oid, &mut sink).await?; + sink.complete(object) + } + + /// Reuse one dependency file across an admitted physical-verification page. + /// Each witness retains its exact immutable range and complete digest. + async fn inspect_to_spool( + &mut self, + oid: ObjectId, + spool: &spool::EdgeSpool, + max_edge_bytes: u64, + ) -> Result { + let mut sink = spool.sink(oid, max_edge_bytes)?; + let object = self.inspect(oid, &mut sink).await?; + sink.complete(object) + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/canopy-server/src/packs/verification/physical.rs b/crates/canopy-server/src/packs/verification/physical.rs new file mode 100644 index 0000000..c05ea7f --- /dev/null +++ b/crates/canopy-server/src/packs/verification/physical.rs @@ -0,0 +1,379 @@ +//! Isolated verification of authenticated native bytes and exact decoded shard +//! coverage. Global typed closure and fenced publication are separate stages. +use super::super::{ + directory::index::IndexError, + metadata::{ + MetadataBuilder, MetadataError, MetadataLimits, MetadataSegment, PAGE_OBJECTS, + SegmentDescriptor, SegmentIdentity, + }, + sources::{NativePackDescriptor, VerifiedPackBinding}, +}; +use super::*; +use crate::{ + git_cache::{CacheError, GitCache}, + git_http::{GitHttpError, GitProcess, WORKER_DEADLINE, read_bounded}, +}; +use canopy_object_storage::{artifact::ArtifactStore, external::MAX_ARTIFACT_BYTES}; +use cellule_ltx::DiskBudget; +use std::{path::PathBuf, process::Stdio, sync::Arc, time::Duration}; + +mod partition; +pub use partition::PhysicalPartition; + +#[derive(Clone, Copy, Debug)] +pub struct PhysicalLimits { + pub max_pack_bytes: u64, + pub max_index_bytes: u64, + pub max_edge_bytes: u64, + pub metadata: MetadataLimits, + /// Deadline for native index validation and actor shutdown. The service + /// separately owns the whole-operation deadline and native process quotas. + pub native_timeout: Duration, +} +impl Default for PhysicalLimits { + fn default() -> Self { + Self { + max_pack_bytes: MAX_ARTIFACT_BYTES, + max_index_bytes: MAX_ARTIFACT_BYTES, + max_edge_bytes: 256 << 20, + metadata: MetadataLimits::default(), + native_timeout: WORKER_DEADLINE, + } + } +} +#[derive(Debug, thiserror::Error)] +pub enum PhysicalError { + #[error("isolated native workspace failed")] + Cache(#[from] CacheError), + #[error("physical artifact binding failed")] + Binding(#[from] IndexError), + #[error("physical metadata assembly failed")] + Metadata(#[from] MetadataError), + #[error("canonical decoded inspection failed")] + Object(#[from] ObjectReadError), + #[error("native physical validation failed")] + Native(#[from] GitHttpError), + #[error("physical verification I/O failed")] + Io(#[from] std::io::Error), + #[error("physical verification task failed")] + Task(#[from] tokio::task::JoinError), + #[error("physical verification is incomplete, canceled or inconsistent")] + Integrity, + #[error("physical verification exceeds its admitted limits")] + Limit, +} + +/// Only successful complete isolated verification constructs this value. It +/// binds every decoded native ordinal to an exact sealed metadata partition. +/// It is an in-process witness, not a persisted publication certificate. +pub struct PhysicalPackWitness { + store: ArtifactStore, + native: NativePackDescriptor, + shard_count: u32, + metadata_digest: [u8; 32], +} +impl PhysicalPackWitness { + pub fn verify_store(&self, store: &ArtifactStore) -> Result<(), PhysicalError> { + if self.store.same_binding(store) { + Ok(()) + } else { + Err(PhysicalError::Integrity) + } + } + pub fn native(&self) -> NativePackDescriptor { + self.native + } + pub fn shard_count(&self) -> u32 { + self.shard_count + } + pub fn metadata_digest(&self) -> [u8; 32] { + self.metadata_digest + } + /// A bounded checker for metadata loading/copying one shard at a time. + pub fn partition(&self) -> PhysicalPartition { + PhysicalPartition::new(self.native, self.shard_count, self.metadata_digest) + } + /// Compare the exact ordered metadata descriptors when assembling the + /// catalog. Merely matching a pack's object count is insufficient. + pub fn verify_segments( + &self, + segments: impl IntoIterator, + ) -> Result<(), PhysicalError> { + let mut partition = self.partition(); + for segment in segments { + partition.add(segment)?; + } + partition.finish() + } +} + +/// Owns a fresh private workspace with exactly one pack/index pair, no loose +/// objects, alternates, refs, replacement objects or host configuration. Native +/// cache cleanup retains admission until all inherited worker fences clear. +/// The service must also retain its native-process admission for this lifetime. +pub struct PhysicalVerifier { + store: ArtifactStore, + // Drop the native actor/index before releasing the fenced workspace. + native: Option, + binding: Arc, + cache: Arc, + descriptor: NativePackDescriptor, + root: PathBuf, + budget: DiskBudget, + limits: PhysicalLimits, + next_ordinal: u32, + shards: u32, + chain: [u8; 32], + failed: bool, +} +impl PhysicalVerifier { + pub async fn download( + root: &Path, + budget: DiskBudget, + store: &ArtifactStore, + descriptor: NativePackDescriptor, + limits: PhysicalLimits, + native: crate::native_resources::NativeScope, + ) -> Result { + descriptor.validate(store.repository(), descriptor.format)?; + if limits.max_pack_bytes > MAX_ARTIFACT_BYTES + || limits.max_index_bytes > MAX_ARTIFACT_BYTES + || descriptor.pack.size > limits.max_pack_bytes + || descriptor.index.size > limits.max_index_bytes + || limits.native_timeout.is_zero() + { + return Err(PhysicalError::Limit); + } + let root = root.to_owned(); + let root = tokio::task::spawn_blocking(move || std::fs::canonicalize(root)).await??; + let cache = GitCache::create( + root.clone(), + budget.clone(), + "refs/heads/main", + descriptor.format, + native, + ) + .await?; + cache.download_native(store, descriptor).await?; + let pinned = Arc::clone(&cache); + let input_claim = cache + .native + .try_admit(crate::native_resources::NativeWork::Read)?; + let binding = tokio::task::spawn_blocking(move || { + let _claim = input_claim; + let path = pack_path(&pinned, descriptor); + descriptor.verify_files(&path, &path.with_extension("idx")) + }) + .await??; + validate_native(Arc::clone(&cache), descriptor, limits.native_timeout).await?; + let native = CanonicalVerifier::new(&cache.git_dir(), descriptor.format, &cache.native)?; + Ok(Self { + store: store.clone(), + native: Some(native), + binding: Arc::new(binding), + cache, + descriptor, + root, + budget, + limits, + next_ordinal: 0, + shards: 0, + chain: seed(descriptor), + failed: false, + }) + } + + /// Produce the next exact ordinal interval. Returned files remain private + /// staging until finish returns the whole-pack witness and closure succeeds. + /// Any error or cancellation permanently prevents reuse/successful finish. + pub async fn inspect_next_shard( + &mut self, + object_count: u32, + ) -> Result, PhysicalError> { + if self.failed { + return Err(PhysicalError::Integrity); + } + self.failed = true; + let end = self + .next_ordinal + .checked_add(object_count) + .filter(|end| *end <= self.descriptor.object_count && object_count != 0) + .ok_or(PhysicalError::Integrity)?; + let identity = shard_identity(self.descriptor, self.next_ordinal, object_count); + let root = self.root.clone(); + let budget = self.budget.clone(); + let limits = self.limits.metadata; + let mut builder = tokio::task::spawn_blocking(move || { + MetadataBuilder::new(&root, budget, identity, limits) + }) + .await??; + let mut ordinal = self.next_ordinal; + while ordinal < end { + let count = (end - ordinal).min(PAGE_OBJECTS as u32); + let binding = Arc::clone(&self.binding); + let cache = Arc::clone(&self.cache); + let ids = tokio::task::spawn_blocking(move || { + let _pin = cache; + binding + .index() + .ids_from(ordinal)? + .take(count as usize) + .collect::>>() + }) + .await??; + if ids.len() != count as usize { + return Err(PhysicalError::Integrity); + } + let mut witnesses = Vec::with_capacity(ids.len()); + let edges = spool::EdgeSpool::new(&self.root, self.budget.clone()); + for oid in ids { + let witness = self + .native + .as_mut() + .ok_or(PhysicalError::Integrity)? + .inspect_to_spool(oid, &edges, self.limits.max_edge_bytes) + .await?; + witnesses.push(witness); + } + // Witnesses and detached SQL workers own the file through every + // replay. Drop the producer handle before handing off this page. + drop(edges); + let cache = Arc::clone(&self.cache); + builder = tokio::task::spawn_blocking(move || { + let _pin = cache; + builder.put_verified_batch(witnesses)?; + Ok::<_, MetadataError>(builder) + }) + .await??; + ordinal += count; + } + let binding = Arc::clone(&self.binding); + let cache = Arc::clone(&self.cache); + let segment = tokio::task::spawn_blocking(move || { + let _pin = cache; + builder.seal(binding.index()).map(Arc::new) + }) + .await??; + self.chain = fold_shard(self.chain, self.shards, segment.descriptor()); + self.shards = self.shards.checked_add(1).ok_or(PhysicalError::Integrity)?; + self.next_ordinal = end; + self.failed = false; + Ok(segment) + } + + pub async fn finish(mut self) -> Result { + if self.failed || self.next_ordinal != self.descriptor.object_count || self.shards == 0 { + return Err(PhysicalError::Integrity); + } + let native = self.native.take().ok_or(PhysicalError::Integrity)?; + tokio::time::timeout(self.limits.native_timeout, native.finish()) + .await + .map_err(|_| GitHttpError::Timeout)??; + Ok(PhysicalPackWitness { + store: self.store, + native: self.descriptor, + shard_count: self.shards, + metadata_digest: self.chain, + }) + } +} + +fn pack_path(cache: &GitCache, descriptor: NativePackDescriptor) -> PathBuf { + cache.git_dir().join(format!( + "objects/pack/pack-{}.pack", + hex::encode(descriptor.git_checksum) + )) +} +async fn validate_native( + cache: Arc, + descriptor: NativePackDescriptor, + deadline: Duration, +) -> Result<(), GitHttpError> { + let mut command = crate::native_git::command(&cache.git_dir())?; + command + .args(["index-pack", "--threads=2", "--verify"]) + .arg(pack_path(&cache, descriptor)) + .stdin(Stdio::null()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()); + let native = cache + .native + .try_admit(crate::native_resources::NativeWork::Pack)?; + let mut process = GitProcess::spawn(command, cache, native)?; + let run = async { + let (_, stderr) = tokio::try_join!( + read_bounded( + process + .child + .stdout + .take() + .ok_or(GitHttpError::Interrupted)?, + 128 + ), + read_bounded( + process + .child + .stderr + .take() + .ok_or(GitHttpError::Interrupted)?, + 64 << 10 + ), + )?; + let status = process.wait().await?; + if !status.success() { + return Err(GitHttpError::GitExit { + status, + stderr: String::from_utf8_lossy(&stderr).into_owned(), + }); + } + Ok(()) + }; + tokio::time::timeout(deadline, run) + .await + .map_err(|_| GitHttpError::Timeout)? +} +fn shard_identity( + native: NativePackDescriptor, + first_ordinal: u32, + object_count: u32, +) -> SegmentIdentity { + SegmentIdentity { + repository: native.repository, + operation: native.operation, + format: native.format, + pack_digest: native.pack.digest, + git_checksum: native.git_checksum, + first_ordinal, + object_count, + } +} +fn seed(native: NativePackDescriptor) -> [u8; 32] { + let mut hash = blake3::Hasher::new(); + hash.update(b"canopy.physical-pack-witness.v1\0"); + hash.update(&native.repository); + hash.update(&native.operation); + hash.update(&[native.format.bytes() as u8]); + hash.update(&native.git_checksum); + hash.update(&native.object_count.to_le_bytes()); + for artifact in [native.pack, native.index] { + hash.update(&artifact.size.to_le_bytes()); + hash.update(&artifact.digest); + hash.update(&artifact.manifest_digest); + } + *hash.finalize().as_bytes() +} +fn fold_shard(previous: [u8; 32], ordinal: u32, segment: SegmentDescriptor) -> [u8; 32] { + let mut record = Vec::with_capacity(160); + record.extend_from_slice(&segment.identity.first_ordinal.to_le_bytes()); + record.extend_from_slice(&segment.identity.object_count.to_le_bytes()); + record.extend_from_slice(&segment.first_oid); + record.extend_from_slice(&segment.last_oid); + record.extend_from_slice(&segment.edge_count.to_le_bytes()); + record.extend_from_slice(&segment.inventory_digest); + record.extend_from_slice(&segment.size.to_le_bytes()); + record.extend_from_slice(&segment.digest); + super::super::metadata::fold(previous, u64::from(ordinal), &record) +} + +#[cfg(test)] +pub(in crate::packs) mod tests; diff --git a/crates/canopy-server/src/packs/verification/physical/partition.rs b/crates/canopy-server/src/packs/verification/physical/partition.rs new file mode 100644 index 0000000..d00fd13 --- /dev/null +++ b/crates/canopy-server/src/packs/verification/physical/partition.rs @@ -0,0 +1,69 @@ +use super::*; + +/// Constructed only from a complete physical witness. Invalid or interrupted +/// partitions cannot finish; retained state is independent of shard count. +pub struct PhysicalPartition { + native: NativePackDescriptor, + expected_count: u32, + expected_digest: [u8; 32], + next: u32, + count: u32, + chain: [u8; 32], + last: Option, + failed: bool, +} +impl PhysicalPartition { + pub(super) fn new( + native: NativePackDescriptor, + expected_count: u32, + expected_digest: [u8; 32], + ) -> Self { + Self { + native, + expected_count, + expected_digest, + next: 0, + count: 0, + chain: seed(native), + last: None, + failed: false, + } + } + pub fn add(&mut self, segment: SegmentDescriptor) -> Result<(), PhysicalError> { + if self.failed { + return Err(PhysicalError::Integrity); + } + self.failed = true; + let identity = segment.identity; + if identity != shard_identity(self.native, self.next, identity.object_count) + || identity.object_count == 0 + || segment.first_oid.format() != self.native.format + || segment.last_oid.format() != self.native.format + || segment.first_oid.is_zero() + || segment.first_oid > segment.last_oid + || self.last.is_some_and(|oid| oid >= segment.first_oid) + { + return Err(PhysicalError::Integrity); + } + self.next = self + .next + .checked_add(identity.object_count) + .filter(|end| *end <= self.native.object_count) + .ok_or(PhysicalError::Integrity)?; + self.chain = fold_shard(self.chain, self.count, segment); + self.count = self.count.checked_add(1).ok_or(PhysicalError::Integrity)?; + self.last = Some(segment.last_oid); + self.failed = false; + Ok(()) + } + pub fn finish(self) -> Result<(), PhysicalError> { + if self.failed + || self.next != self.native.object_count + || self.count != self.expected_count + || self.chain != self.expected_digest + { + return Err(PhysicalError::Integrity); + } + Ok(()) + } +} diff --git a/crates/canopy-server/src/packs/verification/physical/tests.rs b/crates/canopy-server/src/packs/verification/physical/tests.rs new file mode 100644 index 0000000..02a0715 --- /dev/null +++ b/crates/canopy-server/src/packs/verification/physical/tests.rs @@ -0,0 +1,434 @@ +use super::*; +use crate::packs::{ + metadata::tests::{Fixture, fixture, limits}, + sources::{PackCoverage, SourceRecord}, +}; + +pub(in crate::packs) mod independence; +use canopy_object_storage::artifact::{ArtifactDescriptor, ArtifactKey, ArtifactKind}; +use cellule_ltx::DiskBudget; +use object_store::{ObjectStore, ObjectStoreExt, memory::InMemory}; +use std::{future::Future, path::Path}; + +type Result = std::result::Result>; +pub(in crate::packs) struct Prepared { + pub(in crate::packs) fixture: Fixture, + provider: Arc, + pub(in crate::packs) store: Arc, + pub(in crate::packs) descriptor: NativePackDescriptor, +} +pub(in crate::packs) fn physical_limits() -> PhysicalLimits { + PhysicalLimits { + metadata: limits(), + native_timeout: Duration::from_secs(30), + ..PhysicalLimits::default() + } +} +async fn upload_bytes( + store: &ArtifactStore, + key: ArtifactKey, + bytes: &[u8], +) -> Result { + Ok(store + .put( + key, + bytes.len() as u64, + *blake3::hash(bytes).as_bytes(), + &mut &bytes[..], + ) + .await?) +} +pub(in crate::packs) async fn prepared(format: ObjectFormat, blobs: usize) -> Result { + prepared_for_context(format, blobs, [1; 16], [2; 16]).await +} +pub(in crate::packs) async fn prepared_for_context( + format: ObjectFormat, + blobs: usize, + repository: [u8; 16], + operation: [u8; 16], +) -> Result { + let provider: Arc = Arc::new(InMemory::new()); + let store = Arc::new(ArtifactStore::new(provider.clone(), repository)); + prepared_for_store(format, blobs, operation, provider, store).await +} +pub(in crate::packs) async fn prepared_for_store( + format: ObjectFormat, + blobs: usize, + operation: [u8; 16], + provider: Arc, + store: Arc, +) -> Result { + let mut fixture = fixture(format, blobs).await?; + fixture.identity.repository = store.repository(); + fixture.identity.operation = operation; + let path = std::fs::read_dir(fixture.root.path().join("objects/pack"))? + .find_map(|entry| { + entry + .ok() + .map(|entry| entry.path()) + .filter(|path| path.extension().is_some_and(|ext| ext == "idx")) + }) + .ok_or("index")?; + let key = |kind| ArtifactKey { + operation: fixture.identity.operation, + binding_digest: fixture.identity.pack_digest, + kind, + }; + let pack = upload_bytes( + &store, + key(ArtifactKind::Pack), + &std::fs::read(path.with_extension("pack"))?, + ) + .await?; + let index = upload_bytes(&store, key(ArtifactKind::Index), &std::fs::read(path)?).await?; + let descriptor = NativePackDescriptor { + repository: fixture.identity.repository, + operation: fixture.identity.operation, + format, + git_checksum: fixture.identity.git_checksum, + object_count: fixture.index.len(), + pack, + index, + }; + Ok(Prepared { + fixture, + provider, + store, + descriptor, + }) +} +async fn drained(root: &Path, budget: &DiskBudget, expected: u64) -> Result { + tokio::time::timeout(Duration::from_secs(5), async { + while budget.used() != expected { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await?; + if expected == 0 { + assert_eq!(std::fs::read_dir(root)?.count(), 0); + } + Ok(()) +} + +#[tokio::test] +async fn isolated_physical_verification_covers_exact_shards_and_binds_uploaded_sources() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let prepared = prepared(format, 1600).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let mut verifier = PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + prepared.descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + let mut segments = Vec::new(); + let halfway = prepared.descriptor.object_count / 2; + for count in [1, halfway, prepared.descriptor.object_count - halfway - 1] { + segments.push(verifier.inspect_next_shard(count).await?); + } + let witness = verifier.finish().await?; + assert_eq!(witness.native(), prepared.descriptor); + assert_eq!(witness.shard_count(), 3); + let descriptors: Vec<_> = segments + .iter() + .map(|segment| segment.descriptor()) + .collect(); + witness.verify_segments(descriptors.iter().copied())?; + let retained = segments + .iter() + .map(|segment| segment.descriptor().size) + .sum(); + drained(root.path(), &budget, retained).await?; + let mut sources = Vec::new(); + for segment in &segments { + let mut after = None; + let mut count = 0; + loop { + let page = segment.headers_after(after)?; + if page.is_empty() { + break; + } + assert!(page.len() <= PAGE_OBJECTS); + for header in page { + assert_eq!( + header.object, + prepared.fixture.objects[&header.object.oid].0 + ); + after = Some(header.object.oid); + count += 1; + } + } + assert_eq!(count, segment.descriptor().identity.object_count); + let metadata = segment.clone().upload(&prepared.store).await?; + let source = SourceRecord { + metadata, + pack: prepared.descriptor.pack, + index: prepared.descriptor.index, + pack_object_count: prepared.descriptor.object_count, + }; + source.validate(prepared.descriptor.repository, format)?; + assert_eq!(source.native(), witness.native()); + sources.push(source); + } + let mut coverage = PackCoverage::new(sources[0])?; + for source in sources { + coverage.add(source)?; + } + coverage.finish()?; + for bad in 0..5 { + let mut forged = descriptors.clone(); + match bad { + 0 => { + forged.swap(0, 1); + } + 1 => { + forged.pop(); + } + 2 => { + forged[1].digest[0] ^= 1; + } + 3 => { + forged[1].inventory_digest[0] ^= 1; + } + _ => { + forged[1].identity.operation[0] ^= 1; + } + } + assert!(matches!( + witness.verify_segments(forged), + Err(PhysicalError::Integrity) + )); + } + drop(segments); + drained(root.path(), &budget, 0).await?; + } + Ok(()) +} + +#[tokio::test] +async fn incomplete_or_failed_physical_inspection_cannot_finish_or_resume() -> Result { + let prepared = prepared(ObjectFormat::Sha256, 700).await?; + for fail in [false, true] { + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let mut limits = physical_limits(); + if fail { + limits.max_edge_bytes = 0; + } + let mut verifier = PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + prepared.descriptor, + limits, + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + if fail { + assert!( + verifier + .inspect_next_shard(prepared.descriptor.object_count) + .await + .is_err() + ); + assert!(matches!( + verifier.inspect_next_shard(1).await, + Err(PhysicalError::Integrity) + )); + } else { + let segment = verifier.inspect_next_shard(1).await?; + drop(segment); + } + assert!(matches!( + verifier.finish().await, + Err(PhysicalError::Integrity) + )); + drained(root.path(), &budget, 0).await?; + } + Ok(()) +} + +#[tokio::test] +async fn native_index_verification_rejects_forged_crc_despite_valid_artifact_and_index_hashes() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let prepared = prepared(format, 16).await?; + let path = std::fs::read_dir(prepared.fixture.root.path().join("objects/pack"))? + .find_map(|entry| { + entry + .ok() + .map(|entry| entry.path()) + .filter(|path| path.extension().is_some_and(|ext| ext == "idx")) + }) + .ok_or("index")?; + let mut bytes = std::fs::read(&path)?; + let crc_at = 1032 + prepared.descriptor.object_count as usize * format.bytes(); + bytes[crc_at] ^= 1; + let payload = bytes.len() - format.bytes(); + let mut hash = crate::git_format::ObjectHasher::raw(format); + hash.update(&bytes[..payload]); + bytes[payload..].copy_from_slice(&hash.finalize()); + let candidate = tempfile::NamedTempFile::new()?; + std::fs::write(candidate.path(), &bytes)?; + // The bounded native index checker accepts a self-consistent index; + // only the isolated native pack/index verification catches the CRC lie. + crate::git_format::pack_index::PackIndex::open(candidate.path(), format)?; + let index = upload_bytes( + &prepared.store, + prepared.descriptor.key(ArtifactKind::Index)?, + &bytes, + ) + .await?; + let descriptor = NativePackDescriptor { + index, + ..prepared.descriptor + }; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + assert!(matches!( + PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground) + ) + .await, + Err(PhysicalError::Native(GitHttpError::GitExit { .. })) + )); + drained(root.path(), &budget, 0).await?; + } + Ok(()) +} + +#[tokio::test] +async fn artifact_corruption_and_admission_limits_never_return_a_physical_verifier() -> Result { + let prepared = prepared(ObjectFormat::Sha256, 16).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let mut limits = physical_limits(); + limits.max_pack_bytes = prepared.descriptor.pack.size - 1; + assert!(matches!( + PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + prepared.descriptor, + limits, + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground) + ) + .await, + Err(PhysicalError::Limit) + )); + drained(root.path(), &budget, 0).await?; + let mut foreign = prepared.descriptor; + foreign.repository[0] ^= 1; + assert!(matches!( + PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + foreign, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground) + ) + .await, + Err(PhysicalError::Binding(_)) + )); + drained(root.path(), &budget, 0).await?; + let small_budget = + DiskBudget::new(prepared.descriptor.pack.size + prepared.descriptor.index.size - 1); + assert!(matches!( + PhysicalVerifier::download( + root.path(), + small_budget.clone(), + &prepared.store, + prepared.descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground) + ) + .await, + Err(PhysicalError::Metadata(MetadataError::Budget(_))) + )); + drained(root.path(), &small_budget, 0).await?; + let path = prepared.store.path( + prepared.descriptor.key(ArtifactKind::Pack)?, + prepared.descriptor.pack.digest, + )?; + prepared + .provider + .put( + &canopy_object_storage::external::part(&path, 0), + bytes::Bytes::from(vec![0; prepared.descriptor.pack.size as usize]).into(), + ) + .await?; + assert!(matches!( + PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + prepared.descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground) + ) + .await, + Err(PhysicalError::Metadata(MetadataError::Artifact(_))) + )); + drained(root.path(), &budget, 0).await?; + Ok(()) +} + +#[test] +fn canceling_confirmed_queued_shard_assembly_poisons_the_complete_pack() -> Result { + let runtime = tokio::runtime::Builder::new_multi_thread() + .worker_threads(1) + .max_blocking_threads(1) + .enable_all() + .build()?; + let prepared = runtime.block_on(prepared(ObjectFormat::Sha256, 16))?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let mut verifier = runtime.block_on(PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + prepared.descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ))?; + let entered = Arc::new(tokio::sync::Notify::new()); + let worker_entered = entered.clone(); + let (release, wait) = std::sync::mpsc::channel(); + let _blocker = runtime.spawn_blocking(move || { + worker_entered.notify_one(); + wait.recv().expect("release blocker"); + }); + runtime.block_on(entered.notified()); + runtime.block_on(async { + let mut pending = std::pin::pin!(verifier.inspect_next_shard(1)); + std::future::poll_fn(|context| { + assert!(pending.as_mut().poll(context).is_pending()); + std::task::Poll::Ready(()) + }) + .await; + }); + assert!(runtime.block_on(verifier.inspect_next_shard(1)).is_err()); + assert!(runtime.block_on(verifier.finish()).is_err()); + release.send(())?; + runtime.block_on(runtime.spawn_blocking(|| ()))?; + runtime.block_on(drained(root.path(), &budget, 0))?; + Ok(()) +} diff --git a/crates/canopy-server/src/packs/verification/physical/tests/independence.rs b/crates/canopy-server/src/packs/verification/physical/tests/independence.rs new file mode 100644 index 0000000..cd69919 --- /dev/null +++ b/crates/canopy-server/src/packs/verification/physical/tests/independence.rs @@ -0,0 +1,304 @@ +use super::*; +use crate::{ObjectKind, git_format::ObjectHasher}; +use std::io::Write; +use tokio::io::AsyncWriteExt; + +// Independent native-format fixtures, checked through stock Git below. This +// creates a one-object REF_DELTA whose base is absent from the physical pack. +fn crc32(bytes: &[u8]) -> u32 { + let mut crc = u32::MAX; + for byte in bytes { + crc ^= u32::from(*byte); + for _ in 0..8 { + crc = (crc >> 1) ^ (0xedb88320 & (0_u32.wrapping_sub(crc & 1))); + } + } + !crc +} +fn thin_pack(format: ObjectFormat, base: ObjectId) -> Result> { + let mut pack = b"PACK\0\0\0\x02\0\0\0\x01".to_vec(); + pack.push(0x78); // REF_DELTA, eight inflated delta-instruction bytes. + pack.extend_from_slice(&base); + let mut compressed = + flate2::write::ZlibEncoder::new(Vec::new(), flate2::Compression::default()); + compressed.write_all(b"\x04\x05\x05base!")?; + pack.extend_from_slice(&compressed.finish()?); + let mut hash = ObjectHasher::raw(format); + hash.update(&pack); + pack.extend_from_slice(&hash.finalize()); + Ok(pack) +} +fn single_index(format: ObjectFormat, oid: ObjectId, pack: &[u8]) -> Result> { + let width = format.bytes(); + let mut index = b"\xfftOc\0\0\0\x02".to_vec(); + for bucket in 0..256 { + index.extend_from_slice(&u32::from(bucket >= usize::from(oid[0])).to_be_bytes()); + } + index.extend_from_slice(&oid); + index.extend_from_slice(&crc32(&pack[12..pack.len() - width]).to_be_bytes()); + index.extend_from_slice(&12_u32.to_be_bytes()); + index.extend_from_slice(&pack[pack.len() - width..]); + let mut hash = ObjectHasher::raw(format); + hash.update(&index); + index.extend_from_slice(&hash.finalize()); + Ok(index) +} +pub(in crate::packs) async fn upload_pair( + prepared: &Prepared, + oid: ObjectId, + pack: &[u8], +) -> Result { + upload_pair_for_operation(prepared, oid, pack, [9; 16]).await +} +pub(in crate::packs) async fn upload_pair_for_operation( + prepared: &Prepared, + oid: ObjectId, + pack: &[u8], + operation: [u8; 16], +) -> Result { + let format = oid.format(); + let pack_digest = *blake3::hash(pack).as_bytes(); + let key = |kind| ArtifactKey { + operation, + binding_digest: pack_digest, + kind, + }; + let index_bytes = single_index(format, oid, pack)?; + let index = upload_bytes(&prepared.store, key(ArtifactKind::Index), &index_bytes).await?; + let pack_artifact = upload_bytes(&prepared.store, key(ArtifactKind::Pack), pack).await?; + Ok(NativePackDescriptor { + repository: prepared.descriptor.repository, + operation, + format, + git_checksum: ObjectId::try_from(&pack[pack.len() - format.bytes()..])?, + object_count: 1, + pack: pack_artifact, + index, + }) +} +pub(in crate::packs) async fn git_input( + root: &Path, + args: &[&str], + input: &[u8], +) -> Result> { + let mut command = crate::native_git::command(root)?; + let mut child = command + .args(args) + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .kill_on_drop(true) + .spawn()?; + child.stdin.take().ok_or("stdin")?.write_all(input).await?; + let output = child.wait_with_output().await?; + if !output.status.success() { + return Err(String::from_utf8_lossy(&output.stderr).into_owned().into()); + } + Ok(output.stdout) +} + +#[tokio::test] +async fn ambient_duplicate_cannot_hide_an_unresolved_delta_in_isolated_verification() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let prepared = prepared(format, 4).await?; + let base = crate::object_id(format, ObjectKind::Blob, b"base"); + let target = crate::object_id(format, ObjectKind::Blob, b"base!"); + let written = git_input( + prepared.fixture.root.path(), + &["hash-object", "-w", "--stdin"], + b"base", + ) + .await?; + assert_eq!(ObjectId::from_hex(written.trim_ascii())?, base); + // A shared cache can answer the OID from a duplicate loose object even + // when the physical pack cannot decode it independently. + let written = git_input( + prepared.fixture.root.path(), + &["hash-object", "-w", "--stdin"], + b"base!", + ) + .await?; + assert_eq!(ObjectId::from_hex(written.trim_ascii())?, target); + + let pack = thin_pack(format, base)?; + let descriptor = upload_pair(&prepared, target, &pack).await?; + let path = prepared.fixture.root.path().join(format!( + "objects/pack/pack-{}.pack", + hex::encode(descriptor.git_checksum) + )); + std::fs::write(&path, &pack)?; + std::fs::write( + path.with_extension("idx"), + single_index(format, target, &pack)?, + )?; + // Exact bytes, index/native checksums and artifact bindings all pass. + descriptor.verify_files(&path, &path.with_extension("idx"))?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let mut ambient = CanonicalVerifier::new( + prepared.fixture.root.path(), + format, + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )?; + let decoded = ambient + .inspect_to_disk(target, root.path(), budget.clone(), 0) + .await?; + assert_eq!(decoded.object().oid, target); + assert_eq!(decoded.object().size, 5); + assert_eq!(decoded.object().digest, *blake3::hash(b"base!").as_bytes()); + ambient.finish().await?; + drop(decoded); + let outcome = PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await; + match outcome { + Err(PhysicalError::Native(GitHttpError::GitExit { stderr, .. })) => { + assert!(stderr.contains("unresolved"), "{stderr}") + } + _ => panic!("isolated native verification must reject an unresolved delta"), + } + drained(root.path(), &budget, 0).await?; + } + Ok(()) +} + +#[tokio::test] +async fn completed_thin_pack_verifies_every_entry_including_the_appended_external_base() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let prepared = prepared(format, 4).await?; + let base = crate::object_id(format, ObjectKind::Blob, b"base"); + let target = crate::object_id(format, ObjectKind::Blob, b"base!"); + git_input( + prepared.fixture.root.path(), + &["hash-object", "-w", "--stdin"], + b"base", + ) + .await?; + let output = git_input( + prepared.fixture.root.path(), + &["index-pack", "--stdin", "--fix-thin", "--index-version=2"], + &thin_pack(format, base)?, + ) + .await?; + let checksum = ObjectId::from_hex( + output + .split(|byte| byte.is_ascii_whitespace()) + .rfind(|word| !word.is_empty()) + .ok_or("checksum")?, + )?; + let path = prepared + .fixture + .root + .path() + .join(format!("objects/pack/pack-{}.pack", hex::encode(checksum))); + let pack_bytes = std::fs::read(&path)?; + assert_eq!(u32::from_be_bytes(pack_bytes[8..12].try_into()?), 2); + let digest = *blake3::hash(&pack_bytes).as_bytes(); + let key = |kind| ArtifactKey { + operation: [10; 16], + binding_digest: digest, + kind, + }; + let pack = upload_bytes(&prepared.store, key(ArtifactKind::Pack), &pack_bytes).await?; + let index = upload_bytes( + &prepared.store, + key(ArtifactKind::Index), + &std::fs::read(path.with_extension("idx"))?, + ) + .await?; + let descriptor = NativePackDescriptor { + repository: prepared.descriptor.repository, + operation: [10; 16], + format, + git_checksum: checksum, + object_count: 2, + pack, + index, + }; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let mut verifier = PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + let segment = verifier.inspect_next_shard(2).await?; + let witness = verifier.finish().await?; + witness.verify_segments([segment.descriptor()])?; + for (oid, body) in [(base, b"base".as_slice()), (target, b"base!".as_slice())] { + let header = segment.header(oid)?.ok_or("decoded header")?; + assert_eq!(header.object.oid, oid); + assert_eq!(header.object.kind, ObjectKind::Blob); + assert_eq!(header.object.size, body.len() as u64); + assert_eq!(header.object.digest, *blake3::hash(body).as_bytes()); + assert_eq!(header.edge_count, 0); + } + drop(segment); + drained(root.path(), &budget, 0).await?; + } + Ok(()) +} + +#[tokio::test] +async fn isolated_pack_can_have_graph_dependencies_in_other_packs_without_external_delta_bases() +-> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let prepared = prepared(format, 16).await?; + let (commit, edges) = prepared + .fixture + .objects + .values() + .find(|(object, _)| object.kind == ObjectKind::Commit) + .ok_or("commit")?; + let pack = git_input( + prepared.fixture.root.path(), + &["pack-objects", "--stdout", "--no-reuse-delta"], + format!("{}\n", hex::encode(commit.oid)).as_bytes(), + ) + .await?; + assert_eq!(u32::from_be_bytes(pack[8..12].try_into()?), 1); + let descriptor = upload_pair(&prepared, commit.oid, &pack).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let mut verifier = PhysicalVerifier::download( + root.path(), + budget.clone(), + &prepared.store, + descriptor, + physical_limits(), + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + ) + .await?; + let segment = verifier.inspect_next_shard(1).await?; + let witness = verifier.finish().await?; + witness.verify_segments([segment.descriptor()])?; + assert_eq!( + segment.header(commit.oid)?.ok_or("commit header")?.object, + *commit + ); + let dependencies = segment.edges_after(commit.oid, None)?; + assert_eq!(dependencies, *edges); + for edge in dependencies { + assert!(segment.header(edge.child)?.is_none()); + } + // This stage correctly proves physical delta independence, while global + // graph closure still requires resolving these typed dependencies. + drop(segment); + drained(root.path(), &budget, 0).await?; + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/verification/spool.rs b/crates/canopy-server/src/packs/verification/spool.rs new file mode 100644 index 0000000..6b2f445 --- /dev/null +++ b/crates/canopy-server/src/packs/verification/spool.rs @@ -0,0 +1,293 @@ +use super::{CanonicalObject, EdgeSink, ObjectReadError}; +use crate::{ + ObjectId, ObjectKind, + packs::metadata::{AdmittedFile, MetadataError, PAGE_OBJECTS, TypedEdge, kind_code}, +}; +use cellule_ltx::DiskBudget; +use std::{ + io::{Read, Seek, SeekFrom, Write}, + path::{Path, PathBuf}, + sync::{ + Arc, Mutex, + atomic::{AtomicBool, Ordering}, + }, +}; + +/// A decoded canonical witness with privately owned, admitted dependencies. +/// Construction is restricted to a successful complete native inspection. +/// It is not a physical-pack or graph-closure certificate. +pub struct VerifiedObject { + object: CanonicalObject, + range: Option, + bytes: u64, + digest: [u8; 32], +} +struct EdgeRange { + storage: Arc>, + offset: u64, +} +impl VerifiedObject { + pub fn object(&self) -> CanonicalObject { + self.object + } + + /// Replay exact occurrence bytes in bounded pages. SQL deduplicates typed + /// edges; the digest detects changed scratch bytes before transaction commit. + pub(in crate::packs) fn replay( + &mut self, + mut append: impl FnMut(&[TypedEdge]) -> Result<(), MetadataError>, + ) -> Result<(), MetadataError> { + let Some(range) = self.range.as_ref() else { + return if self.bytes == 0 && self.digest == *blake3::hash(&[]).as_bytes() { + Ok(()) + } else { + Err(MetadataError::Integrity) + }; + }; + let mut storage = range.storage.lock().map_err(|_| MetadataError::Integrity)?; + let width = self.object.oid.len(); + let stride = width + 1; + if storage.failed + || self.bytes == 0 + || !self.bytes.is_multiple_of(stride as u64) + || range + .offset + .checked_add(self.bytes) + .is_none_or(|end| end > storage.bytes) + { + return Err(MetadataError::Integrity); + } + let stored_bytes = storage.bytes; + let file = storage.file.as_mut().ok_or(MetadataError::Integrity)?; + if file.file().as_file().metadata()?.len() != stored_bytes { + return Err(MetadataError::Integrity); + } + let input = file.file_mut().as_file_mut(); + input.seek(SeekFrom::Start(range.offset))?; + let mut buffer = [0; PAGE_OBJECTS * 33]; + let mut remaining = self.bytes; + let mut hash = blake3::Hasher::new(); + while remaining != 0 { + let size = remaining.min((PAGE_OBJECTS * stride) as u64) as usize; + input.read_exact(&mut buffer[..size])?; + hash.update(&buffer[..size]); + let mut edges = Vec::with_capacity(size / stride); + for record in buffer[..size].chunks_exact(stride) { + let child = + ObjectId::try_from(&record[..width]).map_err(|_| MetadataError::Integrity)?; + let expected_kind = match record[width] { + 1 => ObjectKind::Blob, + 2 => ObjectKind::Tree, + 3 => ObjectKind::Commit, + 4 => ObjectKind::Tag, + _ => return Err(MetadataError::Integrity), + }; + if child.is_zero() { + return Err(MetadataError::Integrity); + } + edges.push(TypedEdge { + child, + expected_kind, + }); + } + append(&edges)?; + remaining -= size as u64; + } + if hash.finalize().as_bytes() != &self.digest { + return Err(MetadataError::Integrity); + } + Ok(()) + } +} + +struct Storage { + file: Option, + bytes: u64, + failed: bool, +} + +/// One append-only dependency file for a bounded native-object batch. Each +/// complete witness owns an immutable range; keeping any witness or queued job +/// alive retains the complete file and its disk admission. +pub(super) struct EdgeSpool { + root: PathBuf, + budget: DiskBudget, + storage: Arc>, + writing: Arc, +} +impl EdgeSpool { + pub(super) fn new(root: &Path, budget: DiskBudget) -> Self { + Self { + root: root.to_owned(), + budget, + storage: Arc::new(Mutex::new(Storage { + file: None, + bytes: 0, + failed: false, + })), + writing: Arc::new(AtomicBool::new(false)), + } + } + pub(super) fn sink(&self, parent: ObjectId, limit: u64) -> Result { + // Hold exclusivity for an entire object, including between edge pages. + // Queued writes retain this permit after observer cancellation. + self.writing + .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) + .map_err(|_| ObjectReadError::Malformed)?; + Ok(DiskSink { + parent, + root: self.root.clone(), + budget: self.budget.clone(), + limit, + failed: false, + state: Arc::new(Mutex::new(State { + storage: Arc::clone(&self.storage), + offset: None, + bytes: 0, + hash: blake3::Hasher::new(), + _writer: AppendPermit(Arc::clone(&self.writing)), + })), + }) + } +} +struct AppendPermit(Arc); +impl Drop for AppendPermit { + fn drop(&mut self) { + self.0.store(false, Ordering::Release); + } +} +struct State { + storage: Arc>, + offset: Option, + bytes: u64, + hash: blake3::Hasher, + _writer: AppendPermit, +} +pub(super) struct DiskSink { + parent: ObjectId, + root: PathBuf, + budget: DiskBudget, + limit: u64, + failed: bool, + state: Arc>, +} +impl DiskSink { + pub(super) fn new(parent: ObjectId, root: &Path, budget: DiskBudget, limit: u64) -> Self { + // A fresh private spool cannot have another writer. + EdgeSpool::new(root, budget) + .sink(parent, limit) + .expect("fresh edge spool") + } + pub(super) fn complete( + self, + object: CanonicalObject, + ) -> Result { + if self.failed || object.oid != self.parent { + return Err(ObjectReadError::Malformed); + } + let state = Arc::try_unwrap(self.state) + .map_err(|_| ObjectReadError::Malformed)? + .into_inner() + .map_err(|_| ObjectReadError::Malformed)?; + let range = state.offset.map(|offset| EdgeRange { + storage: Arc::clone(&state.storage), + offset, + }); + Ok(VerifiedObject { + object, + range, + bytes: state.bytes, + digest: *state.hash.finalize().as_bytes(), + }) + } +} +impl EdgeSink for DiskSink { + async fn append( + &mut self, + parent: ObjectId, + edges: &[TypedEdge], + ) -> Result<(), ObjectReadError> { + if self.failed { + return Err(ObjectReadError::Malformed); + } + // Set before any await. A canceled append cannot produce a witness; + // queued blocking jobs retain the file and admission independently. + self.failed = true; + if parent != self.parent || edges.is_empty() || edges.len() > PAGE_OBJECTS { + return Err(ObjectReadError::Malformed); + } + let mut bytes = Vec::with_capacity(edges.len() * (parent.len() + 1)); + for edge in edges { + if edge.child.format() != parent.format() || edge.child.is_zero() { + return Err(ObjectReadError::Malformed); + } + bytes.extend_from_slice(&edge.child); + bytes.push(kind_code(edge.expected_kind)); + } + let state = Arc::clone(&self.state); + let root = self.root.clone(); + let budget = self.budget.clone(); + let limit = self.limit; + tokio::task::spawn_blocking(move || { + let mut state = state.lock().map_err(|_| ObjectReadError::Malformed)?; + let next = state + .bytes + .checked_add(bytes.len() as u64) + .filter(|size| *size <= limit) + .ok_or(ObjectReadError::TooLarge)?; + let storage = Arc::clone(&state.storage); + let mut storage = storage.lock().map_err(|_| ObjectReadError::Malformed)?; + if storage.failed { + return Err(ObjectReadError::Malformed); + } + let offset = state.offset.unwrap_or(storage.bytes); + if offset.checked_add(state.bytes) != Some(storage.bytes) { + return Err(ObjectReadError::Malformed); + } + let stored_next = storage + .bytes + .checked_add(bytes.len() as u64) + .ok_or(ObjectReadError::TooLarge)?; + // A partial write cannot be adopted by a later object. Failure + // poisons this file; all retained ranges reject replay. + storage.failed = true; + if storage.file.is_none() { + let reservation = budget + .try_reserve(bytes.len() as u64) + .map_err(std::io::Error::other)?; + let file = tempfile::Builder::new() + .prefix("canopy-verified-edges-") + .tempfile_in(root)?; + storage.file = Some(AdmittedFile::new(file, reservation)); + } else { + storage + .file + .as_mut() + .ok_or(ObjectReadError::Malformed)? + .reservation() + .try_grow(bytes.len() as u64) + .map_err(std::io::Error::other)?; + } + let stored_bytes = storage.bytes; + let file = storage.file.as_mut().ok_or(ObjectReadError::Malformed)?; + if file.file().as_file().metadata()?.len() != stored_bytes { + return Err(ObjectReadError::Malformed); + } + let output = file.file_mut().as_file_mut(); + output.seek(SeekFrom::Start(stored_bytes))?; + output.write_all(&bytes)?; + storage.bytes = stored_next; + storage.failed = false; + state.offset = Some(offset); + state.hash.update(&bytes); + state.bytes = next; + Ok::<_, ObjectReadError>(()) + }) + .await??; + self.failed = false; + Ok(()) + } +} + +#[cfg(test)] +mod tests; diff --git a/crates/canopy-server/src/packs/verification/spool/tests.rs b/crates/canopy-server/src/packs/verification/spool/tests.rs new file mode 100644 index 0000000..e514e90 --- /dev/null +++ b/crates/canopy-server/src/packs/verification/spool/tests.rs @@ -0,0 +1,417 @@ +use super::*; +use crate::{ + ObjectFormat, + packs::metadata::{ + MetadataBuilder, + tests::{fixture, limits}, + }, +}; +use std::future::Future; +type Result = std::result::Result>; + +#[test] +fn canceled_queued_write_retains_file_admission_and_cannot_complete() -> Result { + let runtime = tokio::runtime::Builder::new_multi_thread() + .worker_threads(1) + .max_blocking_threads(1) + .enable_all() + .build()?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(100); + let parent = crate::object_id(ObjectFormat::Sha256, ObjectKind::Tree, b"tree"); + let child = crate::object_id(ObjectFormat::Sha256, ObjectKind::Blob, b"blob"); + let edge = [TypedEdge { + child, + expected_kind: ObjectKind::Blob, + }]; + let mut sink = DiskSink::new(parent, root.path(), budget.clone(), 100); + runtime.block_on(sink.append(parent, &edge))?; + assert_eq!(budget.used(), 33); + let entered = Arc::new(tokio::sync::Notify::new()); + let worker_entered = entered.clone(); + let (release, wait) = std::sync::mpsc::channel(); + let _blocker = runtime.spawn_blocking(move || { + worker_entered.notify_one(); + wait.recv().expect("release blocker"); + }); + runtime.block_on(entered.notified()); + runtime.block_on(async { + let mut pending = std::pin::pin!(sink.append(parent, &edge)); + std::future::poll_fn(|context| { + assert!(pending.as_mut().poll(context).is_pending()); + std::task::Poll::Ready(()) + }) + .await; + // Drop the confirmed queued append here, before releasing its worker. + }); + assert!( + sink.complete(CanonicalObject { + oid: parent, + kind: ObjectKind::Tree, + size: 4, + digest: [0; 32] + }) + .is_err() + ); + assert_eq!(budget.used(), 33); + assert_eq!(std::fs::read_dir(root.path())?.count(), 1); + release.send(())?; + runtime.block_on(runtime.spawn_blocking(|| ()))?; + assert_eq!(budget.used(), 0); + assert_eq!(std::fs::read_dir(root.path())?.count(), 0); + Ok(()) +} + +#[tokio::test] +async fn late_scratch_corruption_rolls_back_entire_witness_batch_and_poison_sealing() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = fixture(format, 1600).await?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let mut builder = + MetadataBuilder::new(root.path(), budget.clone(), fixture.identity, limits())?; + let tree = fixture + .objects + .values() + .find(|(object, _)| object.kind == ObjectKind::Tree) + .ok_or("tree")? + .0 + .oid; + let commit = fixture + .objects + .values() + .find(|(object, _)| object.kind == ObjectKind::Commit) + .ok_or("commit")? + .0 + .oid; + let mut verifier = super::super::CanonicalVerifier::new( + fixture.root.path(), + format, + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )?; + let spool = EdgeSpool::new(root.path(), budget.clone()); + let first = verifier.inspect_to_spool(tree, &spool, 1 << 20).await?; + let corrupt = verifier.inspect_to_spool(commit, &spool, 1 << 20).await?; + assert!(Arc::ptr_eq( + &first.range.as_ref().ok_or("first range")?.storage, + &corrupt.range.as_ref().ok_or("second range")?.storage, + )); + assert!(corrupt.range.as_ref().ok_or("second range")?.offset > 0); + drop(spool); + { + let range = corrupt.range.as_ref().ok_or("spool")?; + let mut storage = range.storage.lock().map_err(|_| "spool lock")?; + let file = storage + .file + .as_mut() + .ok_or("spool file")? + .file_mut() + .as_file_mut(); + file.seek(SeekFrom::Start(range.offset))?; + let mut byte = [0]; + file.read_exact(&mut byte)?; + byte[0] ^= 1; // valid OID bytes; detected after all replay pages. + file.seek(SeekFrom::Start(range.offset))?; + file.write_all(&byte)?; + } + verifier.finish().await?; + assert!(matches!( + builder.put_verified_batch(vec![first, corrupt]), + Err(MetadataError::Integrity) + )); + let path = std::fs::read_dir(root.path())? + .find_map(|entry| { + entry + .ok() + .filter(|entry| { + entry + .file_name() + .to_string_lossy() + .starts_with("canopy-metadata-") + }) + .map(|entry| entry.path()) + }) + .ok_or("metadata")?; + let database = rusqlite::Connection::open_with_flags( + path, + rusqlite::OpenFlags::SQLITE_OPEN_READ_ONLY, + )?; + for table in ["objects", "object_edges"] { + let count: u64 = + database.query_row(&format!("SELECT count(*) FROM {table}"), [], |row| { + row.get(0) + })?; + assert_eq!(count, 0); + } + drop(database); + assert!(matches!( + builder.put_objects(&[fixture.objects[&commit].0]), + Err(MetadataError::Integrity) + )); + assert!(matches!( + builder.seal(&fixture.index), + Err(MetadataError::Integrity) + )); + assert_eq!(budget.used(), 0); + assert_eq!(std::fs::read_dir(root.path())?.count(), 0); + } + Ok(()) +} + +#[tokio::test] +async fn native_structural_page_uses_one_file_with_exact_independent_ranges() -> Result { + use crate::packs::verification::physical::tests::independence::git_input; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = fixture(format, 1).await?; + let initial = fixture + .objects + .values() + .find(|(object, _)| object.kind == ObjectKind::Commit) + .ok_or("initial commit")? + .0 + .oid; + let tree = fixture + .objects + .values() + .find(|(object, _)| object.kind == ObjectKind::Tree) + .ok_or("tree")? + .0 + .oid; + let mut input = Vec::new(); + for n in 1..=PAGE_OBJECTS { + let parent = if n == 1 { + hex::encode(initial) + } else { + format!(":{}", n - 1) + }; + input.extend_from_slice(format!("commit refs/heads/chain\nmark :{n}\ncommitter Test {} +0000\ndata 1\nx\nfrom {parent}\n\n", n + 1).as_bytes()); + } + git_input(fixture.root.path(), &["fast-import", "--quiet"], &input).await?; + let output = git_input( + fixture.root.path(), + &["rev-list", "--max-count=512", "refs/heads/chain"], + b"", + ) + .await?; + let ids = String::from_utf8(output)? + .lines() + .map(|line| { + ObjectId::try_from(hex::decode(line)?.as_slice()).map_err(|error| error.into()) + }) + .collect::>>()?; + assert_eq!(ids.len(), PAGE_OBJECTS); + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(64 << 10); + let spool = EdgeSpool::new(root.path(), budget.clone()); + let mut verifier = super::super::CanonicalVerifier::new( + fixture.root.path(), + format, + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )?; + let mut witnesses: Vec = Vec::with_capacity(PAGE_OBJECTS); + for (n, oid) in ids.iter().copied().enumerate() { + // Replaying an earlier range moves the shared file cursor. A later + // append must explicitly seek to the admitted end, never overwrite. + if let Some(previous) = witnesses.last_mut() { + previous.replay(|_| Ok(()))?; + } + let witness = verifier.inspect_to_spool(oid, &spool, 100).await?; + assert_eq!(witness.object().oid, oid); + assert_eq!(witness.object().kind, ObjectKind::Commit); + assert_eq!(witness.bytes, (2 * (format.bytes() + 1)) as u64); + assert_eq!( + witness.range.as_ref().ok_or("range")?.offset, + (n * 2 * (format.bytes() + 1)) as u64 + ); + witnesses.push(witness); + assert_eq!(std::fs::read_dir(root.path())?.count(), 1); + } + verifier.finish().await?; + drop(spool); + let admitted = (PAGE_OBJECTS * 2 * (format.bytes() + 1)) as u64; + assert_eq!(budget.used(), admitted); + // Reverse and repeat replay to exercise exact offsets and complete + // digest rechecking, including the behavior needed by SQLite retries. + for n in (0..PAGE_OBJECTS).rev() { + let expected = [ + TypedEdge { + child: tree, + expected_kind: ObjectKind::Tree, + }, + TypedEdge { + child: ids.get(n + 1).copied().unwrap_or(initial), + expected_kind: ObjectKind::Commit, + }, + ]; + for _ in 0..2 { + let mut actual = Vec::new(); + witnesses[n].replay(|page| { + actual.extend_from_slice(page); + Ok(()) + })?; + assert_eq!(actual, expected); + } + } + while witnesses.len() > 1 { + drop(witnesses.pop()); + } + assert_eq!(budget.used(), admitted); + assert_eq!(std::fs::read_dir(root.path())?.count(), 1); + drop(witnesses); + assert_eq!(budget.used(), 0); + assert_eq!(std::fs::read_dir(root.path())?.count(), 0); + } + Ok(()) +} + +#[test] +fn canceled_shared_writer_blocks_reuse_until_queued_work_drains() -> Result { + let runtime = tokio::runtime::Builder::new_multi_thread() + .worker_threads(1) + .max_blocking_threads(1) + .enable_all() + .build()?; + let root = tempfile::TempDir::new()?; + let budget = DiskBudget::new(100); + let parent = crate::object_id(ObjectFormat::Sha256, ObjectKind::Tree, b"tree"); + let child = crate::object_id(ObjectFormat::Sha256, ObjectKind::Blob, b"blob"); + let edge = [TypedEdge { + child, + expected_kind: ObjectKind::Blob, + }]; + let spool = EdgeSpool::new(root.path(), budget.clone()); + let mut sink = spool.sink(parent, 100)?; + runtime.block_on(sink.append(parent, &edge))?; + assert!(spool.sink(parent, 100).is_err()); + let (entered, wait_entered) = tokio::sync::oneshot::channel(); + let (release, wait) = std::sync::mpsc::channel(); + let blocker = runtime.spawn_blocking(move || { + let _ = entered.send(()); + wait.recv().expect("release"); + }); + runtime.block_on(wait_entered)?; + runtime.block_on(async { + let mut pending = std::pin::pin!(sink.append(parent, &edge)); + std::future::poll_fn(|cx| { + assert!(pending.as_mut().poll(cx).is_pending()); + std::task::Poll::Ready(()) + }) + .await; + }); + let object = CanonicalObject { + oid: parent, + kind: ObjectKind::Tree, + size: 4, + digest: [0; 32], + }; + assert!(sink.complete(object).is_err()); + assert!(spool.sink(parent, 100).is_err()); + assert_eq!(budget.used(), 33); + assert_eq!(std::fs::read_dir(root.path())?.count(), 1); + release.send(())?; + runtime.block_on(blocker)?; + runtime.block_on(runtime.spawn_blocking(|| ()))?; + assert_eq!(budget.used(), 66); + // The canceled object's bytes remain admitted but cannot enter a witness. + // A later writer starts after that orphan range, with its own exact digest. + let next_child = crate::object_id(ObjectFormat::Sha256, ObjectKind::Blob, b"next"); + let next_edge = [TypedEdge { + child: next_child, + expected_kind: ObjectKind::Blob, + }]; + let mut next = spool.sink(parent, 100)?; + runtime.block_on(next.append(parent, &next_edge))?; + let mut witness = next.complete(object)?; + assert_eq!(witness.range.as_ref().ok_or("range")?.offset, 66); + let mut actual = Vec::new(); + witness.replay(|edges| { + actual.extend_from_slice(edges); + Ok(()) + })?; + assert_eq!(actual, next_edge); + drop(spool); + assert_eq!(budget.used(), 99); + drop(witness); + assert_eq!(budget.used(), 0); + assert_eq!(std::fs::read_dir(root.path())?.count(), 0); + Ok(()) +} + +#[tokio::test] +async fn failed_shared_growth_and_file_length_changes_reject_retained_ranges() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let parent = crate::object_id(format, ObjectKind::Tree, b"tree"); + let child = crate::object_id(format, ObjectKind::Blob, b"blob"); + let edge = [TypedEdge { + child, + expected_kind: ObjectKind::Blob, + }]; + let object = CanonicalObject { + oid: parent, + kind: ObjectKind::Tree, + size: 4, + digest: [0; 32], + }; + for failure in ["admission", "write", "append", "truncate"] { + let root = tempfile::TempDir::new()?; + let stride = (format.bytes() + 1) as u64; + let budget = DiskBudget::new(if failure == "admission" { stride } else { 100 }); + let spool = EdgeSpool::new(root.path(), budget.clone()); + let mut sink = spool.sink(parent, 100)?; + sink.append(parent, &edge).await?; + let mut first = sink.complete(object)?; + let mut retained = stride; + if failure == "admission" { + let mut denied = spool.sink(parent, 100)?; + assert!(denied.append(parent, &edge).await.is_err()); + assert!(denied.complete(object).is_err()); + assert_eq!(budget.used(), stride); + } else if failure == "write" { + { + let range = first.range.as_ref().ok_or("range")?; + let mut storage = range.storage.lock().map_err(|_| "lock")?; + let file = storage.file.as_mut().ok_or("file")?.file_mut(); + // Keep the same admitted inode but make writes fail after + // credit growth and a successful end seek. + let read_only = std::fs::File::open(file.path())?; + *file.as_file_mut() = read_only; + } + let mut broken = spool.sink(parent, 100)?; + assert!(matches!( + broken.append(parent, &edge).await, + Err(ObjectReadError::Io(_)) + )); + assert!(broken.complete(object).is_err()); + retained = stride * 2; + assert_eq!(budget.used(), retained); + let mut later = spool.sink(parent, 100)?; + assert!(matches!( + later.append(parent, &edge).await, + Err(ObjectReadError::Malformed) + )); + assert!(later.complete(object).is_err()); + } else { + let range = first.range.as_ref().ok_or("range")?; + let storage = range.storage.lock().map_err(|_| "lock")?; + let file = storage.file.as_ref().ok_or("file")?.file().as_file(); + file.set_len(if failure == "append" { + stride + 1 + } else { + stride - 1 + })?; + } + assert!(matches!( + first.replay(|_| Ok(())), + Err(MetadataError::Integrity) + )); + drop(spool); + assert_eq!(budget.used(), retained); + drop(first); + assert_eq!(budget.used(), 0); + assert_eq!(std::fs::read_dir(root.path())?.count(), 0); + } + } + Ok(()) +} diff --git a/crates/canopy-server/src/packs/verification/tests.rs b/crates/canopy-server/src/packs/verification/tests.rs new file mode 100644 index 0000000..c11e22c --- /dev/null +++ b/crates/canopy-server/src/packs/verification/tests.rs @@ -0,0 +1,404 @@ +use super::*; +use crate::{ + ObjectKind, + packs::metadata::{PAGE_OBJECTS, TypedEdge, tests::fixture}, +}; +use std::{sync::Arc, time::Duration}; +use tokio::{io::AsyncWriteExt, process::Command}; + +type Result = std::result::Result>; + +#[tokio::test] +async fn admitted_native_witnesses_assemble_exact_metadata_and_release_scratch() -> Result { + use crate::packs::metadata::{MetadataBuilder, tests::limits}; + use cellule_ltx::DiskBudget; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = fixture(format, 1600).await?; + let scratch = tempfile::TempDir::new()?; + // A 16 MiB shard ceiling must not require a 48 MiB reservation before + // verification. This budget admits the real database and edge spools. + let budget = DiskBudget::new(2 << 20); + let mut builder = + MetadataBuilder::new(scratch.path(), budget.clone(), fixture.identity, limits())?; + let baseline = budget.used(); + let mut verifier = CanonicalVerifier::new( + fixture.root.path(), + format, + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )?; + let mut witnesses = Vec::with_capacity(PAGE_OBJECTS); + let mut spool = spool::EdgeSpool::new(scratch.path(), budget.clone()); + for oid in fixture.index.ids() { + let witness = verifier.inspect_to_spool(oid?, &spool, 1 << 20).await?; + assert_eq!(witness.object(), fixture.objects[&witness.object().oid].0); + witnesses.push(witness); + if witnesses.len() == PAGE_OBJECTS { + drop(spool); + builder.put_verified_batch(std::mem::take(&mut witnesses))?; + assert_eq!(budget.used(), builder.admitted_bytes()); + spool = spool::EdgeSpool::new(scratch.path(), budget.clone()); + } + } + drop(spool); + if !witnesses.is_empty() { + builder.put_verified_batch(witnesses)?; + } + // Repeated complete witnesses are canonical overlap, not extra objects + // or dependencies. Exercise the wide tree's multi-page replay twice. + let tree = fixture + .objects + .values() + .find(|(object, _)| object.kind == ObjectKind::Tree) + .ok_or("tree")? + .0 + .oid; + for _ in 0..2 { + let repeated = verifier + .inspect_to_disk(tree, scratch.path(), budget.clone(), 1 << 20) + .await?; + builder.put_verified(repeated)?; + } + verifier.finish().await?; + assert_eq!(budget.used(), builder.admitted_bytes()); + assert!(budget.used() > baseline); + let segment = builder.seal(&fixture.index)?; + for (oid, (expected, edges)) in &fixture.objects { + let header = segment.header(*oid)?.ok_or("header")?; + assert_eq!(header.object, *expected); + assert_eq!(header.edge_count, edges.len() as u64); + let mut actual = Vec::new(); + loop { + let page = + segment.edges_after(*oid, actual.last().map(|edge: &TypedEdge| edge.child))?; + if page.is_empty() { + break; + } + assert!(page.len() <= PAGE_OBJECTS); + actual.extend(page); + } + let mut expected = edges.clone(); + expected.sort_unstable_by_key(|edge| edge.child); + assert_eq!(actual, expected); + } + drop(segment); + assert_eq!(budget.used(), 0); + assert_eq!(std::fs::read_dir(scratch.path())?.count(), 0); + } + Ok(()) +} + +#[tokio::test] +async fn canonical_and_typed_overlap_conflicts_poison_verified_assembly() -> Result { + use crate::packs::metadata::{MetadataBuilder, MetadataError, tests::limits}; + use cellule_ltx::DiskBudget; + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = fixture(format, 700).await?; + let (tree, edges) = fixture + .objects + .values() + .find(|(object, _)| object.kind == ObjectKind::Tree) + .ok_or("tree")?; + for conflict_in_edges in [false, true] { + let scratch = tempfile::TempDir::new()?; + let budget = DiskBudget::new(128 << 20); + let mut builder = + MetadataBuilder::new(scratch.path(), budget.clone(), fixture.identity, limits())?; + let mut previous = *tree; + if !conflict_in_edges { + previous.digest[0] ^= 1; + } + builder.put_objects(&[previous])?; + if conflict_in_edges { + let edge = edges + .iter() + .find(|edge| edge.expected_kind == ObjectKind::Blob) + .ok_or("blob edge")?; + builder.put_edges( + tree.oid, + &[TypedEdge { + child: edge.child, + expected_kind: ObjectKind::Tree, + }], + )?; + } + let mut verifier = CanonicalVerifier::new( + fixture.root.path(), + format, + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )?; + let witness = verifier + .inspect_to_disk(tree.oid, scratch.path(), budget.clone(), 1 << 20) + .await?; + verifier.finish().await?; + assert!(matches!( + builder.put_verified(witness), + Err(MetadataError::IdentityConflict) + )); + let path = std::fs::read_dir(scratch.path())? + .find_map(|entry| { + entry + .ok() + .filter(|entry| { + entry + .file_name() + .to_string_lossy() + .starts_with("canopy-metadata-") + }) + .map(|entry| entry.path()) + }) + .ok_or("metadata")?; + let database = rusqlite::Connection::open_with_flags( + path, + rusqlite::OpenFlags::SQLITE_OPEN_READ_ONLY, + )?; + let objects: u64 = + database.query_row("SELECT count(*) FROM objects", [], |row| row.get(0))?; + let edges: u64 = + database.query_row("SELECT count(*) FROM object_edges", [], |row| row.get(0))?; + assert_eq!(objects, 1); + assert_eq!(edges, u64::from(conflict_in_edges)); + drop(database); + assert!(matches!( + builder.seal(&fixture.index), + Err(MetadataError::Integrity) + )); + assert_eq!(budget.used(), 0); + } + } + Ok(()) +} + +#[tokio::test] +async fn edge_spool_quota_failure_prevents_witness_and_actor_reuse() -> Result { + use cellule_ltx::DiskBudget; + let fixture = fixture(ObjectFormat::Sha256, 1600).await?; + let tree = fixture + .objects + .values() + .find(|(object, _)| object.kind == ObjectKind::Tree) + .ok_or("tree")? + .0 + .oid; + for (capacity, limit) in [(512 * 33, 1 << 20), (1 << 20, 512 * 33)] { + let scratch = tempfile::TempDir::new()?; + let budget = DiskBudget::new(capacity); + let mut verifier = CanonicalVerifier::new( + fixture.root.path(), + ObjectFormat::Sha256, + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )?; + assert!( + verifier + .inspect_to_disk(tree, scratch.path(), budget.clone(), limit) + .await + .is_err() + ); + assert!( + verifier + .inspect(tree, &mut Collector::default()) + .await + .is_err() + ); + assert!(verifier.finish().await.is_err()); + assert_eq!(budget.used(), 0); + assert_eq!(std::fs::read_dir(scratch.path())?.count(), 0); + } + Ok(()) +} +#[derive(Default)] +struct Collector { + edges: Vec, + parent: Option, + largest: usize, +} +impl EdgeSink for Collector { + async fn append( + &mut self, + parent: ObjectId, + edges: &[TypedEdge], + ) -> std::result::Result<(), ObjectReadError> { + assert!(!edges.is_empty() && edges.len() <= PAGE_OBJECTS); + assert!(self.parent.is_none_or(|expected| expected == parent)); + self.parent = Some(parent); + self.largest = self.largest.max(edges.len()); + self.edges.extend_from_slice(edges); + Ok(()) + } +} +#[tokio::test] +async fn native_streamed_witnesses_match_headers_and_typed_edges_with_bounded_batches() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = fixture(format, 700).await?; + let mut verifier = CanonicalVerifier::new( + fixture.root.path(), + format, + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )?; + // File-backed index iteration replaces a heap inventory for this path. + for oid in fixture.index.ids() { + let oid = oid?; + let mut sink = Collector::default(); + let canonical = verifier.inspect(oid, &mut sink).await?; + let (expected, edges) = fixture.objects.get(&oid).ok_or("object")?; + assert_eq!(canonical, *expected); + sink.edges + .sort_unstable_by_key(|edge| (edge.child, edge.expected_kind.git_name())); + sink.edges.dedup(); + let mut expected = edges.clone(); + expected.sort_unstable_by_key(|edge| (edge.child, edge.expected_kind.git_name())); + assert_eq!(sink.edges, expected); + assert!(sink.largest <= PAGE_OBJECTS); + } + verifier.finish().await?; + } + Ok(()) +} + +#[tokio::test] +async fn large_native_commit_messages_are_hashed_without_entering_the_edge_inventory() -> Result { + for format in [ObjectFormat::Sha1, ObjectFormat::Sha256] { + let fixture = fixture(format, 4).await?; + let tree = fixture + .objects + .values() + .find(|(header, _)| header.kind == ObjectKind::Tree) + .ok_or("tree")? + .0 + .oid; + let mut body = format!("tree {}\nauthor Verifier 1 +0000\ncommitter Verifier 1 +0000\n\n", hex::encode(tree)).into_bytes(); + body.extend(std::iter::repeat_n(b'x', 8 << 20)); + let expected = crate::object_id(format, ObjectKind::Commit, &body); + let mut command: Command = crate::native_git::command(fixture.root.path())?; + let mut child = command + .arg("--git-dir") + .arg(fixture.root.path()) + .args(["hash-object", "-t", "commit", "-w", "--stdin"]) + .stdin(std::process::Stdio::piped()) + .stdout(std::process::Stdio::piped()) + .stderr(std::process::Stdio::piped()) + .kill_on_drop(true) + .spawn()?; + child.stdin.take().ok_or("stdin")?.write_all(&body).await?; + let output = child.wait_with_output().await?; + assert!( + output.status.success(), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + assert_eq!(ObjectId::from_hex(output.stdout.trim_ascii())?, expected); + let mut verifier = CanonicalVerifier::new( + fixture.root.path(), + format, + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )?; + let mut sink = Collector::default(); + let canonical = verifier.inspect(expected, &mut sink).await?; + assert_eq!(canonical.size, body.len() as u64); + assert_eq!(canonical.digest, *blake3::hash(&body).as_bytes()); + assert_eq!( + sink.edges, + vec![TypedEdge { + child: tree, + expected_kind: ObjectKind::Tree + }] + ); + verifier.finish().await?; + } + Ok(()) +} +struct FailedSink; +impl EdgeSink for FailedSink { + async fn append( + &mut self, + _: ObjectId, + _: &[TypedEdge], + ) -> std::result::Result<(), ObjectReadError> { + Err(std::io::Error::other("injected spool failure").into()) + } +} +#[tokio::test] +async fn sink_failure_poisoning_prevents_another_inspection_or_successful_finish() -> Result { + let fixture = fixture(ObjectFormat::Sha256, 4).await?; + let tree = fixture + .objects + .values() + .find(|(header, _)| header.kind == ObjectKind::Tree) + .ok_or("tree")? + .0 + .oid; + let mut verifier = CanonicalVerifier::new( + fixture.root.path(), + ObjectFormat::Sha256, + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )?; + assert!(matches!( + verifier.inspect(tree, &mut FailedSink).await, + Err(ObjectReadError::Io(_)) + )); + assert!(matches!( + verifier.inspect(tree, &mut Collector::default()).await, + Err(ObjectReadError::Malformed) + )); + assert!(matches!( + verifier.finish().await, + Err(ObjectReadError::Malformed) + )); + Ok(()) +} +struct PausedSink { + entered: Arc, +} +impl EdgeSink for PausedSink { + async fn append( + &mut self, + _: ObjectId, + _: &[TypedEdge], + ) -> std::result::Result<(), ObjectReadError> { + self.entered.notify_one(); + std::future::pending().await + } +} +#[tokio::test] +async fn canceling_a_confirmed_pending_sink_write_poisons_native_frame_reuse() -> Result { + let fixture = fixture(ObjectFormat::Sha256, 700).await?; + let tree = fixture + .objects + .values() + .find(|(header, _)| header.kind == ObjectKind::Tree) + .ok_or("tree")? + .0 + .oid; + let mut verifier = CanonicalVerifier::new( + fixture.root.path(), + ObjectFormat::Sha256, + &crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), + )?; + let entered = Arc::new(tokio::sync::Notify::new()); + let mut sink = PausedSink { + entered: Arc::clone(&entered), + }; + let mut inspection = Box::pin(verifier.inspect(tree, &mut sink)); + tokio::time::timeout(Duration::from_secs(5), async { + tokio::select! { + _ = entered.notified() => Ok::<_, Box>(()), + result = inspection.as_mut() => Err(format!("inspection unexpectedly finished: {result:?}").into()), + } + }).await??; + drop(inspection); + assert!(matches!( + verifier.inspect(tree, &mut Collector::default()).await, + Err(ObjectReadError::Malformed) + )); + assert!(matches!( + verifier.finish().await, + Err(ObjectReadError::Malformed) + )); + Ok(()) +} diff --git a/crates/canopy-server/src/packs/wire_request.rs b/crates/canopy-server/src/packs/wire_request.rs new file mode 100644 index 0000000..869d8a1 --- /dev/null +++ b/crates/canopy-server/src/packs/wire_request.rs @@ -0,0 +1,183 @@ +//! Immutable original request bytes and metadata. This root is custody, not +//! signature, native-result, ref-CAS or publication authority. +use super::directory::index::codec::{artifact, fixed, read_artifact}; +use crate::{ObjectFormat, git_http::GitHttpRequest, packs::publication::BeginRequest}; +use canopy_object_storage::artifact::{ + ArtifactDescriptor, ArtifactKey, ArtifactKind, ArtifactStore, +}; +use cellule_runtime::{ + ApplicationId, TenantId, + codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue}, +}; + +const DOMAIN: &[u8] = b"canopy.wire-request.v1\0"; +pub const REQUEST_ROOT_BYTES: u32 = 64 << 10; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct WireRequestRoot(super::input_artifact::StoredInputRoot); +impl WireRequestRoot { + pub fn operation(self) -> [u8; 16] { + self.0.operation + } + pub fn artifact(self) -> ArtifactDescriptor { + self.0.artifact + } + pub(crate) fn validate(self) -> Result<(), CodecError> { + self.0.validate(REQUEST_ROOT_BYTES) + } + pub(crate) async fn upload( + store: &ArtifactStore, + record: WireRequest, + ) -> Result { + record.validate()?; + if record.identity.repository != store.repository() { + return Err(WireRequestError::Context); + } + Ok(Self( + super::input_artifact::StoredInputRoot::upload( + store, + record.operation, + &record, + REQUEST_ROOT_BYTES, + ) + .await?, + )) + } + pub(crate) async fn read(self, store: &ArtifactStore) -> Result { + let record: WireRequest = self.0.read(store, REQUEST_ROOT_BYTES).await?; + if record.operation != self.operation() || record.identity.repository != store.repository() + { + return Err(WireRequestError::Context); + } + Ok(record) + } +} +impl WireValue for WireRequestRoot { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.validate()?; + self.0.encode(e) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + let value = Self(super::input_artifact::StoredInputRoot::decode(d)?); + value.validate()?; + Ok(value) + } +} +pub(crate) struct WireRequest { + pub tenant: TenantId, + pub application: ApplicationId, + pub operation: [u8; 16], + pub identity: BeginRequest, + pub format: ObjectFormat, + pub request: GitHttpRequest, +} +impl WireRequest { + pub(crate) fn matches( + &self, + target: &cellule_runtime::CellTarget, + check: &crate::packs::publication::LeaseCheck, + format: ObjectFormat, + ) -> bool { + self.tenant == target.tenant() + && self.application == target.application() + && self.format == format + && self.identity.repository == check.token.repository + && self.identity.operation == check.token.operation + && self.identity.request_digest == check.token.request_digest + && self.identity.actor == check.actor + && crate::repository_target(self.tenant, self.application, self.identity.repository) + .is_ok_and(|expected| expected == *target) + } + fn validate(&self) -> Result<(), CodecError> { + super::publication::codec::artifact_valid(self.operation)?; + crate::repository_target(self.tenant, self.application, self.identity.repository) + .map_err(|_| CodecError::Invalid("wire request target"))?; + if self.operation == [0; 16] + || self.request.method != "POST" + || self.request.path_info != "/repo.git/git-receive-pack" + || !self.request.authenticated + || self.request.body.size > canopy_object_storage::external::MAX_ARTIFACT_BYTES + || self.request.body.manifest_digest == [0; 32] + { + return Err(CodecError::Invalid("wire request metadata")); + } + Ok(()) + } + pub(crate) fn body_key(&self) -> ArtifactKey { + ArtifactKey { + operation: self.operation, + binding_digest: self.request.body.digest, + kind: ArtifactKind::InputBody, + } + } +} +impl WireValue for WireRequest { + fn encode(&self, e: &mut BoundedEncoder) -> Result<(), CodecError> { + self.validate()?; + e.write_bytes(DOMAIN)?; + e.write_bytes(self.tenant.as_bytes())?; + e.write_bytes(self.application.as_bytes())?; + e.write_bytes(&self.operation)?; + self.identity.encode(e)?; + e.write_u8(self.format.bytes() as u8)?; + e.write_text(&self.request.method)?; + e.write_text(&self.request.path_info)?; + e.write_text(&self.request.query)?; + e.write_bool(self.request.content_type.is_some())?; + if let Some(content_type) = &self.request.content_type { + e.write_text(content_type)?; + } + e.write_bool(self.request.gzip)?; + e.write_bool(self.request.protocol_v2)?; + artifact(e, self.request.body) + } + fn decode(d: &mut BoundedDecoder<'_>) -> Result { + if d.read_bytes()? != DOMAIN { + return Err(CodecError::Invalid("wire request domain")); + } + let tenant = TenantId::from_bytes(fixed(d)?); + let application = ApplicationId::from_bytes(fixed(d)?); + let operation = fixed(d)?; + let identity = BeginRequest::decode(d)?; + let format = match d.read_u8()? { + 20 => ObjectFormat::Sha1, + 32 => ObjectFormat::Sha256, + _ => return Err(CodecError::Invalid("wire request format")), + }; + let request = GitHttpRequest { + method: d.read_text()?.into(), + path_info: d.read_text()?.into(), + query: d.read_text()?.into(), + content_type: if d.read_bool()? { + Some(d.read_text()?.into()) + } else { + None + }, + gzip: d.read_bool()?, + protocol_v2: d.read_bool()?, + body: read_artifact(d)?, + authenticated: true, + }; + let value = Self { + tenant, + application, + operation, + identity, + format, + request, + }; + value.validate()?; + Ok(value) + } +} +#[derive(Debug, thiserror::Error)] +pub enum WireRequestError { + #[error("original request metadata transport failed")] + Root(#[from] super::InputRootError), + #[error("wire request artifact failed")] + Artifact(#[from] canopy_object_storage::artifact::ArtifactError), + #[error("wire request codec failed")] + Codec(#[from] CodecError), + #[error("wire request context differs")] + Context, +} diff --git a/crates/canopy-server/src/push/certificate.rs b/crates/canopy-server/src/push/certificate.rs index 37c5a34..e5ee332 100644 --- a/crates/canopy-server/src/push/certificate.rs +++ b/crates/canopy-server/src/push/certificate.rs @@ -10,10 +10,13 @@ pub struct PushCertificateReceipt { pub recorded_at_ms: i64, } -pub(crate) struct VerifiedPushCertificate { - pub body: Vec, - pub signer: String, - pub key: String, +/// Native-verified signed push. Its construction is confined to the gateway. +pub struct VerifiedPushCertificate { + pub(crate) target: cellule_runtime::CellTarget, + pub(crate) request_digest: [u8; 32], + pub(crate) body: Vec, + pub(crate) signer: String, + pub(crate) key: String, } pub(super) struct CertificateMeta { diff --git a/crates/canopy-server/src/push/mod.rs b/crates/canopy-server/src/push/mod.rs index 9cadb2d..0a8e8d8 100644 --- a/crates/canopy-server/src/push/mod.rs +++ b/crates/canopy-server/src/push/mod.rs @@ -39,14 +39,13 @@ pub(crate) fn valid_options(options: &[String]) -> bool { mod certificate; mod plan; -pub use certificate::PushCertificateReceipt; -pub(crate) use certificate::VerifiedPushCertificate; use certificate::{CertificateMeta, certificate_complete}; +pub use certificate::{PushCertificateReceipt, VerifiedPushCertificate}; pub(crate) mod report; use plan::StagedPlan; -const CHUNK_BYTES: usize = 512 * 1024; -const MAX_RESPONSE_BYTES: usize = 64 * 1024 * 1024; +pub(crate) const CHUNK_BYTES: usize = 512 * 1024; +pub(crate) const MAX_RESPONSE_BYTES: usize = 64 * 1024 * 1024; #[derive(Debug, thiserror::Error)] pub enum PushError { @@ -320,6 +319,11 @@ impl RepositoryCell { ))); } let certificate = if let Some(certificate) = &input.certificate { + if certificate.target != self.target || certificate.request_digest != input.digest { + return Err(InvocationError::NotStarted(Error::Command( + "signed push witness context differs", + ))); + } Some(self.stage_push_certificate(input.id, certificate).await?) } else { None diff --git a/crates/canopy-server/src/push/report.rs b/crates/canopy-server/src/push/report.rs index 475cefc..f1e9454 100644 --- a/crates/canopy-server/src/push/report.rs +++ b/crates/canopy-server/src/push/report.rs @@ -4,6 +4,98 @@ pub(crate) const REJECTED: &str = "Canopy publication rejected: refs, permissions or policy changed; fetch and retry"; const PACKET_BYTES: usize = 65520; +/// Bind native success reports to the exact updates being published. Native +/// failures can coexist with successful updates in a non-atomic push. A failure +/// or empty-command outcome cannot acknowledge any unvalidated successful ref. +pub(crate) fn publication_matches( + response: &GitHttpResponse, + plan: Option<&PushPlan>, +) -> Result<(), PushError> { + use std::{borrow::Cow, collections::BTreeSet}; + if response.status != 200 { + return if plan.is_none() { + Ok(()) + } else { + Err(PushError::InvalidResponse) + }; + } + let mut bytes = response.body.as_slice(); + let first = if bytes.is_empty() { + None + } else { + packet(&mut bytes)? + }; + let data = if first.is_some_and(|payload| matches!(payload.first(), Some(1..=3))) { + let mut data = Vec::new(); + let mut next = first; + while let Some(payload) = next { + match payload.split_first() { + Some((1, body)) => data.extend_from_slice(body), + Some((2, _)) => {} + _ => return Err(PushError::InvalidResponse), + } + next = packet(&mut bytes)?; + } + if !bytes.is_empty() { + return Err(PushError::InvalidResponse); + } + Cow::Owned(data) + } else { + Cow::Borrowed(response.body.as_slice()) + }; + // A client may decline report-status. The service's private native witness + // then supplies the ref plan; final CAS and the certificate still bind it. + if data.is_empty() || data.as_ref() == b"0000" { + return Ok(()); + } + let mut bytes = data.as_ref(); + let unpack = packet(&mut bytes)?.ok_or(PushError::InvalidResponse)?; + if !unpack.starts_with(b"unpack ") + || !unpack.ends_with(b"\n") + || (plan.is_some() && unpack != b"unpack ok\n") + { + return Err(PushError::InvalidResponse); + } + let mut expected: BTreeSet<&str> = plan + .into_iter() + .flat_map(|plan| plan.updates.iter().map(|update| update.name.as_str())) + .collect(); + let mut seen = BTreeSet::new(); + while let Some(payload) = packet(&mut bytes)? { + let (success, name) = if let Some(name) = payload + .strip_prefix(b"ok ") + .and_then(|value| value.strip_suffix(b"\n")) + { + (true, name) + } else if let Some(failure) = payload + .strip_prefix(b"ng ") + .and_then(|value| value.strip_suffix(b"\n")) + { + let mut parts = failure.splitn(2, |byte| *byte == b' '); + let name = parts.next().ok_or(PushError::InvalidResponse)?; + if parts.next().is_none_or(|reason| reason.is_empty()) { + return Err(PushError::InvalidResponse); + } + (false, name) + } else { + return Err(PushError::InvalidResponse); + }; + let name = std::str::from_utf8(name).map_err(|_| PushError::InvalidResponse)?; + if !crate::refs::valid_ref_name(name) + || !seen.insert(name) + || seen.len() > crate::refs::MAX_UPDATES + || (success && (unpack != b"unpack ok\n" || !expected.remove(name))) + || (!success && expected.contains(name)) + { + return Err(PushError::InvalidResponse); + } + } + if !bytes.is_empty() || !expected.is_empty() { + return Err(PushError::InvalidResponse); + } + Ok(()) +} + pub(crate) fn rejected_commands<'a>( names: impl Iterator, report_status: bool, diff --git a/crates/canopy-server/src/refs.rs b/crates/canopy-server/src/refs.rs index 0aa4cc1..c498186 100644 --- a/crates/canopy-server/src/refs.rs +++ b/crates/canopy-server/src/refs.rs @@ -204,28 +204,9 @@ impl RepositoryCell { impl WireValue for PushPlan { fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - if validate_component(&self.actor).is_err() { - return Err(CodecError::Invalid("invalid push actor")); - } - if self.updates.is_empty() || self.updates.len() > MAX_UPDATES { - return Err(CodecError::Invalid("push update count is outside bounds")); - } - encoder.write_text(&self.actor)?; - encoder.write_count(self.updates.len())?; + self.encode_prefix(encoder)?; for update in &self.updates { - encoder.write_text(&update.name)?; - encoder.write_bool(update.expected.is_some())?; - if let Some(expected) = &update.expected { - encoder.write_bool(expected.oid.is_some())?; - if let Some(oid) = expected.oid { - encoder.write_bytes(&oid)?; - } - encoder.write_i64(expected.version)?; - } - encoder.write_bool(update.new_oid.is_some())?; - if let Some(oid) = update.new_oid { - encoder.write_bytes(&oid)?; - } + encode_update(update, encoder)?; } Ok(()) } @@ -268,6 +249,67 @@ impl WireValue for PushPlan { Ok(Self { actor, updates }) } } +impl PushPlan { + pub(crate) fn encode_range( + &self, + range: std::ops::Range, + encoder: &mut BoundedEncoder, + ) -> Result<(), CodecError> { + let updates = self + .updates + .get(range) + .ok_or(CodecError::Invalid("push plan range"))?; + encode_plan_prefix(&self.actor, updates.len(), encoder)?; + for update in updates { + encode_update(update, encoder)?; + } + Ok(()) + } + /// Shared wire prefix and update encoding allow bounded incremental hashing + /// without allocating a second full copy of a large mirror plan. + pub(crate) fn encode_prefix(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { + encode_plan_prefix(&self.actor, self.updates.len(), encoder) + } +} +fn encode_plan_prefix( + actor: &str, + count: usize, + encoder: &mut BoundedEncoder, +) -> Result<(), CodecError> { + if validate_component(actor).is_err() { + return Err(CodecError::Invalid("invalid push actor")); + } + if count == 0 || count > MAX_UPDATES { + return Err(CodecError::Invalid("push update count is outside bounds")); + } + encoder.write_text(actor)?; + encoder.write_count(count) +} +pub(crate) fn encode_update( + update: &RefUpdate, + encoder: &mut BoundedEncoder, +) -> Result<(), CodecError> { + encoder.write_text(&update.name)?; + encode_update_suffix(update, encoder) +} +pub(crate) fn encode_update_suffix( + update: &RefUpdate, + encoder: &mut BoundedEncoder, +) -> Result<(), CodecError> { + encoder.write_bool(update.expected.is_some())?; + if let Some(expected) = &update.expected { + encoder.write_bool(expected.oid.is_some())?; + if let Some(oid) = expected.oid { + encoder.write_bytes(&oid)?; + } + encoder.write_i64(expected.version)?; + } + encoder.write_bool(update.new_oid.is_some())?; + if let Some(oid) = update.new_oid { + encoder.write_bytes(&oid)?; + } + Ok(()) +} fn read_oid(decoder: &mut BoundedDecoder<'_>) -> Result { decoder @@ -305,16 +347,41 @@ pub(crate) fn apply_refs( plan: &PushPlan, merge: Option<&crate::pulls::merge::ReviewedMerge>, ) -> cellule_runtime::Result { - if plan.updates.is_empty() || plan.updates.len() > MAX_UPDATES { + let Some(validated) = validate_refs(context, plan)? else { + return Ok(false); + }; + if !crate::graph::certified_roots(context, plan)? + || !crate::branch_rules::policies_allow(context, plan, merge)? + { return Ok(false); } + validated.apply(context)?; + Ok(true) +} + +/// Only ref/ACL validation constructs this result. Catalog membership and +/// current branch/merge policy must also pass before consuming it. It borrows +/// the immutable plan and cannot be reused by another admitted command. +pub(crate) struct ValidatedRefs<'plan> { + plan: &'plan PushPlan, + target: cellule_runtime::CellTarget, + owner: cellule_runtime::registry::OwnerFence, + sequence: u64, +} +pub(crate) fn validate_refs<'plan>( + context: &CommandContext<'_, '_>, + plan: &'plan PushPlan, +) -> cellule_runtime::Result>> { + if plan.updates.is_empty() || plan.updates.len() > MAX_UPDATES { + return Ok(None); + } if validate_component(&plan.actor).is_err() || !decode_access(&context.sql(&SqlBatch { statements: vec![access_statement(&plan.actor)], })?)? .is_some_and(|level| level >= TokenScope::Write) { - return Ok(false); + return Ok(None); } let identity = context.sql(&SqlBatch { statements: vec![SqlStatement { @@ -339,7 +406,7 @@ pub(crate) fn apply_refs( .and_then(|old| old.oid) .is_some_and(|oid| oid.format() != format) }) { - return Ok(false); + return Ok(None); } let updates: BTreeMap<_, _> = plan .updates @@ -347,7 +414,7 @@ pub(crate) fn apply_refs( .map(|update| (update.name.as_str(), update)) .collect(); if updates.len() != plan.updates.len() { - return Ok(false); + return Ok(None); } // A first mirror push can contain thousands of refs. One transactional // emptiness check avoids a point lookup and namespace scan for every new @@ -375,15 +442,15 @@ pub(crate) fn apply_refs( .as_ref() .is_some_and(|old| old.version <= 0 || old.version == i64::MAX) { - return Ok(false); + return Ok(None); } if update.new_oid.is_none() && update.expected.as_ref().and_then(|old| old.oid).is_none() { - return Ok(false); + return Ok(None); } if (refs_empty && update.expected.is_some()) || (!refs_empty && current_ref(context, &update.name)? != update.expected) { - return Ok(false); + return Ok(None); } } for update in plan @@ -392,16 +459,27 @@ pub(crate) fn apply_refs( .filter(|update| update.new_oid.is_some()) { if existing_namespace_conflict(context, &updates, &update.name, refs_empty)? { - return Ok(false); + return Ok(None); } } - if !crate::graph::certified_roots(context, plan)? - || !crate::branch_rules::policies_allow(context, plan, merge)? - { - return Ok(false); - } - for update in &plan.updates { - let result = match (&update.expected, update.new_oid) { + Ok(Some(ValidatedRefs { + plan, + target: context.target().clone(), + owner: context.owner_fence(), + sequence: context.sequence(), + })) +} +impl ValidatedRefs<'_> { + pub(crate) fn apply(self, context: &mut CommandContext<'_, '_>) -> cellule_runtime::Result<()> { + if context.target() != &self.target + || context.owner_fence() != self.owner + || context.sequence() != self.sequence + { + return Err(Error::Command("ref validation belongs to another command")); + } + let plan = self.plan; + for update in &plan.updates { + let result = match (&update.expected, update.new_oid) { (None, Some(new_oid)) => context.sql(&SqlBatch { statements: vec![SqlStatement { sql: "INSERT INTO refs (name, oid, version) VALUES (?1, ?2, 1)".into(), @@ -425,14 +503,15 @@ pub(crate) fn apply_refs( })?, (None, None) => return Err(Error::Command("empty ref mutation")), }; - if result.first().is_none_or(|set| set.rows_affected != 1) { - return Err(Error::Command("ref CAS changed no rows")); + if result.first().is_none_or(|set| set.rows_affected != 1) { + return Err(Error::Command("ref CAS changed no rows")); + } } + // Both typed pushes and HTTP completion pass here. Advance only with the + // ref transaction so paginated readers reject mixed generations, including ABA. + advance_generation(context)?; + Ok(()) } - // Both typed pushes and HTTP completion pass here. Advance only with the - // ref transaction so paginated readers reject mixed generations, including ABA. - advance_generation(context)?; - Ok(true) } pub(crate) fn server_owned_ref(name: &str) -> bool { diff --git a/crates/canopy-server/src/server/lifecycle.rs b/crates/canopy-server/src/server/lifecycle.rs index 32e9bd6..a3b4df8 100644 --- a/crates/canopy-server/src/server/lifecycle.rs +++ b/crates/canopy-server/src/server/lifecycle.rs @@ -109,6 +109,7 @@ impl CanopyServer { impl RunningServer { async fn shutdown(self) -> Result<(), ServerError> { + self.native.close(); self.maintenance_stop.cancel(); self.ingress_stop.cancel(); let serving = self.serving.await; @@ -119,6 +120,10 @@ impl RunningServer { }; self.tasks.close(); self.tasks.wait().await; + // Detached native reapers and blocking verifiers outlive their request + // observers. Keep Cell authority, heartbeat and workspace until every + // admitted owner releases its claim. Uncertain drain stays pending. + self.native.drain().await; let drained = self.node.shutdown().await; if drained.is_ok() { self.local.confirm_drained(); @@ -140,3 +145,6 @@ impl RunningServer { Ok(()) } } + +#[cfg(test)] +mod tests; diff --git a/crates/canopy-server/src/server/lifecycle/tests.rs b/crates/canopy-server/src/server/lifecycle/tests.rs new file mode 100644 index 0000000..2873276 --- /dev/null +++ b/crates/canopy-server/src/server/lifecycle/tests.rs @@ -0,0 +1,89 @@ +use super::*; +use crate::native_resources::{NativeClass, NativeLimits, NativeWork}; +use object_store::memory::InMemory; + +#[tokio::test(flavor = "multi_thread")] +async fn native_drain_retains_cell_workspace_and_lease_past_one_lease() +-> Result<(), Box> { + let files = tempfile::TempDir::new()?; + let data = files.path().join("node"); + let server = RunningServer::start( + ServerConfig { + tenant: TenantId::from_bytes([51; 16]), + application: ApplicationId::from_bytes([52; 16]), + node: NodeId::from_bytes([54; 16]), + fleet: Digest::from_bytes([55; 32]), + image: Digest::from_bytes([56; 32]), + signing_key: SigningKey::from_bytes(&[57; 32]), + owner: "canopy".into(), + token: "local-test-token".into(), + public_url: "http://127.0.0.1".into(), + peer_endpoint: "https://canopy.test".into(), + peer_ca_pem: None, + listen: "127.0.0.1:0".parse()?, + ssh: None, + data_dir: data.clone(), + store_prefix: StorePath::from("native-shutdown-test"), + native_limits: NativeLimits::default(), + local_disk_limit_bytes: 1 << 30, + max_active_repositories: 3, + }, + Arc::new(InMemory::new()), + None, + ) + .await?; + let native = server.native.clone(); + let foreground = native + .scope(NativeClass::Foreground) + .try_admit(NativeWork::Read)?; + let maintenance = native + .scope(NativeClass::Maintenance) + .try_admit(NativeWork::Pack)?; + let node = Arc::clone(&server.node); + let directory = server.directory.clone(); + let session = server.advertisement.lock().await.advertisement().session(); + let mut shutdown = tokio::spawn(server.shutdown()); + tokio::time::timeout(Duration::from_secs(5), async { + loop { + match native + .scope(NativeClass::Foreground) + .try_admit(NativeWork::Read) + { + Err(error) if crate::native_resources::is_exhausted(&error) => break, + Err(error) => return Err(error), + Ok(claim) => drop(claim), + } + tokio::task::yield_now().await; + } + Ok::<_, std::io::Error>(()) + }) + .await??; + // A finite shutdown timeout would hide premature authority withdrawal. + // The native owner must retain renewal past the original signed lease. + tokio::time::sleep(Duration::from_millis(u64::try_from(LEASE_MS)? + 2000)).await; + assert!(!shutdown.is_finished()); + assert!( + !node.is_shutting_down(), + "Cell shutdown preceded native drain" + ); + assert!(directory.is_live(session, unix_now_ms()?).await?); + assert!( + workspace::Workspace::open(&data) + .is_err_and(|error| error.kind() == std::io::ErrorKind::WouldBlock) + ); + drop(foreground); + // Canceling the wait must not stop the supervisor or its maintenance owner. + assert!( + tokio::time::timeout(Duration::from_millis(50), &mut shutdown) + .await + .is_err() + ); + assert!(!node.is_shutting_down()); + assert!(directory.is_live(session, unix_now_ms()?).await?); + drop(maintenance); + tokio::time::timeout(Duration::from_secs(10), shutdown).await???; + assert!(node.is_shutting_down()); + assert!(!directory.is_live(session, unix_now_ms()?).await?); + let _restored = workspace::Workspace::open(&data)?; + Ok(()) +} diff --git a/crates/canopy-server/src/server/mod.rs b/crates/canopy-server/src/server/mod.rs index 429cb75..0fae530 100644 --- a/crates/canopy-server/src/server/mod.rs +++ b/crates/canopy-server/src/server/mod.rs @@ -108,6 +108,7 @@ pub struct ServerConfig { pub data_dir: PathBuf, pub store_prefix: StorePath, pub local_disk_limit_bytes: u64, + pub native_limits: crate::native_resources::NativeLimits, pub max_active_repositories: usize, } @@ -168,6 +169,7 @@ struct RunningServer { ssh_serving: Option>>, tasks: TaskTracker, local: Arc, + native: crate::native_resources::NativeResources, } pub(crate) struct RepositoryManager { @@ -183,6 +185,7 @@ pub(crate) struct RepositoryManager { local: Arc, external_store: Arc, disk_budget: DiskBudget, + native: crate::native_resources::NativeResources, pub(crate) owner: String, pub(crate) public_url: String, pub(crate) ready: Arc bool + Send + Sync>, @@ -426,6 +429,7 @@ impl RunningServer { return Err(ServerError::Http("Git access token is required")); } http::validate_public_url(&config.public_url).map_err(ServerError::Http)?; + let native = crate::native_resources::NativeResources::new(config.native_limits)?; let data_dir = config.data_dir.clone(); let local = Arc::new( tokio::task::spawn_blocking(move || workspace::Workspace::open(&data_dir)).await??, @@ -601,6 +605,7 @@ impl RunningServer { local: Arc::clone(&local), external_store, disk_budget, + native: native.clone(), owner: config.owner, public_url: config.public_url, ready, @@ -624,6 +629,11 @@ impl RunningServer { let (api, peer, manager) = match startup { Ok(api) => api, Err(error) => { + maintenance_stop.cancel(); + native.close(); + tasks.close(); + tasks.wait().await; + native.drain().await; match node.shutdown().await { Ok(()) => local.confirm_drained(), Err(cleanup) => tracing::error!(error = %cleanup, "startup drain failed"), @@ -682,6 +692,7 @@ impl RunningServer { serving, tasks, local, + native, }) } } diff --git a/crates/canopy-server/src/server/residency/mod.rs b/crates/canopy-server/src/server/residency/mod.rs index 4d064d2..7f57850 100644 --- a/crates/canopy-server/src/server/residency/mod.rs +++ b/crates/canopy-server/src/server/residency/mod.rs @@ -532,6 +532,7 @@ impl RepositoryManager { self.local.path().to_path_buf(), Arc::clone(&self.external_store), self.disk_budget.clone(), + self.native.clone(), ) .with_signer_directory(Arc::clone(&self.directory)), ); diff --git a/crates/canopy-server/src/server/workspace/tests.rs b/crates/canopy-server/src/server/workspace/tests.rs index dee1c23..6d23bc6 100644 --- a/crates/canopy-server/src/server/workspace/tests.rs +++ b/crates/canopy-server/src/server/workspace/tests.rs @@ -120,6 +120,8 @@ async fn orphan_git_descendant_prevents_reclamation_until_it_exits() -> Result { cellule_ltx::DiskBudget::new(1 << 20), "refs/heads/main", crate::ObjectFormat::Sha1, + crate::native_resources::NativeResources::default() + .scope(crate::native_resources::NativeClass::Foreground), ) .await?; let data = workspace.path().join("directory.sqlite"); diff --git a/crates/canopy-server/tests/directory_cell/compatibility.rs b/crates/canopy-server/tests/directory_cell/compatibility.rs index 7ed6ccb..41e1816 100644 --- a/crates/canopy-server/tests/directory_cell/compatibility.rs +++ b/crates/canopy-server/tests/directory_cell/compatibility.rs @@ -3,9 +3,10 @@ use cellule_runtime::Digest; // Captured read-only from the unchanged RustFS deployment. The canonical // descriptor's SHA-256 is 8b85c842995e4a2b0bcc6d1f3d361ccdbd2c33a1c52592241b50ad8cfe9aeb88. -// This is contract compatibility, not proof that old Cells have been restored. +// The full historical application is refused by the Repository hard cutover. +// Directory compatibility is checked separately; no old binary restore is implied. #[test] -fn bounded_authentication_retains_the_exact_selected_predecessor() +fn bounded_authentication_retains_the_exact_selected_predecessor_directory() -> Result<(), Box> { let previous = include_bytes!("fixtures/c51-selected-release.json").trim_ascii_end(); assert_eq!( @@ -17,8 +18,14 @@ fn bounded_authentication_retains_the_exact_selected_predecessor() "directory-compatibility-test", ))?; let registry = application.registry(); - // Directory compatibility remains independent of the repository pack - // schema. Whole-release rolling compatibility is fenced separately below. + assert!(matches!( + registry.verify_rolling_from(previous), + Err(cellule_runtime::Error::Registry( + "rolling release does not retain predecessor module code" + )) + )); + let scoped = retained_directory::predecessor(®istry, previous)?; + registry.verify_rolling_from(&scoped)?; let predecessor: serde_json::Value = serde_json::from_slice(previous)?; let old_directory = predecessor["modules"] .as_array() diff --git a/crates/canopy-server/tests/directory_cell/main.rs b/crates/canopy-server/tests/directory_cell/main.rs index 2b04f8c..009e7ad 100644 --- a/crates/canopy-server/tests/directory_cell/main.rs +++ b/crates/canopy-server/tests/directory_cell/main.rs @@ -4,6 +4,8 @@ mod compatibility; mod expiry; #[path = "../support/objects.rs"] mod objects; +#[path = "../support/retained_directory.rs"] +mod retained_directory; mod ssh_keys; use std::{ diff --git a/crates/canopy-server/tests/git_http.rs b/crates/canopy-server/tests/git_http.rs index 8ebb594..d5ceaf9 100644 --- a/crates/canopy-server/tests/git_http.rs +++ b/crates/canopy-server/tests/git_http.rs @@ -9,6 +9,8 @@ async fn git_backend_advertises_smart_fetch_and_authenticated_push() cellule_ltx::DiskBudget::new(1 << 20), "refs/heads/main", canopy_server::ObjectFormat::Sha1, + canopy_server::native_resources::NativeResources::default() + .scope(canopy_server::native_resources::NativeClass::Foreground), ) .await?; for (service, content_type) in [ @@ -53,3 +55,85 @@ async fn git_backend_advertises_smart_fetch_and_authenticated_push() } Ok(()) } + +#[tokio::test] +async fn backend_instances_share_native_capacity_and_recover_after_release() +-> Result<(), Box> { + use canopy_server::native_resources::{NativeClass, NativeResources, NativeUsage, NativeWork}; + let root = tempfile::TempDir::new()?; + let disk = cellule_ltx::DiskBudget::new(2 << 20); + let native = NativeResources::default(); + let scope = native.scope(NativeClass::Foreground); + let mut backends = Vec::new(); + for _ in 0..2 { + backends.push( + GitHttpBackend::initialize( + root.path().into(), + disk.clone(), + "refs/heads/main", + canopy_server::ObjectFormat::Sha256, + scope.clone(), + ) + .await?, + ); + } + let mut active = Vec::new(); + while let Ok(permit) = scope.try_admit(NativeWork::Read) { + active.push(permit); + } + let held = native.usage()?; + for backend in &backends { + let input = canopy_server::git_input::GitInput::receive( + axum::body::Body::empty(), + root.path(), + &disk, + Some(1 << 20), + None, + ) + .await?; + let error = backend + .run(GitHttpRequest { + method: "GET".into(), + path_info: "/repo.git/info/refs".into(), + query: "service=git-upload-pack".into(), + content_type: None, + gzip: false, + protocol_v2: false, + body: input, + authenticated: true, + }) + .await + .unwrap_err(); + assert!( + matches!(error, canopy_server::git_http::GitHttpError::Io(error) if error.kind() == std::io::ErrorKind::WouldBlock) + ); + assert_eq!(native.usage()?, held); + } + drop(active); + assert_eq!(native.usage()?, NativeUsage::default()); + for backend in &backends { + let input = canopy_server::git_input::GitInput::receive( + axum::body::Body::empty(), + root.path(), + &disk, + Some(1 << 20), + None, + ) + .await?; + let response = backend + .run(GitHttpRequest { + method: "GET".into(), + path_info: "/repo.git/info/refs".into(), + query: "service=git-upload-pack".into(), + content_type: None, + gzip: false, + protocol_v2: false, + body: input, + authenticated: true, + }) + .await?; + assert_eq!(response.status, 200); + } + assert_eq!(native.usage()?, NativeUsage::default()); + Ok(()) +} diff --git a/crates/canopy-server/tests/multi_server/main.rs b/crates/canopy-server/tests/multi_server/main.rs index ba39708..5f1be58 100644 --- a/crates/canopy-server/tests/multi_server/main.rs +++ b/crates/canopy-server/tests/multi_server/main.rs @@ -14,6 +14,8 @@ mod paused_blobs; mod pulls; mod push_options; mod rebase; +#[path = "../support/retained_directory.rs"] +mod retained_directory; mod sha256; mod size; mod ssh; @@ -590,6 +592,7 @@ fn config(address: std::net::SocketAddr, data_dir: std::path::PathBuf) -> Server ssh: None, data_dir, store_prefix: StorePath::from("single-server-test"), + native_limits: canopy_server::native_resources::NativeLimits::default(), local_disk_limit_bytes: 1 << 30, max_active_repositories: 3, } diff --git a/crates/canopy-server/tests/multi_server/residency/faults/git_discovery.rs b/crates/canopy-server/tests/multi_server/residency/faults/git_discovery.rs index 99e4b7f..e603618 100644 --- a/crates/canopy-server/tests/multi_server/residency/faults/git_discovery.rs +++ b/crates/canopy-server/tests/multi_server/residency/faults/git_discovery.rs @@ -32,10 +32,10 @@ async fn ref_discovery_does_not_wait_for_a_full_history_restore() -> Result { assert!(blob.replace(meta.location).is_none()); } } - // Evict the gateway and its full Git cache before the native clone starts. - for name in ["fourth", "fifth", "sixth"] { - create(&fixture.client, fixture.address, name).await?; - } + // Publication and maintenance can temporarily exclude the original from + // eviction. Establish the actual cold precondition with the same bounded + // helper as the other cold-restore tests before starting discovery. + fixture.make_original_cold().await?; assert!(!fixture.repository_dir.exists()); *fixture.store.paused_read.lock().unwrap() = Some(blob.ok_or("external Git body missing")?); let destination = fixture.workspace.path().join("cold-clone"); diff --git a/crates/canopy-server/tests/multi_server/retained_catalog.rs b/crates/canopy-server/tests/multi_server/retained_catalog.rs index 7bab047..88ab6ae 100644 --- a/crates/canopy-server/tests/multi_server/retained_catalog.rs +++ b/crates/canopy-server/tests/multi_server/retained_catalog.rs @@ -1,6 +1,7 @@ //! Real startup seam for an immutable predecessor Directory catalog entry. //! This owned in-memory fixture is not an upgrade of the live RustFS corpus -//! or proof that an old executable produced the stored bytes. +//! or proof that an old executable produced the stored bytes. Only the exact +//! historical Directory contract is selected; other modules remain current. use super::*; use bytes::Bytes; @@ -79,8 +80,14 @@ async fn retained_fixture() -> Result { blake3::hash(previous).to_hex().as_str(), "e31bf1a951e2fa19d91e9f964b2ddeade1a81b05a20ad628362819a1487c16b1" ); - // This explicit fixture activation carries only a Directory Cell. The - // repository pack schema separately rejects whole-release rolling upgrade. + assert!(matches!( + registry.verify_rolling_from(previous), + Err(cellule_runtime::Error::Registry( + "rolling release does not retain predecessor module code" + )) + )); + let scoped = retained_directory::predecessor(®istry, previous)?; + registry.verify_rolling_from(&scoped)?; let descriptor: serde_json::Value = serde_json::from_slice(previous)?; let module = descriptor["modules"] .as_array() @@ -93,14 +100,13 @@ async fn retained_fixture() -> Result { .try_into() .map_err(|_| "invalid predecessor code")?, ); - assert!(registry.supports_cell(directory::DIRECTORY, CatalogRole::Sql, old_code, 1)); let releases = ReleaseStore::new(layout.clone(), identity)?; let image = format!("sha256:{}", hex::encode(configuration.image.as_bytes())); let old_operation = RequestId::from_bytes(uuid::Uuid::new_v4().into_bytes()); let prepared = releases .prepare( - previous, - Digest::from_bytes(*blake3::hash(previous).as_bytes()), + &scoped, + Digest::from_bytes(*blake3::hash(&scoped).as_bytes()), 0, &image, old_operation, diff --git a/crates/canopy-server/tests/multi_server/ssh/mod.rs b/crates/canopy-server/tests/multi_server/ssh/mod.rs index d1a8bff..d76dc25 100644 --- a/crates/canopy-server/tests/multi_server/ssh/mod.rs +++ b/crates/canopy-server/tests/multi_server/ssh/mod.rs @@ -936,12 +936,18 @@ async fn openssh_authenticates_registered_rsa_and_ecdsa_keys() -> Result { ssh_key::private::Ed25519Keypair::from_seed(&[11; 32]).into(), "test", )?; - let address = available_address().await?; + // Let the server own the ephemeral bind; probing and releasing a port + // before startup races other listeners in the parallel integration suite. let server = CanopyServer::start( - server_config(address, workspace.path().join("server"), &host)?, + server_config( + "127.0.0.1:0".parse()?, + workspace.path().join("server"), + &host, + )?, Arc::new(InMemory::new()), ) .await?; + let address = server.local_addr(); create_repository(address, "algorithms").await?; let ssh_address = server.ssh_addr().ok_or("SSH listener missing")?; let known = workspace.path().join("known_hosts"); diff --git a/crates/canopy-server/tests/owner_restart.rs b/crates/canopy-server/tests/owner_restart.rs index 2c8c4c9..bb4866c 100644 --- a/crates/canopy-server/tests/owner_restart.rs +++ b/crates/canopy-server/tests/owner_restart.rs @@ -114,6 +114,7 @@ async fn a_second_node_clones_from_the_published_root_after_local_disk_loss() first_disk.path().to_path_buf(), Arc::clone(&object_store), DiskBudget::new(1 << 30), + canopy_server::native_resources::NativeResources::default(), )); let (address, stop, server) = serve(first_gateway).await?; let first_url = format!("http://{address}/canopy/example.git"); @@ -325,6 +326,7 @@ async fn a_second_node_clones_from_the_published_root_after_local_disk_loss() second_disk.path().to_path_buf(), object_store, DiskBudget::new(1 << 30), + canopy_server::native_resources::NativeResources::default(), )); let (address, stop, server) = serve(second_gateway).await?; let second_url = format!("http://{address}/canopy/example.git"); diff --git a/crates/canopy-server/tests/smart_http/encoded_input.rs b/crates/canopy-server/tests/smart_http/encoded_input.rs index b3e2adf..5d486de 100644 --- a/crates/canopy-server/tests/smart_http/encoded_input.rs +++ b/crates/canopy-server/tests/smart_http/encoded_input.rs @@ -135,8 +135,27 @@ pub async fn delete_with_admission_retry( .output .is_some_and(|state| state.oid.is_none()) ); + // Admit only the encoded spool: any repeated gzip expansion must fail. + // Completed replay must resolve the original response before decoding. + let occupied = budget.try_reserve(budget.capacity() - warm - wire.len() as u64)?; let replay = request().send().await?.error_for_status()?.bytes().await?; assert_eq!(replay, response); + assert_eq!(budget.used(), occupied.bytes() + warm); + drop(occupied); + // Changing only gzip metadata preserves decoded commands, but is a + // different encoded request and cannot reuse the completed operation ID. + let mut changed = wire.clone(); + changed[4..8].copy_from_slice(&1u32.to_le_bytes()); + let conflict = client + .post(format!("{url}/git-receive-pack")) + .bearer_auth("local-test-token") + .header("Content-Type", "application/x-git-receive-pack-request") + .header("Content-Encoding", "gzip") + .header("Idempotency-Key", &id) + .body(changed) + .send() + .await?; + assert_eq!(conflict.status(), reqwest::StatusCode::CONFLICT); assert_eq!( repository.refs_page("", None).await?.output.generation, generation + 1 diff --git a/crates/canopy-server/tests/smart_http/main.rs b/crates/canopy-server/tests/smart_http/main.rs index 5b8aef7..d1c7c3b 100644 --- a/crates/canopy-server/tests/smart_http/main.rs +++ b/crates/canopy-server/tests/smart_http/main.rs @@ -34,6 +34,10 @@ mod ref_snapshots; #[tokio::test(flavor = "multi_thread")] async fn stock_git_push_and_clone_are_backed_by_one_repository_cell() -> Result<(), Box> { + let _ = tracing_subscriber::fmt() + .with_env_filter("canopy_server=warn") + .with_test_writer() + .try_init(); let application = Arc::new(CanopyApplication::compile(build_descriptor( include_bytes!("../../../../Cargo.lock"), "smart-http-test", @@ -103,7 +107,7 @@ async fn stock_git_push_and_clone_are_backed_by_one_repository_cell() )?; let repository = Arc::new(RepositoryCell::new( &application_handle, - target.clone(), + target, repository_id, canopy_server::ObjectFormat::Sha1, )?); @@ -118,6 +122,7 @@ async fn stock_git_push_and_clone_are_backed_by_one_repository_cell() scratch.path().to_path_buf(), Arc::clone(&blob_store), disk_budget.clone(), + canopy_server::native_resources::NativeResources::default(), )); let invalid_oid = [0; 32]; assert!(matches!( @@ -441,26 +446,18 @@ async fn stock_git_push_and_clone_are_backed_by_one_repository_cell() }) .await?; drop(teardown_gateway); - // The canonical Cell also owns the shared pack reader. Drop both - // owners to prove complete cache teardown before the cold clone. - drop(repository); assert_eq!( disk_budget.used(), 0, - "repository teardown releases retained object and snapshot charges" + "gateway teardown releases retained object and snapshot charges" ); - let repository = Arc::new(RepositoryCell::new( - &application_handle, - target, - repository_id, - canopy_server::ObjectFormat::Sha1, - )?); let gateway = Arc::new(GitGateway::new( Arc::clone(&repository), scratch.path().to_path_buf(), blob_store, DiskBudget::new(1 << 30), + canopy_server::native_resources::NativeResources::default(), )); let listener = TcpListener::bind("127.0.0.1:0").await?; let address = listener.local_addr()?; diff --git a/crates/canopy-server/tests/smart_http/publication.rs b/crates/canopy-server/tests/smart_http/publication.rs index a2891dd..d58bb8e 100644 --- a/crates/canopy-server/tests/smart_http/publication.rs +++ b/crates/canopy-server/tests/smart_http/publication.rs @@ -308,6 +308,7 @@ async fn verify_preparation( scratch.path().into(), store.clone(), DiskBudget::new(1 << 30), + canopy_server::native_resources::NativeResources::default(), )); let listener = TcpListener::bind("127.0.0.1:0").await?; let address = listener.local_addr()?; @@ -475,6 +476,7 @@ async fn gateway_request( scratch.path().into(), store, DiskBudget::new(1 << 30), + canopy_server::native_resources::NativeResources::default(), ); let response = gateway .handle( diff --git a/crates/canopy-server/tests/support/retained_directory.rs b/crates/canopy-server/tests/support/retained_directory.rs new file mode 100644 index 0000000..b83d588 --- /dev/null +++ b/crates/canopy-server/tests/support/retained_directory.rs @@ -0,0 +1,32 @@ +//! Owned fixture for a Directory-only predecessor release. Other module +//! identities stay current: the packed Repository module has a hard cutover. + +use cellule_runtime::Registry; + +type Result = std::result::Result>; + +pub fn predecessor(registry: &Registry, historical: &[u8]) -> Result> { + let historical: serde_json::Value = serde_json::from_slice(historical)?; + let old_directory = historical["modules"] + .as_array() + .ok_or("historical modules missing")? + .iter() + .find(|module| module["name"] == "directory") + .ok_or("historical Directory missing")?; + let mut selected: serde_json::Value = serde_json::from_slice(registry.release_bytes())?; + let modules = selected["modules"] + .as_array_mut() + .ok_or("current modules missing")?; + let directory = modules + .iter_mut() + .find(|module| module["name"] == "directory") + .ok_or("current Directory missing")?; + assert_ne!(directory["code"], old_directory["code"]); + *directory = old_directory.clone(); + // Preserve the exact historical Directory contract, including its code, + // migration and operation bounds. No old Repository code is selected. + selected["build"]["source_revision"] = "owned-directory-predecessor-fixture".into(); + let encoded = serde_json::to_vec(&selected)?; + registry.verify_rolling_from(&encoded)?; + Ok(encoded) +} diff --git a/deploy/README.md b/deploy/README.md index 1a28668..65ce1a4 100644 --- a/deploy/README.md +++ b/deploy/README.md @@ -62,6 +62,32 @@ public/peer URLs for this deployment. Set these container-specific values: { "listen": "0.0.0.0:8080", "data_dir": "/var/lib/canopy", + "native_limits": { + "total": { + "processes": 16, + "cpu_units": 16, + "memory_bytes": 2147483648, + "descriptors": 1024 + }, + "maintenance_reserved": { + "processes": 2, + "cpu_units": 5, + "memory_bytes": 805306368, + "descriptors": 128 + }, + "read": { + "processes": 1, + "cpu_units": 1, + "memory_bytes": 134217728, + "descriptors": 32 + }, + "pack": { + "processes": 1, + "cpu_units": 4, + "memory_bytes": 536870912, + "descriptors": 64 + } + }, "local_disk_limit_bytes": 1610612736, "max_active_repositories": 3 } diff --git a/docs/README.md b/docs/README.md index 9254f07..9d97092 100644 --- a/docs/README.md +++ b/docs/README.md @@ -14,6 +14,7 @@ Use this page to choose a document by task. Canopy's hosting core supports stock | Find the right Rust crate | [Rust workspace](workspace.md) | Crate ownership, dependencies and build commands | | Decide whether a release gate is closed | [Delivery plan](delivery-plan.md) | Required proof, current state and chronological implementation evidence | | Plan or evaluate capacity | [Repository density and latency](performance-plan.md) | Workloads, targets, measured results and limits of each result | +| Implement large-repository storage for a large team | [Packed storage design](large-repository-storage-design.md), [implementation plan](large-repository-implementation-plan.md) and [large-team amendment](large-team-scalability.md) | Hard-cutover design, required scalability changes and single-hot-repository release gates; capacity remains unqualified | | Inspect the original RustFS corpus upgrade | [Full-corpus activation and remote verification](performance/2026-10-01-original-corpus-activation.md) | Admission of 10,003 Cells, full Git/LFS verification, failed diagnostic load windows and open owner-loss gates | | Track three nodes behind a proxy and the latest dependency candidate | [Three-node proxy qualification](performance/2026-09-30-three-node-proxy.md) | Baseline/candidate pins, complete-corpus recovery, failing load windows and open gates | | Inspect the merged workspace and RustFS verification | [Workspace and RustFS verification](performance/2026-09-30-workspace-rustfs.md) | The `70bd25f` revision, conflict resolution and end-to-end gates | diff --git a/docs/archive/pr20-progress-through-8bb0ee7.md b/docs/archive/pr20-progress-through-8bb0ee7.md new file mode 100644 index 0000000..acc2099 --- /dev/null +++ b/docs/archive/pr20-progress-through-8bb0ee7.md @@ -0,0 +1,190 @@ +Large Git histories need canonical object lookup and publication without a heap-sized OID inventory or authoritative per-object placement rows in the Repository Cell. This PR adds immutable pack/catalog storage, verified preparation, owner-fenced atomic publication, bounded directory maintenance and shared final-command dispatch. Long imports can now retain their creating namespace through staging and acquire a catalog generation floor after physical verification. + +## Resulting behavior + +- Direct-ref policies now prepare in at most 128 updates/256 KiB per page. Guards reuse the existing plan, lease, catalog certificate and immutable ref transition; rare configuration changes advance an epoch while CI reports invalidate indexed exact required-run dependencies. Private guarded snapshot signing rechecks original intent, conditional refs and fresh readiness. Registration/cleanup are transactional and bounded; live invalid tombstones prevent old pages from recreating readiness. The final atomic root/native-outcome command and production cutover remain required. + +- Fresh empty repositories can now install authenticated catalog/ref roots through a private empty preparation and one bounded owner-fenced transaction. Initialization reuses the catalog certificate, GenerationFact and existing lease/pin model, refuses prior ref history and records one immutable exact outcome. Original mutation and logical replay survive actual owner restore; pending old attempts cannot write. No SQL-ref conversion or raw-root signing adapter is added. Production repository creation and full root publication still require the coordinated hard cutover. + +- Query-derived ref preparation now reuses the catalog's immutable generation and retention floor. GenerationFact, leases/frontier queries and catalog certificates carry the selected snapshot; a private factory loads it through the prepared catalog's own store and refuses a missing snapshot. Compaction carries the exact ref root forward, while the inline SQL ref publisher refuses new writes against a selected root. The new v4 attestation layout encodes shared context once and retains the 1 KiB certificate cap. Production initialization wiring, atomic final guard/policy validation and short root/outcome publication remain required before producer/reader cutover. + +- Immutable ref state now reuses RangeIndex/NodeRef, authenticated artifacts, RefExpectation and the canonical PushPlan digest. Variable-length shared keys preserve lexical ordering; encoded-byte splits support names up to 65,535 bytes. Tombstones retain exact delete/recreate versions, and authenticated live counts skip dead subtrees without missing later namespace conflicts. Initial inventories and existing-base batches use the shared streaming builder. Sorted rewrites load affected paths and reuse authenticated untouched subtrees; one completed block plus the active block per level balances final tails by bytes and fanout. Exact expectations and namespace checks precede node writes, while the canonical digest preserves original caller intent order. A snapshot root keeps long default-branch/root fences outside the command envelope; wire-request/native-result typed caps remain 64/128 KiB. These are conditional data-plane primitives. Authoritative atomic catalog/ref/response root publication remains required; the serving path still uses SQL refs. + +- Native completion now survives local cache loss through a bounded authenticated result root in the existing input checkpoint. It reuses PushCompletionRequest, PushPlan, GitHttpResponse, SignedPushAnnotation, ArtifactDescriptor and GitInput. The exact response/options, versioned ref intent and scoped native signature witness are retained before Bind; attaching the result freezes the captured native inventory. Plans stream in at most 64 KiB frames with 32 canonical updates per chunk, rather than allocating a second full encoded plan. Recovery requires exact committed MAC custody, authenticated bodies and fresh logical scope. Signed recovery additionally requires the exact tenant/application Directory, an enabled account and enabled write-scoped fingerprint. Whole-root takeover adoption copies no large bodies and remains independent of source-pin expiry after destination registration. Original request/result bodies and metadata share InputBody/InputRoot in the create-only git-inputs artifact family, with no old-path/domain adapter. The final short ref-plan command interface and production takeover/producer/reader orchestration remain required. + +- The packed preparation API now retains the original encoded push before native receive. It reuses GitInput, GitHttpRequest, BeginRequest, authenticated create-only artifacts and NativeInputIndex. Request and artifact digests share one 64 KiB scan; upload independently checks the cached digest and queued cancellation retains spool admission. A request-only checkpoint precedes native work, then captured descriptors append through the same index under exact predecessor CAS. The v3 checkpoint domain has no compatibility adapter; the existing lease row carries a predecessor digest and at most 256 unbound revisions. SQL rejects arbitrary replacement, partial/removal/revision changes, append after Bind and simultaneous Bind/append. The service reuses its completed slot only after known success with the exact predecessor, preserving uncertain commands and old observer receipts. Staging/bound reopening authenticates root/body, recomputes scoped request identity and freshly rechecks custody. Restored-owner adoption preserves all exact roots without copying, independently of source-pin expiry after destination commit. Raw signed text remains intent. The final short ref-plan command interface, production selection/orchestration, complete collection/isolated restore and capacity qualification remain open. + +- The production receive-pack gateway now retains an owned authenticated preflight using the existing GitInput spool, BeginRequest and command parser. The v3 digest binds Cell scope, repository format, actor, operation, HTTP metadata and exact encoded bytes before completed replay. Native input and parsed intent consume that same identity. Repository-format checks include zero, signed and shallow OIDs; signature verification remains native. Module source-digest coverage now includes preflight, push handling, parsing and spooling. Completed gzip replay preserves the original response before expansion, with exact encoded-identity conflict checks. This shared handoff still requires production wiring to packed capture/publication and the short final ref-plan command interface. + +- Final push/compaction publication now belongs to the bound lifecycle. StagingTicket::publish synchronously reserves the existing private ready value as a Held job in PublicationCoordinator and stores only a small observer ticket. Shared session scope/clock/fence/ceiling matching, existing class/account/byte admission and exact command identity are preserved. Existing workers/results, due renewal and accepted checkpoints drain before activation, with fresh custody at the latest known receipt. Known outcomes preserve their original receipt without querying the retired preparation record. Pre-activation fencing discards only proven unexecuted work; queued submission and proven-absent recovery check local custody, while known commits resolve first. Closed/canceled/shared-coordinator recovery retains ownership. Native receive now uses this worker-to-publication handoff before cold Git clone/fsck; maintenance uses the reserved class. + +- Bound preparation now remains service-owned after Bind, and ReadyStaging::claim_bound admits takeover into the same staging job. Automatic RenewPreparation, bound workers/results and adopted checkpoint registration reuse the existing actor/operation admission, worker semaphores, WorkSlots, exact command slot and independent pins. open_base shares the supervisor session and fence. A separate immutable local residence ceiling starts before Bind preparation or Claim admission (60s default, at most MAX_LEASE_MS); raw SQL deadlines schedule renewal so clipping cannot cause a tight renewal loop. Canceled observers retain work/results and exact commands. Known Bind/Claim/Renew/checkpoint receipts survive failed fresh custody; close renews until accepted work/results drain, then fences the shared session before releasing admission. SQL pin expiry and remote artifacts remain independently retained. Production integration and durable reconstruction remain required. + +- Bound Claim/Renew now share exact service ownership through ReadyPreparation::claim and PreparationSession::ready_renew. Private bounded requests retain commands 12/13 with an 8 KiB foreground reservation in the existing coordinator. Cancellation, closed recovery, rejected ready-value reuse and absent/lost/panicked replies preserve exact identity and original receipts. PreparationCommandOutcome separates the known commit from a freshly queried session; failed current custody never erases the receipt or revives a fenced renewal session. Renewal keeps the original floor/namespace, while Claim creates a new attempt/current floor and retains the old independent pin. The private request is boxed to preserve the shared ready enum size. Actual restored-owner Claim/recovery/renewal and real retained-input publication use these APIs. Automatic bound renewal, worker/result/checkpoint ownership and residence now reuse the staging supervisor. Durable takeover reconstruction remains required. + +- Bound adopted-input checkpoint registration now reuses PublicationCoordinator through the private PreparationSession::ready_inputs factory. Foreground account/operation quotas, fair queues, cancellation ownership and exact SDK recovery cover the accepted command. Checkpoints reserve 8 KiB while pushes retain 8 MiB; each existing job carries its factory-derived byte reservation. RegisteredNativeInputs preserves the original committed receipt separately from fresh checkpoint/bound-session custody. Revocation, expiry or Claim fences the shared session without erasing the known commit. Refused admission returns the original ready value; checkpoint results cannot become Git push responses. Automatic bound renewal and Claim/worker/checkpoint ownership now reuse the staging supervisor. Durable reconstruction remains required. + +- Completed native capture now explicitly unlocks its exclusive fence only when the last owned file pin drops. Closing a CLOEXEC descriptor alone can leave an unrelated pre-exec child holding its lock and spuriously refuse the next native admission. The fix preserves queued-upload ownership, active-worker exclusion and descendant drain. A deterministic regression reproduces the defect before the fix on macOS and Linux and verifies release while an unrelated child remains paused. + +- Adopted native inputs now publish through authenticated physical custody. CatalogPreparation::begin_retained_pack checks exact full-descriptor membership in the current successor checkpoint and freshly verifies its bound attempt/floor. A private proof gates closure verification while the ordinary raw namespace guard remains strict. Original native/metadata incarnations are preserved; new source-index nodes use the successor namespace. Catalog certificate v4 binds the joint generation and immutable checkpoint digest through reconciliation, issuer checks and final transactional pin/MAC/expiry/authority checks. The old v2/v3 certificate domains are rejected. Completed exact/logical recovery preserves original results. The shared bounded input-index cache reuses existing data structures; no per-object SQL inventory is introduced. Production producer selection, complete collection/isolated restore and large-team capacity qualification remain open. + +- Durable native input checkpoints now reuse NativePackDescriptor, the shared RangeIndex/NodeRef and independent lease rows, with a bounded purpose-separated envelope and immutable digest. Command 29 and query 30 recheck current access, owner/attempt, expiry, scope and authentication. Staging and bound preparation recovery adopt the exact retained root without copying index nodes or uploading native pairs; final registration rechecks source custody and root equality atomically. ReadyStaging::claim retains exact SDK dispatch through cancellation and uncertain replies. StagingTicket::register_inputs now synchronously admits one checkpoint and its mutation identity into that same dispatcher; accepted registration precedes Bind or graceful stop, and a due renewal precedes registration. Dropped observers recover through pending_inputs, while unknown commands retain their exact evidence. One additional 4 KiB reservation bounds the admitted checkpoint slot. Known registration stores the original receipt before a fresh live query, preserving committed recovery while refusing revoked custody. Checkpoints establish descriptor custody; independent physical/canonical/closure verification remains required. Production staging wiring, durable reconstruction and collection integration remain open. + +- Native staged receives now have an explicit `run_native_receive` API that retains pack/index pairs and shares the bounded CGI collector. `stage_native_packs` captures at most 32 request-private inputs under an admitted StagingContext, reuses native descriptors and the pinned-file uploader, and holds cache/disk ownership plus an exclusive native fence through queued work and cancellation. Scope, size, loose-input, checksum and active-worker failures reject before descriptors escape. A SHA-1/SHA-256 composition test receives through actual Git, stages authenticated pairs, drops receive/source caches, independently verifies, atomically publishes, and clones/fscks from a rebuilt cache selected through the committed catalog. The fixture supplies authority and orchestrates the APIs; production HTTP/SSH conversion remains open. + +- Refused and empty pushes now issue a bounded, purpose-separated outcome proof directly from PreparationSession, without a catalog loader, upload, scratch or native worker. Session and catalog readers share the existing authoritative lease/deadline/renewal fence. Outcome proofs reuse command 19, response/options/signed-witness tables and foreground dispatch. New outcomes check current write access, owner/attempt/pin/expiry and immutable floor; completed recovery preserves original results. A moving catalog does not invalidate a ref-free outcome. Exact uncertain commands retain only the session rather than prepared catalog artifacts. Production producer conversion remains open. + +- Native shutdown now permanently closes the shared admission pool across all cloned scopes, joins accepted/tracked work, and waits for healthy zero foreground and maintenance claims before Cell shutdown, workspace release, heartbeat stop and advertisement withdrawal. Startup failure after enrollment follows the same ordering. Canceled drain observers cannot reopen admission or release claims; poison, underflow and quarantined native owners keep drain unproven. Closed admission maps to HTTP 503. +- Every shared native Git spawn now requires a private four-dimension claim for process slots, CPU admission units, memory and descriptors. One pool is configured per node and shared through production gateways/caches/readers; preparation APIs require its explicit scope. Foreground and maintenance shares are disjoint. Claims follow the native drain guard through cancellation/quarantine; repack workers release after drain before validation and returned caches retain their original scope. Typed exhaustion becomes HTTP 503. Required `native_limits` is a configuration hard cutover; checked-in config producers/examples and benchmark metadata now supply it. These claims are estimates; OS containment, account/fair preparation scheduling and full-history qualification remain open. +- Native transport, verification, decoded-object/history, cache maintenance, blob extraction, candidates and ref-list workers now reuse one process ownership guard. Unix completion waits for inherited-descriptor EOF before leader reaping. Cancellation signals the unreaped private group and transfers the child, descriptor and generic owner to a bounded independent reaper; failed, saturated or shutdown drain quarantines owners. The fence proves drain only for participating descendants that preserve it. Hard OS CPU/RSS/file/process containment, fair account/preparation admission, profile qualification and equivalent non-Unix descendant drain remain open. +- Metadata, directory and closure construction reuse the existing SQLite tables and DiskBudget with admitted geometric growth. Empty default spools initially charge 192 KiB rather than 768 MiB. Fully rolled-back SQLite capacity failures can double the page cap only after disk admission; commit precedes cursor/digest advancement, and verified edge files rehash on every replay. Closure lookup creation and pending-degree initialization use indexed pages of at most 512. Immutable sealing shrinks to exact file bytes. Explicit ceilings, cancellation ownership and conservative cleanup remain enforced. +- Physical verification shares one admitted append-only dependency file per at-most-512-object metadata page, reducing dependency files from as many as 512 to one. Existing decoded witnesses retain exact private offset/length/digest ranges and rehash each range on SQL replay. Exclusive object-writer permits survive queued cancellation, storage failure poisons reuse, and full file credit remains held until the last producer/witness/worker drains. This is a per-page bound; global file/I/O admission and native profile qualification remain open. +- Ancestry reuses admitted SQLite growth and the existing visits/answers tables with a 192 KiB initial charge. A mutable walker borrow spans each traversal; memo answers bind to the exact catalog descriptor. Failure or cancellation permanently fences reuse, queued workers retain scratch credit through drain, and fresh lease checks cover identical tips and cached answers. Queue reset uses indexed transactions of at most 512 keys while preserving memo answers. +- File-backed SHA-1/SHA-256 native Git indexes, authenticated operation-scoped artifacts, canonical metadata segments, directory/source indexes and admitted catalog readers provide immutable lookup foundations. Physical pack verification and disk-backed canonical/dependency/closure verification compose through private factories; raw descriptors or caller-selected certification flags cannot authorize publication. +- A fresh schema removes legacy Git bodies and per-object placement. Independent attempt pins, immutable generation facts, monotonic creating namespaces, exact deferred operation/pin bindings and bounded indexed reaping preserve custody through Claim, Abort and owner restoration. SQL reaping does not authorize remote deletion. +- Staging reuses the same tokens, operation rows and lease rows with a NULL generation. Begin/Renew/Check/ClaimStaging retain inputs without holding intervening catalog facts. BindStaging selects the current floor once, preserves namespace and expiry, and needs no additional pin. A generated non-null binding value prevents SQLite's nullable composite-FK escape. Staging cannot open a preparation base or certify publication; the existing assembler accepts pre-bind physical witnesses after authoritative binding. +- StagingCoordinator owns Begin/Claim/Renew/RegisterStagedInputs/Bind commands and input producers independently of observers. Defaults bound 32 operations/eight per actor and 64 worker/result slots/eight per actor. Automatic renewal uses fresh authoritative queries; completed typed results remain charged until single handoff and are retrievable after a canceled observer. Seal blocks new workers and renews custody while slots drain before Bind. Failure cancels and joins producers and drops completed resources before credits. Unknown commands retain exact identities/evidence, sharing invocation/resolution with final publication; close/drain preserves uncertainty and original receipts. +- Trusted catalog/ref proofs bind exact ref-plan bytes, canonical membership and ancestry. Final commands recheck admitted owner fencing, current ACL/policies/checks, expected ref identities/versions and selected catalog CAS. Catalog, refs, exact native response, options and signed-push annotations commit atomically. Exact/logical recovery preserves original durable receipts. Reconciliation reuses physical inputs and admitted closure scratch against a queried current catalog. +- Streamed bounded output files share an indexed ingress root. StoredRun reuses physical file facts and logical RunCoverage; partial projections retain exact parent/prefix/suffix inventories. Directory-root v3/range-index v2 reject old layouts. Nodes remain within 64 KiB and point selection within 48 candidates; projections share physical cache admission. +- Certified compaction merges 2–32 ingress roots and supports bounded adjacent-level windows, retaining verified suffixes in existing physical files. Path-copy updates preserve unrelated runs, source/version identity, refs and old readers; reconciliation rejects changed or newly overlapping inputs. Geometric policy rotates ingress, levels and indexed cursors while bounding urgent bursts. +- One service-owned publication coordinator dispatches privately prepared pushes, compactions, bound input checkpoints and Claim/Renew commands through typed outcomes and exact uncertainty recovery. Defaults reserve four of 32 slots for maintenance, cap maintenance waits at two of eight total, and allow at most three foreground starts before eligible maintenance. Separate class/account quotas and byte credits retain ambiguous work; proof ownership drops before admission is released. Cancellation cleanup releases private workspace roots before reader/disk credits. + +## Release scope and dependencies + +Node configuration now requires `native_limits`; missing limits, unknown fields and negative capacities reject parsing, and invalid resource shares reject before workspace/provider work. The example and small evaluation profiles require measured hardware/OS headroom before production use. + +The fresh schema and registry are **not selected by production HTTP/SSH handlers yet**. Remaining work includes producer/reader conversion, production staging lifecycle wiring and durable reconstruction, full-history input/resource qualification and ancestry acceleration, continuous preparation/maintenance and CPU/I/O admission, publication progress against moving roots, native physical pack rewriting and accelerated reads, complete retained-root collection, isolated restore and deployment hard cutover. Outcome growth, provider durability/grouping and full-history Kubernetes/Linux/Chromium mixed-load campaigns remain unqualified. This PR does not claim capacity for 10,000+ engineers. + +Binding preserves borrowed input expiry; a five-minute staging renewal can leave a five-minute catalog floor. Remaining-floor capacity and lease/admission configuration must be qualified before production selection. Ordinary direct preparation retains its two-command minimum; bulk staging adds Bind, measured renewals and each selected input checkpoint registration. + +This branch builds on the merged pack-storage prerequisite from [Canopy #19](https://github.com/crabbuild/canopy/pull/19). Cellule manifest/lockfile entries pin main's `0f4ca0919b0dfe20a3dcd964d21da03135e42eed`, including the owner-fencing API from merged [Cellule #38](https://github.com/crabbuild/cellule/pull/38). + +## Main-branch conflict resolution + +Merged main at `64db462` and the newly merged pack prerequisite at `877dc33` through commits `5f98a3a` and `a286214`. The resolution uses main's Cellule pin, retains both documentation entries, combines listener/fence regressions with native admission and deferred cleanup, and preserves bounded pack indexes and weak gateway-owned readers. Retained Directory fixtures select the exact historical Directory contract while keeping other modules current; both the old Repository code and whole historical application are explicitly refused. The golden descriptor is unchanged. Main's benchmark source identity helper/test are retained. + +Validation: all 458 workspace library tests passed; final all-target workspace Clippy passed with warnings denied; all 96 Python harness tests passed; Directory/repository boundary checks, four listener checks, three retained-catalog startup scenarios, signed SHA-256 SSH/fresh-disk restore, both HTTP backend checks and the stock-Git push/clone fault workflow passed. Initial failures were one stale startup test signature during compilation and three retained-catalog fixtures demanding whole-release compatibility before startup. Both causes were corrected; the original runtime scenarios passed. Detailed timing and historical results are preserved in the implementation status. The final prerequisite merge leaves validated Rust production sources byte-for-byte unchanged. Policy-guard work was excluded from this merge validation and is validated separately below. + +CI at `a286214`: [37077055118](https://github.com/crabbuild/canopy/actions/runs/37077055118) completed successfully; [37077052177](https://github.com/crabbuild/canopy/actions/runs/37077052177) failed one multi-server eviction-precondition assertion. The unchanged focused test passed locally; `7ff1fe1` reuses the existing bounded cold-state helper and preserves all discovery deadlines/integrity assertions. Its focused check passed (3.63s). The failed CI invocation remains failed. New-head CI, hard cutover and full-history/large-team qualification remain open. + +## Validation + +- Durable request checkpoint increment (`5e48a3a`): all 424 unique workspace library tests passed (6 Git-format, 14 object-storage, 404 server; server 145.67s). Six new tests cover large plain/gzip request recovery in both formats and native append, actual restored-owner adoption after source-pin expiry, multipart corruption/partial-spool release and revoked custody, absent/lost/panicked append recovery with old observer receipts, queued hash/upload-open cancellation ownership, and exact SQL predecessor/unbound phase/256-revision guards. Artifact tests now cover Request/RequestRoot create-only replay and retired incarnation isolation. The real native receive/publication/cold-clone/fsck composition now checkpoints the request before native receive, appends actual captured pairs and reopens the original request after local cache loss. Synthetic signed/pack request text qualifies byte/intent retention, not native validity. All-target workspace Clippy with warnings denied passed (29.37s); rebuilt stock-Git smart-HTTP passed (44.88s). Formatting/diff checks, whitespace for 28 changed/new files and 62 local doc links passed. New-head CI is pending. + + Initial real composition reproduced Duplicate twice: the completed service checkpoint slot still refused native append; exact-predecessor replacement after known success fixes it. A new SQL regression reproduced simultaneous Bind/append acceptance because the trigger checked only OLD.generation; the updated trigger requires both phases to remain unbound. The initial restored-owner fixture used time zero with a wall-clock mutation identity and now uses the existing clock helper. The first broad run passed 403 server tests and failed one corruption fixture (180.17s): its body was below the actual 8 MiB part boundary, so its injected second key was never read. The fixture now derives size from PART_BYTES and asserts the boundary; the unchanged integrity/refusal assertions pass in the final run. The earlier broad run remains failed. No deadline or authority/integrity assertion was weakened. These results do not qualify full-history import, production hard cutover, remote durability, isolated restore or large-team capacity. Both CI runs at the previous `6b62045` head passed: [run 1](https://github.com/crabbuild/canopy/actions/runs/37023717381), [run 2](https://github.com/crabbuild/canopy/actions/runs/37023710550). + +- Owned preflight increment (`6b62045`): all 418 unique workspace library tests passed on final source (6 Git-format, 14 object-storage, 398 server; server 113.20s), including five new preflight checks. Stock-Git smart HTTP passed (37.68s); signed/push-option selection passed four and explicitly ignored one isolated provider check (9.20s); SHA-256 selection passed five and explicitly ignored three isolated provider checks (18.38s), including signed SSH and fresh-disk restore. Workspace/all-target Clippy with warnings denied passed (29.93s); formatting, diff checks, whitespace for 11 changed/new files and 46 local documentation targets passed. Initial focused/library/HTTP runs also passed before the final source-digest and all-zero regression additions. No runtime failure, assertion weakening or deadline widening occurred in this increment. Fresh CI is pending for 6b62045. The selected production publication path still uses the legacy schema; durable wire-plan/response recovery, producer/reader conversion, fresh-schema selection, complete collection/isolated restore, containment, hot-root progress and full-history/large-team qualification remain required. + +- Final lifecycle increment (`b2a7949`): all 413 unique workspace library tests passed on the final source (6 Git-format, 14 object-storage, 393 server; server 109.79s). Thirteen new tests cover held admission and contention, original ready-value reuse, cancellation/resource ownership, activation/discard races, worker/result drain, serialized renewal/checkpoint recovery, original final receipts after revocation/expiry, queued expiry before initial execution, shared-coordinator recovery and maintenance publication. Native receive/cold clone/fsck passes for both formats through the new handoff. All-target workspace Clippy with warnings denied passed (26.15s); formatting/diff/20-file whitespace/59 local documentation targets passed. Initial new tests exposed fixture/permission-observation mistakes and virtual-clock offsets crossing unrelated SDK deadlines; the corrected fixtures distinguish internal original results from authenticated replay, isolate virtual time per scenario and use a tighter one-second local ceiling. Production bounds and SDK deadlines are unchanged. Both CI runs passed at b2a7949, including workspace integrations, Python harness and RustFS compatibility: [run 1](https://github.com/crabbuild/canopy/actions/runs/37013508378), [run 2](https://github.com/crabbuild/canopy/actions/runs/37013518780). Production selection, durable takeover, complete collection/isolated restore and full-history/large-team capacity remain unqualified. + +- Bound lifecycle ownership increment (`0c38d4a`): all 400 unique workspace library tests passed (6 Git-format, 14 object-storage, 380 server; server 128.60s). Eight new checks cover automatic bound renewal, canceled worker/result ownership and close drain, exact absent/lost/panicked Claim/Renew with closed recovery, original receipts with revoked/expired/superseded custody, stage/bound phase separation and residence fencing, shared native bases and inflight work, reused actor worker/result caps and resource-drop order, serialized adopted checkpoint recovery/root reuse, and actual owner restoration followed by workers/renewal. Native assembly now runs in a bound-owned worker before its existing attestation checks. All nine unchanged staging tests passed (1.36s) and all 24 focused service checks passed (10.61s). A final regression checks native base reads against a past ceiling while the shared fence remains unset; renewal/refresh also recheck the ceiling after a delayed query. The final changes passed all 24 focused checks (9.31s) and all-target workspace Clippy with warnings denied (31.84s). The initial compile exposed a deadline borrow conflict; the first new-test compile had three test-plumbing errors (Receipt access/Result alias), corrected without widening deadlines or assertions. No focused runtime test failed. Formatting, diff, whitespace for 13 changed/new files and 51 local documentation targets passed. Both CI runs passed at 0c38d4a, including workspace integrations, Python harness and RustFS compatibility. These checks do not qualify production cutover, SQL floor-capacity limits, complete collection/isolated restore or full-history/large-team capacity. + +- Bound Claim/Renew dispatch increment (`7353e55`): the full workspace library run passed all 392 unique tests (6 Git-format, 14 object-storage, 372 server; server 136.79s), including six new checks. They cover both formats, exact absent/lost/panicked dispatch, canceled observers, closed recovery, refused command reuse, known receipts with revoked/expired/superseded custody, current authority after absence, expired-source Claim, invalid/oversized factory input, permanent session fencing, unchanged renewal floors and retained old pins. The actual Cell-owner restoration fixture recovers a lost Claim reply, checks its new owner/namespace and original pin, and renews the resulting session. Real retained-pair publication now acquires its bound session through supervised Claim. The initial focused run passed five checks (1.85s). Initial Clippy rejected the enlarged shared ready enum and two nested-if lints; the private request was boxed and guards collapsed without suppressions or changed authority/deadline checks. Final all-target Clippy with warnings denied passed (41.67s); all 17 focused bound/admission/ownership/recovery/retained-input checks passed on the final representation (8.89s). Formatting, diff, whitespace for 13 changed/new files and 49 local documentation targets passed. Both CI runs passed at 7353e55, including workspace integrations, Python harness and RustFS compatibility: [run 1](https://github.com/crabbuild/canopy/actions/runs/37004520762), [run 2](https://github.com/crabbuild/canopy/actions/runs/37004516197). These fixtures do not qualify automatic lifecycle scheduling, production cutover, complete collection/isolated restore or full-history/large-team capacity. + +- Bound checkpoint dispatch increment (`df7d79f`): all 386 unique workspace library tests passed (6 Git-format, 14 object-storage, 366 server; server 119.49s). Six new checks cover both OID formats, absent/lost/panicked exact dispatch, canceled observers, closed recovery, refused ready-value reuse, original committed recovery with failed fresh custody, denied absent recovery without attaching inventory, real retained-pair verification/publication after source-pin expiry and mixed checkpoint/push actor/byte admission. The existing 300-record adoption test now uses supervised registration while preserving the exact original root. The corrected focused run passed six checks (9.32s); its first attempt failed to compile due to test plumbing, corrected without changing production guards or deadlines. All-target workspace Clippy with warnings denied passed (39.96s), plus formatting, diff checks, whitespace for 13 changed/new files and 44 local documentation targets. Both CI runs passed at df7d79f, including workspace integrations, Python harness and RustFS compatibility: [run 1](https://github.com/crabbuild/canopy/actions/runs/37002670400), [run 2](https://github.com/crabbuild/canopy/actions/runs/37002664555). These checks do not qualify production cutover, complete collection/isolated restore or full-history/large-team capacity. + +- Capture fence release increment (`f59617d`): all 380 unique workspace library tests passed locally (6 Git-format, 14 object-storage, 360 server; server 120.67s) and under an unprivileged Linux Docker user (same counts; server 147.38s). Linux used Rust 1.97.1/Git 2.39.5, while local macOS used Rust 1.98.0. The new fork regression, original capture rejection fixture, queued-upload ownership, actual receive/publication/cold clone and both permissions checks passed. The initial macOS regression failed because /bin/true was absent; correcting executable lookup exposed the intended WouldBlock failure before the production fix. The pre-fix Linux run passed 357 server tests and failed three (225.22s): the new regression and two permissions fixtures incorrectly run as root. Its original capture rejection fixture passed. This proves the inherited exclusive-lock defect; the precise phase of the earlier CI failure remains unproven. Diagnostic instrumentation was removed before final Linux validation and all-target Clippy (warnings denied, 29.98s). Formatting, diff, whitespace and local documentation targets passed; the disposable container was cleaned up. Both CI runs passed at f59617d, including workspace integrations, Python harness and RustFS compatibility: [run 1](https://github.com/crabbuild/canopy/actions/runs/36999237560), [run 2](https://github.com/crabbuild/canopy/actions/runs/36999233339). These checks do not qualify full-history import, production cutover, collection/isolated restore or large-team capacity. + +- Retained physical custody increment (`63299e3`): all 379 unique workspace library tests passed (6 Git-format, 14 object-storage, 359 server; server 152.75s). Six new custody tests cover moving nonempty bases and cold Git clone in both formats, absent successor checkpoint, raw namespace rejection, absent inventory membership, exact index incarnation, signed digest substitution/old-domain refusal, and issuer/final-command revocation, expiry and Claim. The actual owner-restoration test now publishes after old-pin expiry and verifies independent pin/original checkpoint receipt recovery after completion. The initial run passed 378 tests and failed its post-publication active-query assertion: completion correctly retires the active operation. The corrected assertion checks active custody before publication and retained pin/exact receipt afterward; production guards and deadlines are unchanged. All-target workspace Clippy with warnings denied passed (1m 09s); formatting, diff, whitespace for 16 changed/new files and local documentation targets passed. Both CI runs passed at 63299e3, including workspace integrations, Python harness and RustFS compatibility: [run 1](https://github.com/crabbuild/canopy/actions/runs/36997309621), [run 2](https://github.com/crabbuild/canopy/actions/runs/36997304829). This does not qualify production cutover or full-history/large-team capacity. + +- Staging checkpoint supervision increment (`c200d63`): all 373 unique workspace library tests passed (6 Git-format, 14 object-storage, 353 server; server 145.02s). Five new service tests cover canceled observers, exact absent/lost/panicked dispatch in both formats, registration-before-Bind ordering, one-slot and foreign/duplicate/closed refusal without execution, committed recovery with revoked current access, and expiry after absence without attaching an inventory. The real receive/publication/cold-clone fixture uses supervised registration in both formats. The initial run passed 372 unique tests and failed the new expiry fixture (server 352 pass/one fail, 172.62s): changing only pin expiry violated the deferred operation/pin foreign key. The fixture now changes both expiries in one transaction, with production deadlines, guards and assertions unchanged. Final all-target workspace Clippy with warnings denied passed (1m 22s); formatting, diff, whitespace for 11 changed files and 18 local documentation targets passed. Both CI runs at c200d63 failed the existing native-capture rejection fixture with Input(Io(Kind(WouldBlock))) after 352 server tests passed and one failed (101.72s and 99.44s): [run 1](https://github.com/crabbuild/canopy/actions/runs/36993719788), [run 2](https://github.com/crabbuild/canopy/actions/runs/36993713588). Those c200d63 CI runs remain failed. The later f59617d regression/fix and green CI establish the inherited capture-lock defect; the precise phase of these earlier failures remains unproven. These results do not qualify production cutover or full-history/large-team capacity. + +- Input checkpoint increment (`b30feb0`): all 368 unique workspace library tests passed (6 Git-format, 14 object-storage, 348 server; server 126.73s). Seven new tests cover bounded multi-node inventories in both formats, exact replay/immutable SQL, authority and tamper rejection, source expiry, actual owner restoration with independent decoding of a retained real pack after old-pin expiry, bound preparation adoption without node copying, and Claim uncertainty after absent/lost replies and panic. The real receive/publication/cold-clone composition now checkpoints inputs in both formats. All-target workspace Clippy with warnings denied passed (37.16s), along with formatting, diff checks, whitespace for 15 changed/new files and 18 local documentation targets. Both CI runs passed at `b30feb0`, including workspace integrations, Python harness and RustFS compatibility. These are bounded correctness/composition checks, not full-history or large-team capacity qualification. + +- Native-input increment: the initial workspace run passed all 361 unique library tests (6 Git-format, 14 object-storage, 341 server), two CLI checks, ten Directory Cell checks and two Git HTTP checks. All three new capture tests passed. Multi-server passed 84, failed 12 and ignored nine (420.76s), exposing legacy loose-file/external-blob fixture assumptions under globally forced packed receive, plus Cell deadlines and system-wide file exhaustion. Packed receive is now scoped to the new API pending production hard cutover. Final composition/rejection checks passed (1.75s), and queued-upload ownership passed separately. All 12 unchanged failing integration cases passed sequentially after the correction (360.52s). **The original concurrent workspace run remains failed; this is not a clean full-workspace or capacity pass.** Final queued-upload ownership passed (0.10s). Rebuilt owner-restart, Repository Cell and stock-Git smart-HTTP checks passed (4.11, 27.88 and 51.35s). Final workspace/all-target Clippy with warnings denied passed (33.40s). Formatting, diff checks, whitespace for all 12 changed/new files and 17 local documentation targets passed. + +- Outcome-only increment: workspace library tests passed 357 and failed one existing native ancestry case with `Base(Inactive)` (server: 337 pass/one fail, 240.26s). All six new tests passed, covering both object formats, exact response-only persistence, moving/unavailable roots, payload/purpose rejection, write/read revocation, canceled observers, exact uncertain dispatch recovery and genuine owner restoration. The unchanged ancestry case passed in isolation (31.86s), with original assertions and lease deadline. **The concurrent library run remains failed.** + +- Outcome-only increment owner-restart and Repository Cell checks passed (5.60 and 43.36s). Stock-Git smart-HTTP passed after rebuilding (80.15s); its first attempt could not execute because the test binary was missing from the shared target directory. + +- Final `cargo +1.98.0 clippy --workspace --all-targets --locked -- -D warnings` passed (1m 12s). Formatting, diff checks, new-file whitespace and 31 local documentation links passed. + +Earlier native shutdown increment: + +- `cargo +1.98.0 test --workspace --locked` passed all 352 unique library tests (6 Git-format, 14 object-storage, 332 server), two CLI checks, ten Directory Cell checks and two Git HTTP checks. Five new resource tests cover terminal closure, admission/closure races, multiple and canceled observers, two-worker final-release races and poisoned/underflowed drain refusal. The new real-server check retains both native classes longer than one lease, verifies live renewal and workspace exclusion before Cell shutdown, cancels an observer, then verifies withdrawal and workspace reuse after release. Existing closed-stream and escaped-session fault fixtures now await the pool drain directly. +- The parallel multi-server run passed 94, failed two and ignored nine (403.96s). Large-object push returned HTTP 503 and bulk mirror returned HTTP 500; logs record Cell SQL deadlines and pending ingestion mutations. Both unchanged cases passed together sequentially (204.54s), with original assertions and deadlines. **The mixed run remains failed; this is not a clean full-workspace or capacity pass.** +- Owner-restart, Repository Cell and stock-Git smart-HTTP passed separately (4.06, 30.17 and 72.68s). These checks exercise the current production path; they do not qualify the missing packed-catalog cutover. +- Final `cargo +1.98.0 clippy --workspace --all-targets --locked -- -D warnings` passed (48.74s). Formatting, diff checks, new-file whitespace and 32 local documentation links passed. Workspace doctest targets passed with zero declared tests. +- The library suite includes native SHA-1/SHA-256 physical verification, canonical collision/typed closure rejection, admitted growth and rollback, shared edge-file witness replay, exact catalog ancestry, output projection/compaction inventories, staging renewal/ownership, account/class publication dispatch, owner-fenced atomic catalog/ref publication and original-receipt recovery. These are bounded primitive/composition checks, not full-history or 10,000-engineer throughput measurements. +- Both CI runs passed at native shutdown commit `b36a3a6`, including workspace integrations, Python harness and RustFS compatibility. Both CI runs also passed at outcome-only completion commit `a34ea8d`, including workspace integrations, Python harness and RustFS compatibility. Both CI runs passed at native-input capture commit `95ef9ea`, including workspace integrations, Python harness and RustFS compatibility. + +## Paged current ref policy validation + +`8bb0ee7` adds commands 33/35, query 34 and private guarded-root signing, with 4,096 guards, 2,097,152 watches and at most 512 indexed watch deletions per cleanup. Full workspace libraries passed all 469 unique tests with four threads (6 Git-format, 14 object-storage, 449 server; server 623.06s). Ten guard checks cover long-name/count paging, nonaligned evidence, ordering/MAC/ACL refusals, exact replay, current check/config invalidation, late-write rollback, capacities, immutable state, bounded cleanup and actual owner restore. The separate ancestry byte-bound regression passed; all-target workspace Clippy passed with warnings denied (1m56s), and all three retained-catalog integration cases passed (0.94s). Formatting/diff checks and 55 local document links passed. + +Earlier failures are retained: two compile API mismatches; one fixture alias mistaken for an unrelated OID; then eight passing guard cases and one long-name SQL codec failure. Ancestry queries now count exact wire bytes within 256 KiB/128 rows; the fixture derives production's existing 1 MiB descriptor. Production limits/deadlines are unchanged. The first Clippy pass found one test parity-style warning; its equivalent correction passed. See the [policy contract](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/paged-ref-policy-guards.md) and [status](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/large-repository-implementation-status.md). Final joint publication, failure-response roots, reviewed merges, service recovery/fairness, producer/reader cutover, containment/collection/restore and full-history/10,000-developer capacity remain mandatory unfinished scope. New-head CI is pending. + +## Native completion recovery validation + +- Final local workspace library validation passed all 429 unique tests: 6 Git-format, 14 object-storage and 409 server (192.20s). Five new checks cover 20,000-update plans beyond the 4 MiB inline envelope, multipart native responses and both OID formats; frozen inventories, refusal before registration and staging/bound reopening; corruption and repository revocation; foreign Directory/disabled account/read-scope/revoked-key refusal; malformed framing/count/actor/trailing bytes and disk rejection; and the unchanged 1 KiB envelope with all roots, adoption, predecessor and maximum actor. Existing queued cancellation now includes the consuming plan reader. Owner restoration reopens response/plan/options after source-pin expiry. The real native receive/publication/cold-clone/fsck composition publishes its recovered completion in both formats. Synthetic signer/completion fixtures qualify custody, not cryptographic/native validity. +- The initial full run passed 428 tests and failed the maximum-envelope fixture (408 server passed, one failed; 155.22s). Its synthetic repository UUID was noncanonical, so token validation refused it before sizing. Correcting the UUID bits preserved all maximum-field, integrity and size assertions; no production bound or deadline changed. That earlier run remains failed. +- All-target workspace Clippy with warnings denied passed (1m 54s). Rebuilt stock-Git smart-HTTP passed (64.98s), and signed-push/registered-key/audit-after-restore integration passed (3.54s). Formatting, diff, 30 changed/new files' whitespace and 63 local documentation links passed. +- Both Rust and harness CI runs passed at native-result commit `1319d0a`. These correctness fixtures do not establish hard cutover, provider durability, complete collection/isolated restore, OS containment or full-history/10,000-engineer mixed-load capacity. + +## Fresh empty initialization validation + +- Four focused initialization tests passed (3.02s), covering SHA-1/SHA-256 private preparation and deterministic retry, bounded input/reply framing and truncation, first ref preparation from the installed root, exact mutation/logical recovery, current identity/admin access, history/head/expiry/MAC/scope refusal, immutable outcome guards, rollback after late writes, concurrent attempts with exactly one winner and actual restored-owner recovery with stale pending-worker refusal. The factory future retains a Send assertion. The command input stays within 2 KiB, the shared result within 512 bytes and the existing catalog certificate within 1 KiB. +- The first focused compilation used a target accessor absent from the pinned runtime's QueryContext. The read-only lookup now uses the routed Cell capability and verifies its persisted repository identity and current admin role, following the existing compaction lookup. No deadline, limit or authority assertion changed. +- The full library invocation passed 446 unique cases and failed five existing native/catalog cases (server 426 passed/five failed; 324.74s). Physical partitions/base reuse, ancestry pair reuse, full ingress compaction and final current policies returned Base(Inactive); denied ancestry growth failed its expected budget-error assertion. All four new tests passed in that run. Ranked hypotheses before probing were preparation lease expiry, shared clock/fence interference and a schema/query regression. +- All five unchanged cases passed in separate sequential reproductions: 95.77s, 92.99s, 39.35s, 199.90s and 79.90s respectively. The original 60-second preparation leases, assertions and caps were retained. This supports contention/expiry as an explanation; the broad invocation remains failed and does not qualify concurrency or capacity. All 451 unique library cases passed across the broad run and these reproductions, without claiming a passing full-workspace invocation. +- All-target workspace Clippy with warnings denied passed (1m 32s). Rebuilt stock-Git smart HTTP passed (129.34s); signed SHA-256 SSH push, registered-key/audit assertions and fresh-disk restore passed (3.11s). Formatting/diff checks, whitespace for all 12 changed/new files and 52 local documentation links passed. The original checkout's protected staged index is unchanged. +- The singleton outcome retains the small initial catalog/ref roots; complete collection and isolated restore must include them. Production initialization wiring, short catalog/ref/outcome publication with current policy/check binding, every producer/reader, OS containment, continuous fair maintenance and full-history/large-team qualification remain open. +- Prior-head CI at `ecc5c6c`: [37067561966](https://github.com/crabbuild/canopy/actions/runs/37067561966) completed successfully. It precedes the main merge and does not qualify capacity. + +## Joint catalog/ref generation validation + +- All 447 unique workspace library tests passed: 6 Git-format, 14 object-storage and 427 server (server 160.29s). Four new tests cover both Git formats, query-derived base/root and canonical intent binding, retained old roots, deterministic retries, tombstones, stale-base/actor refusal, cross-format/future/missing metadata, joint-fact framing, MAC binding, maximum populated certificate fields, generation-floor retention, inline-publisher refusal with a whole-state oracle, and exact ref-root preservation through compaction/replay. The factory future retains an explicit Send assertion; existing native/catalog/recovery and current-policy tests also passed. +- The first all-target check/focused compile attempt failed on a missing Arc import. The first focused run then passed two tests and failed one (0.72s) because a mechanical fixture edit decoded four checkpoint columns from a three-column query. Restoring that fixture projection preserves the rejection assertions; the generation-state oracle independently includes the new ref descriptor. +- The first full library run passed 446 unique tests and failed the new maximum-shape certificate check (server 426 passed/one failed; 183.10s). The isolated check reproduced CodecError::Limit (1.16s). The v4 payload now encodes the already-bound repository/format/creating-operation context once and reconstructs the shared structures on decode. All three focused tests passed (1.63s), and the final complete suite passed with the same maximum-shape assertion, 960-byte payload cap and 1 KiB certificate cap. The prior v3 domain is refused without an adapter. +- All-target workspace Clippy with warnings denied passed (29.36s). Rebuilt stock-Git smart-HTTP passed (87.63s); signed-push/registered-key/audit-after-restore passed (3.44s). Formatting/diff checks, whitespace for all 21 changed/new files and 51 local document links passed. +- Prior-head CI had one success and one failure at `95650f0`, as recorded below. The failed SSH fixture released a probed HTTP port before server startup; it now lets the server bind port zero and uses the actual bound address. The same OpenSSH RSA/ECDSA case passed locally (1.87s). Authentication assertions and deadlines are unchanged. Both Verify runs [37060379452](https://github.com/crabbuild/canopy/actions/runs/37060379452) and [37060375239](https://github.com/crabbuild/canopy/actions/runs/37060375239) completed successfully at `07be856`. They precede the initialization increment; fresh-head results are recorded above. +- The production serving path remains unconverted. Production initialization wiring, membership/ancestry plus current policy/check proof, short atomic catalog/ref/outcome publication, every producer/reader, complete collection/isolated restore, OS containment and full-history/10,000-engineer mixed-load qualification remain mandatory. + +## Coalesced ref batch validation + +- All 443 unique workspace library tests passed: 6 Git-format, 14 object-storage and 423 server (server 189.50s). Six new ref-state tests and the extended 20,000-ref case cover existing-base rewrites, sparse prefix/gap/suffix and namespace changes, long-name byte splits in both formats, late-input failure/retry, old-root retention, exact versions and unchanged canonical plan digests. Independent BTreeMap inventories verify every sparse/long record; the preparation future retains an explicit Send assertion. +- A reversed 512-update batch against 20,000 refs stays within 40 artifact writes/loaded nodes. Sparse changes against 100,000 refs stay within 100 new objects and 64 loads. Rewriting all 20,000 long-name refs against an existing base stays within 500 new objects and 600 loads. Repeated prefix insertion retains 544 refs within 11 cold-loaded nodes, and 17,024 refs within 145 nodes at height two. These are fixture bounds, not sustained throughput results. +- The original per-update path-copy regression failed its write bound (11.21s). The original density regression loaded 37 nodes against its unchanged limit of 11 (0.44s). Coalesced traversal and balancing the final pair fixed both regressions. An all-target compile check passed (1m 34s), but adding the Send assertion exposed borrowed mapped-iterator lifetime errors; the first closure-only adjustment remained insufficient. A named streaming adapter fixed those lifetimes without allocating a second record inventory or dropping the assertion; the first 12 focused tests passed (25.69s). +- The first broader coalescer suite passed 440 unique tests and failed the existing queued-spool cancellation check (server 171.22s). An isolated run passed, then the unchanged case reproduced on iteration 15. Spool cleanup released disk credit before transfer admission, making zero disk usage observable before that owner's admission release. File-first, admission-next, disk-last destruction fixes the cleanup order. The unchanged case passed 1,000 repetitions (34.33s) and the final full suite. The earlier suite remains recorded as failed; no deadline, limit or cancellation assertion was weakened. +- All-target workspace Clippy with warnings denied passed (39.38s). Rebuilt stock-Git smart-HTTP passed (61.99s); signed-push/registered-key/audit-after-restore integration passed (4.01s). Formatting/diff checks, whitespace for all 10 changed/new files, and 51 local document links passed. +- Both prior-head Rust/harness runs passed at `487ebd2`: runs 37049237641 and 37049229572. At `95650f0`, Verify run [37056401256](https://github.com/crabbuild/canopy/actions/runs/37056401256) passed and [37056394903](https://github.com/crabbuild/canopy/actions/runs/37056394903) failed the multi-server OpenSSH RSA/ECDSA case with AddrInUse (95 passed, one failed, nine ignored; 470.72s). The affected fixture fix and local result are recorded in the joint-generation validation above. These checks do not establish authoritative ref-root publication, production hard cutover, complete collection/isolated restore, OS resource containment, or full-history/10,000-engineer capacity. + +## Immutable ref state validation + +- All 437 unique workspace library tests passed: 6 Git-format, 14 object-storage and 417 server (173.27s). Eight new tests cover exact versions, tombstone/ABA refusal, namespace swaps and Unicode, a 20,000-ref initial plan beyond 4 MiB, bounded node construction and one-path cold lookup/seek, bounded single-update path writes, 40,000-record deleted-subtree seeks and their actual namespace-check seam, long-name leaf/internal byte splits, large snapshot fences, narrower typed-root caps, purpose/format/repository/digest refusal and sorted-input/zero-live checks. Existing native/catalog/recovery/publication compositions also passed. +- The first focused run passed four tests and failed two (13.58s): the fixture requested an encoder bound above the runtime codec limit, and a live seek lost a later sibling. Bounded range encoding corrected the fixture. The expanded pre-fix run passed six and failed both cursor and actual namespace-check regressions (12.51s). Advancing from an exhausted seek branch to later ancestor siblings fixes the missed conflict; both regressions passed unchanged in the full suite. Existing typed-root bounds, deadlines and integrity assertions were preserved. +- All-target workspace Clippy with warnings denied passed (1m 25s). Rebuilt stock-Git smart-HTTP passed (142.36s). Signed-push/registered-key/audit-after-restore integration passed (4.57s). Formatting/diff checks, whitespace for 24 changed/new files and 53 local document links passed. +- Both Rust/harness CI runs passed at native-result commit `1319d0a` and ref-state commit `487ebd2` (runs 37049237641 and 37049229572). These correctness checks do not establish production hard cutover or full-history/10,000-engineer capacity. + +## Design and remaining requirements + +- [Immutable versioned ref state and root publication requirements](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/immutable-ref-state.md) + +- [Durable native completion and framed plan recovery](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/durable-native-result.md) + +- [Durable original push request and native append](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/durable-push-request.md) + +- [Owned production push preflight](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/owned-push-preflight.md) + +- [Final publication lifecycle and held ownership](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/final-publication-lifecycle.md) + +- [Bound lifecycle ownership, renewal and residence](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/bound-preparation-lifecycle.md) + +- [Bound preparation command ownership](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/bound-preparation-dispatch.md) + +- [Native input checkpoints and adoption](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/native-input-checkpoint.md) + +- [Native receive input capture](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/native-input-capture.md) + +- [Outcome-only push completion](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/outcome-only-completion.md) + +- [Node native resource admission, configuration and claims](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/native-resource-admission.md) +- [Native process ownership and descendant drain](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/native-process-ownership.md) +- [Shared dependency spool ownership and replay](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/verified-edge-spool.md) +- [Admitted SQLite construction growth](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/admitted-sqlite-growth.md) +- [Implementation status](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/large-repository-implementation-status.md) and [executable implementation plan](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/large-repository-implementation-plan.md) +- [Large-team workload budgets and mandatory qualification](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/large-team-scalability.md) +- [Staged input retention and late binding](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/staged-input-retention.md) +- [Service-owned staging lifecycle](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/staging-service-lifecycle.md) +- [Physical/logical run coverage](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/directory-run-coverage.md), [geometric maintenance](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/geometric-directory-maintenance.md), and [shared dispatch](https://github.com/crabbuild/canopy/blob/8bb0ee7cfbb7e11816b38b08cc9c81438be87ae6/docs/design/shared-publication-dispatch.md) + diff --git a/docs/design/admitted-sqlite-growth.md b/docs/design/admitted-sqlite-growth.md new file mode 100644 index 0000000..a91a9dd --- /dev/null +++ b/docs/design/admitted-sqlite-growth.md @@ -0,0 +1,41 @@ +# Admitted SQLite growth for repository construction + +MetadataBuilder, DirectoryBuilder, ClosureVerifier and the ref ancestry Walker now grow their private SQLite databases under the existing Cellule DiskBudget. They reuse the same SQLite tables, canonical identities, typed edges, native ordinal partitions and immutable artifact descriptors. No serving format changes or compatibility adapter are involved. + +## Allocation and transaction contract + +MetadataLimits.max_file_bytes remains an explicit main-file ceiling, validated as a multiple of 4 KiB. Each builder initially admits three times the smaller of 64 KiB and that ceiling, before creating its file. SQLite uses 4 KiB pages, DELETE rollback journals, its configured bounded cache and no mmap. The threefold charge covers main-file capacity, rollback journal and conservative overhead; it is a reservation rather than a claim about actual bytes written. + +Every mutable construction path executes through the shared admitted transaction helper. Only SQLite's DiskFull error can trigger growth. Preserve its source through MetadataError so validation limits, integrity failures, cancellation, I/O failures and identity conflicts cannot masquerade as a retryable capacity condition. A failed transaction must leave the connection in autocommit before replay is allowed. Otherwise fail closed. + +Before a retry, double the cap, clipped exactly to the configured ceiling. Grow the existing DiskReservation first; then raise max_page_count and verify the resulting cap. Failed admission leaves the old cap and credit intact. A SQLite cap update failure retains the enlarged reservation conservatively until cleanup. At the configured ceiling, return MetadataError::Limit. Actual filesystem exhaustion may produce the same SQLite error; retries are finite and remain bounded by the ceiling and shared budget. + +The replay body has no externally visible effects. Inputs remain owned until every attempt finishes. Commit before adopting its returned cursors, counts, digest folds or graph-processing state. Native metadata sealing consumes index ordinals once outside the replay body, then processes the same bounded object page on each SQL attempt. Closure copies recompute local inventory state per attempt. Topological processing copies its constant-size active reverse-fanout cursor and adopts it only after commit. + +Verified native objects retain their admitted dependency storage through the entire metadata batch. Physical verification shares one append-only file per page; each witness owns a private offset/length/digest range. Replay seeks to that exact range, checks the whole admitted file length and range bounds, hashes every occurrence byte, validates typed edges and checks the final digest again on each attempt. See the [shared edge-spool contract](verified-edge-spool.md). The outer consuming batch retains ownership and permanently poisons sealing on an unrecovered failure. A partial replay cannot become a complete witness or acknowledged publication. + +Closure lookup creation pages incoming OIDs and distinct child OIDs through their existing indexes, at most 512 keys per transaction. Pending-degree initialization also pages incoming vertices. Topological processing still performs at most 512 vertex/edge updates per transaction, and pages reverse fanout independently. A single large object's edge copy remains atomic and can require multiple cap increases; graph history and object bodies do not enter the heap. + +## Ancestry traversal and reuse + +The ref ancestry fallback reuses the existing visits queue, visits_ready index and answers table. Its constructor admits the same initial 192 KiB charge, and schema creation, parent insertion, expansion markers, memoized answers and queue cleanup all use the same admitted transaction helper. A denied cap increase cannot produce a negative ancestry answer. The configured ceiling remains a limit on the database, not on history length or an inferred count of commits. + +A Walker requires an exclusive mutable borrow across the entire traversal. This serializes pairs within one private proof operation; it introduces no repository-wide preparation lock. Bind memoized pairs to the exact existing StoredCatalog identity, including its artifact incarnation. Reusing a walker with another catalog rejects and fences it, even when endpoint OIDs match. Check the preparation session before and after asynchronous traversal, including identical-tip and memo-hit paths. + +Mark a traversal failed before its first await. A cancellation guard interrupts scratch work on error or dropped traversal; only a completed answer permits reuse. Failed/canceled walkers permanently reject later calls. Queued/running blocking callbacks retain the existing scratch Arc and file/workspace admission until they drain. Cancellation cannot turn partially expanded visits into a cached negative result or allow an old queued write to contaminate another pair. + +Clear the previous visits queue through its primary key, at most 512 keys per transaction, while preserving exact-catalog pair answers. Indexed selection avoids a temporary sort. SQLite can reuse freed pages for another pair, but their physical file capacity stays charged until the walker and all workers are dropped. Do not release credit merely because the queue has fewer rows. + +## Ownership and operational limits + +Sealing closes SQLite, removes its journal, synchronizes and hashes the file, then shrinks admission to the exact immutable length. Failed or canceled work keeps the existing AdmittedFile ordering: SQLite closes and files/workspaces are deleted before admission is released. Queued work retains ownership; cleanup failure retains disk credit for recovery. + +At the 256 MiB main-file default, one empty construction spool initially charges 192 KiB instead of 768 MiB. Two such spools initially charge 384 KiB. Growing to the full ceiling can still charge 768 MiB per spool, and old immutable files, cached downloads, native workspaces and edge files retain their independent reservations. Geometric growth can deny a larger quantum before all shared free space is consumed; this is bounded backpressure, not permission to oversubscribe disk. Do not equate the initial charge with peak operation capacity. + +Ancestry now uses admitted growth too. Native Git descendant RSS, CPU and I/O admission, large-file profiles, full-history throughput, durable authenticated input adoption, producer integration and continuous maintenance remain separate release gates. These construction changes do not select the production hard-cutover registry or prove capacity for 10,000 engineers. + +## Evidence + +Dedicated growth checks exercise rollback after a written prefix, repeated growth, main-file/journal bytes under held credit, denied admission preserving prior committed rows and the old cap, a non-power-of-two ceiling, and non-capacity failures that must not retry. Existing native SHA-1/SHA-256 verification now constructs a 1,600-blob inventory and wide tree with a 16 MiB file ceiling under a 2 MiB shared budget, checking exact metadata/edges and edge-file credit release. Existing deep-chain/wide-fanout, artifact-integrity, incomplete-input, canceled-worker, old-reader and compaction checks exercise the same growing builders. These are correctness/resource fixtures; full-history and mixed-load campaigns remain mandatory. + +Ancestry-specific checks reuse 1,200-commit native SHA-1/SHA-256 catalogs to exercise positive/negative memo reuse, exact-catalog rejection, denied growth with no negative answer, permanent fencing and canceled queued-worker ownership. The 100,001-entry synthetic scratch check now runs under a 64 MiB shared budget with a 32 MiB database ceiling, verifies indexed selection/cleanup, clears in bounded pages and reuses the queue while preserving pair answers. This remains a scratch-resource fixture, not a native 100k-history latency result. diff --git a/docs/design/bound-preparation-dispatch.md b/docs/design/bound-preparation-dispatch.md new file mode 100644 index 0000000..7a8a37c --- /dev/null +++ b/docs/design/bound-preparation-dispatch.md @@ -0,0 +1,25 @@ +# Exact bound preparation command ownership + +Long bound preparations and owner takeover need an exact recovery path for ClaimPreparation and RenewPreparation. ReadyPreparation::claim and PreparationSession::ready_renew prepare those SDK commands before admission to the existing PublicationCoordinator. They reuse LeaseCheck, LeaseRequest, PreparationToken, independent generation pins and the same coordinator job, actor queue and command evidence. The private request is boxed so its exact command/context does not enlarge every ReadyPublication value. No schema, command ID, outbox representation or compatibility adapter is added. + +## Admission and recovery + +Both commands enter the foreground class with an 8 KiB reservation for two bounded encoded copies. Account/operation limits, operation-ID exclusivity, FIFO account rotation, reserved maintenance slots and concurrent durability waits are shared with checkpoints and push completions. An uncertain command keeps its exact identity, bytes and credits. Dropping an observer does not cancel an accepted command. pending, recover and close_and_drain retain their existing behavior; recovery remains available after closing. Refused admission returns the original ready value. + +The factories check repository target equality, a nonzero lease duration within MAX_LEASE_MS and a 4 KiB encoded request. Renewal checks its shared session before and after SDK preparation. Claim intentionally accepts a previous-owner or expired token: the authoritative command checks exact operation identity, actor, phase, current write access, current admitted owner and SQL pin/quota invariants. An expired source may be claimed while it remains present; it cannot be renewed. Claim does not grant custody over old input bytes. Adopt and register the authenticated retained checkpoint before using borrowed physical inputs. + +## Original outcome and fresh custody + +PublicationOutcome::Preparation returns PreparationCommandOutcome with the command kind, original Committed and a separate session Result. Neither the reply's recorded timestamps nor replay alone constructs usable custody. Claim freshly opens CheckPreparation at the original receipt, matches the granted token, floor and format, and exposes a new private session. Renewal freshly queries the same attempt/floor/format and updates the existing shared conservative deadline. Existing fences remain permanent. Each query measures its local deadline from before request dispatch so queue/transport time shortens usable custody. + +A known success remains a known success when current permission, expiry or a later Claim prevents usable custody. The original receipt is returned alongside the custody error. Renewal failures fence the original shared session; an ambiguous command retains its old conservative deadline until resolved. Exact rejected/not-started renewals fence that session. Claim does not depend on or revive a previous local session. Final proof factories and authoritative commands continue to recheck custody independently. Preparation outcomes cannot become Git push responses. + +Renewal preserves the original generation floor and creating namespace. Claim creates a new admitted attempt and selects the current floor while preserving the previous independent pin. It does not advance an existing floor in place. An indefinitely renewed floor can still exhaust retained-generation capacity; the moving-root progress and capacity gates remain required. + +## Remaining lifecycle work + +This dispatcher owns one accepted Claim or Renew command, not the entire bound preparation lifecycle. Automatic renewal, bound worker/result ownership, checkpoint serialization, a separate local residence ceiling and shutdown drain now reuse the staging lifecycle; see the [bound lifecycle contract](bound-preparation-lifecycle.md). Final-publication lifecycle serialization, production producer integration, SQL floor-capacity qualification and durable takeover reconstruction remain required. The process-local exact command survives caller cancellation, not process loss. Unknown or expired SDK evidence cannot justify issuing a replacement command; durable exact/logical recovery must preserve that distinction. Complete retained-root enumeration, writer/reader drain and isolated restore remain required before collection. No remote deletion authority is introduced. + +## Validation scope + +Six checks cover actual Cell owner restoration and both OID formats, absent/lost/panicked dispatch, canceled observers, refused ready-value reuse and closed recovery, current revocation/expiry/superseding Claim, original receipts with failed fresh custody, permanent session fencing and unchanged renewal floors/old Claim pins. These are bounded correctness fixtures, not full-history import, production cutover, provider durability, complete restore or large-team throughput qualification. diff --git a/docs/design/bound-preparation-lifecycle.md b/docs/design/bound-preparation-lifecycle.md new file mode 100644 index 0000000..1eb28e5 --- /dev/null +++ b/docs/design/bound-preparation-lifecycle.md @@ -0,0 +1,29 @@ +# Service ownership through bound preparation + +StagingCoordinator now keeps an operation admitted after BindStaging and can admit a bound takeover through ReadyStaging::claim_bound. It reuses the existing job, actor/operation admission, shared actor worker semaphore, WorkSlots, exact command slot and independent SQL pins. Bound preparation does not need a second lifecycle service, worker queue or stored inventory layout. The standalone [exact Claim/Renew factories](bound-preparation-dispatch.md) remain available for direct dispatch; this lifecycle drives renewal without a caller tick. + +## Handoff and ownership + +Seal drains staged workers/results before Bind as before. A known Bind or bound Claim stores its original lease and receipt in bound_result before fresh queries. Successful fresh CheckPreparation token/floor/format matching constructs the shared bound PreparationSession. Staging contexts become inactive at that phase transition. Admission remains charged until explicit stop, loss of custody or the residence ceiling; Bound is a recorded handoff result, not release of service ownership. + +bound_session returns a locally live shared session only in the usable bound phase. open_base refreshes at the original bound receipt and constructs the existing PreparationBaseResolver with that session's deadline/fence. Base reads, reconciliation and private proof factories observe the same session. spawn_bound admits a callback into the existing global/actor worker slots and typed StagingTask handoff. Dropped callers retain execution/results in service ownership. Retrieved results transfer once; failed/expired results drop before their credits. Use the existing admitted native/disk/reader primitives inside callbacks; these worker counters do not account for arbitrary heap, unjoined descendants or detached I/O. + +A public Bound result remains recoverable after graceful stop and failed fresh custody. It is not usable authority. bound_session/open_base reject stopped or fenced jobs. Previously opened bases and session clones observe the shared permanent fence and residence ceiling. + +## Renewal and residence + +After binding or bound Claim, the same exact slot owns RenewPreparation. Its identity is created once, retained through absent/lost/panicked replies, and resolved before another renewal or checkpoint begins. A known renewal is saved in bound_renewal before fresh CheckPreparation. Renewing preserves the original attempt, namespace and generation floor; a current authorization/expiry/superseding-Claim failure fences the session without erasing the original renewal result. Unknown evidence remains admitted until exact recovery, including after close. + +The existing overall staging lifetime still applies. bound_lifetime_ms adds a separate local ceiling from before preparation of Bind, or admission of bound Claim, rather than resetting it after a delayed/replayed reply. Default is 60 seconds; valid profiles use a nonzero ceiling up to MAX_LEASE_MS. The ceiling is fixed in the shared session and all derived bases. A direct fresh lease query or renewal cannot extend local use past it. The supervisor uses the raw fresh SQL deadline to schedule renewal and the residence ceiling to stop work; clipping the renewal schedule to the ceiling would cause a tight renewal loop near the limit. + +The local ceiling does not shorten or remove the independent SQL pin. Already admitted or queued renewals can leave remaining SQL retention beyond local shutdown. Remaining-floor/fact capacity, command queue delay, server clock bounds and lease configuration still need qualification against the [large-team budgets](../large-team-scalability.md). Long physical imports belong in unbound staging; indefinitely renewing a bound floor is not permitted by this lifecycle. An expired local session cannot revive through exact replay or a successful later query. + +## Adopted checkpoint and shutdown + +A bound Claim may adopt its authenticated retained input root through the shared session. register_inputs now accepts that matching adopted certificate in the bound phase and uses the existing single 4 KiB checkpoint/result slot, adding it to the 8 KiB command-copy reservation. Due renewal precedes queued registration. RegisterStagedInputs shares the exact slot with renewal; its original receipt is stored before fresh checkpoint-digest and bound-session queries. Fresh custody failure fences the job while the committed registration remains observable through pending_inputs. The original creating namespace and input root are reused; no nodes or native pairs are copied. A staged checkpoint already consumes that operation's single slot. + +stop/close refuse new staged or bound workers, keep renewing while accepted tasks and retained results drain, and retain uncertain exact evidence/credits. Service consumers must retrieve completed results to finish graceful drain. Once drained, the shared bound session is fenced before operation admission is returned. Reached residence, worker error/panic or lost custody aborts and joins outstanding callbacks and discards untransferred results before credit release. SQL pins and remote artifacts remain independently retained. + +## Remaining integration and release gates + +Final publication now serializes through the [same lifecycle and account-fair coordinator](final-publication-lifecycle.md), retaining a small held ticket while that coordinator owns the exact proof/command reservation. Workers/results, due renewal and accepted checkpoints drain before activation; known final results fence the shared session without querying the retired preparation record. A generic producer result or this process-local supervisor is not a durable publication outbox, authenticated wire-plan/response recovery or owner-loss reconstruction. Production HTTP/SSH/mirror/generated producer and reader conversion, OS CPU/RSS/file/PID/I/O containment, hot-root progress and continuous maintenance, physical pack rewriting/accelerated readers, complete retained-root enumeration/drain/collection, source-independent isolated restore and hard cutover remain required. Full Linux/Kubernetes/Chromium history and 10,000-engineer mixed-load/recovery gates remain unqualified. No remote deletion authority is introduced. diff --git a/docs/design/catalog-root-v1-sha1.hex b/docs/design/catalog-root-v1-sha1.hex new file mode 100644 index 0000000..5566718 --- /dev/null +++ b/docs/design/catalog-root-v1-sha1.hex @@ -0,0 +1 @@ +0000001763616e6f70792e636174616c6f672d726f6f742e76310000000010010101010101010101010101010101010000001005050505050505050505050505050505140000001002020202020202020202020202020202000000000000008000000020030303030303030303030303030303030303030303030303030303030303030300000020040404040404040404040404040404040404040404040404040404040404040401000000100c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0000000000001000000000200a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a000000200b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b010000003006060606060606060606060606060606070707070707070707070707070707070707070707070707070707070707070700000030080808080808080808080808080808080909090909090909090909090909090909090909090909090909090909090909000000000000000b0000000000000040 diff --git a/docs/design/catalog-root-v1-sha256.hex b/docs/design/catalog-root-v1-sha256.hex new file mode 100644 index 0000000..9a04b87 --- /dev/null +++ b/docs/design/catalog-root-v1-sha256.hex @@ -0,0 +1 @@ +0000001763616e6f70792e636174616c6f672d726f6f742e76310000000010010101010101010101010101010101010000001005050505050505050505050505050505200000001002020202020202020202020202020202000000000000008000000020030303030303030303030303030303030303030303030303030303030303030300000020040404040404040404040404040404040404040404040404040404040404040401000000100c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0000000000001000000000200a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a000000200b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b010000003006060606060606060606060606060606070707070707070707070707070707070707070707070707070707070707070700000030080808080808080808080808080808080909090909090909090909090909090909090909090909090909090909090909000000000000000b0000000000000040 diff --git a/docs/design/check_native_pack_contract.py b/docs/design/check_native_pack_contract.py new file mode 100644 index 0000000..198c41e --- /dev/null +++ b/docs/design/check_native_pack_contract.py @@ -0,0 +1,73 @@ +#!/usr/bin/env python3 +"""Disposable native Git smoke fixture for the proposed storage contract. + +Checks SHA-1/SHA-256, native pack/index installation, and thin-pack completion. +Does not exercise Canopy durability or simulate production scale. +""" +from pathlib import Path +import os +import shutil +import subprocess +import tempfile + + +def git(path, *args, data=None): + env = {k:v for k,v in os.environ.items() if not k.startswith('GIT_')} + env.update(GIT_CONFIG_NOSYSTEM='1', GIT_CONFIG_GLOBAL=os.devnull, + GIT_AUTHOR_NAME='Fixture', GIT_AUTHOR_EMAIL='fixture@example.invalid', + GIT_COMMITTER_NAME='Fixture', GIT_COMMITTER_EMAIL='fixture@example.invalid') + return subprocess.run(['git','-C',str(path),*args],input=data,stdout=subprocess.PIPE, + stderr=subprocess.PIPE,env=env,check=True,timeout=60).stdout + + +def run(fmt, root): + source, target = root / 'source', root / 'target' + source.mkdir(parents=True) + target.mkdir() + git(source,'init',f'--object-format={fmt}') + git(target,'init','--bare',f'--object-format={fmt}') + (source/'data').write_text('line\n'*10000) + git(source,'add','data') + git(source,'commit','-m','base') + base=git(source,'rev-parse','HEAD').strip() + prefix=target/'objects/pack/pack' + checksum=git(source,'pack-objects','--revs','--index-version=2','--delta-base-offset', + '--threads=2','--window-memory=64m','--depth=50','--max-pack-size=1g',str(prefix), + data=base+b'\n').strip().decode() + basepack=prefix.with_name('pack-'+checksum+'.pack') + git(target,'index-pack','--verify',str(basepack)) + git(target,'update-ref','refs/heads/main',base.decode()) + (source/'data').write_text('line\n'*9999+'changed\n') + git(source,'add','data') + git(source,'commit','-m','next') + tip=git(source,'rev-parse','HEAD').strip() + thin=git(source,'pack-objects','--revs','--thin','--stdout',data=tip+b'\n^'+base+b'\n') + git(target,'index-pack','--stdin','--fix-thin','--index-version=2',data=thin) + git(target,'update-ref','refs/heads/main',tip.decode()) + git(target,'symbolic-ref','HEAD','refs/heads/main') + git(target,'fsck','--full','--strict') + expected=git(source,'rev-list','--objects','--all').splitlines() + actual=git(target,'rev-list','--objects','--all').splitlines() + assert sorted(expected)==sorted(actual) + git(target,'multi-pack-index','write') + git(target,'multi-pack-index','verify') + git(target,'commit-graph','write','--reachable','--changed-paths') + git(target,'commit-graph','verify') + # Prove the completed thin pack can decode every physical object by itself. + for pack in (target/'objects/pack').glob('*.pack'): + isolated=root/pack.stem + isolated.mkdir() + git(isolated,'init','--bare',f'--object-format={fmt}') + for ext in ('.pack','.idx'): + shutil.copyfile(pack.with_suffix(ext),isolated/'objects/pack'/pack.with_suffix(ext).name) + git(isolated,'index-pack','--verify',str(isolated/'objects/pack'/pack.name)) + ids=git(isolated,'cat-file','--batch-all-objects','--batch-check=%(objectname)') + git(isolated,'cat-file','--batch',data=ids) + print(f'{fmt}: pack/index, thin completion, isolated decode, MIDX, commit graph passed') + + +if __name__=='__main__': + print(subprocess.check_output(['git','--version'],text=True).strip()) + with tempfile.TemporaryDirectory(prefix='canopy-pack-design-') as directory: + for fmt in ('sha1','sha256'): + run(fmt,Path(directory)/fmt) diff --git a/docs/design/check_packed_repository_schema.py b/docs/design/check_packed_repository_schema.py new file mode 100644 index 0000000..cce64ae --- /dev/null +++ b/docs/design/check_packed_repository_schema.py @@ -0,0 +1,142 @@ +#!/usr/bin/env python3 +"""Execute the proposed DDL with Canopy's unchanged product tables. + +Design validation only: no Canopy processes, deployments or persistent databases. +Run from any directory with Python 3 (stdlib only). +""" +from pathlib import Path +import re +import sqlite3 +import unittest + +ROOT = Path(__file__).resolve().parents[2] +REPLACED = { + "object_uploads", "object_chunks", "objects", "object_edges", "object_closure", + "object_pending", "refs", "ref_generation", +} + + +def schema(): + # Split using SQLite's parser so comments and quoted semicolons are safe. + statements, pending = [], "" + for line in (ROOT / "crates/canopy-server/src/schema.sql").read_text().splitlines(keepends=True): + pending += line + if sqlite3.complete_statement(pending): + clean = re.sub(r"--[^\n]*", "", pending).strip() + table = re.search(r"^CREATE TABLE (\w+)", clean) + index = re.search(r"^CREATE (?:UNIQUE )?INDEX \w+ ON (\w+)", clean) + insert = re.search(r"^INSERT INTO (\w+)", clean) + target = table or index or insert + if target is None or target.group(1) not in REPLACED: + statements.append(pending) + pending = "" + if pending.strip(): + raise AssertionError("Unparsed source SQL") + return (ROOT / "docs/design/packed-repository-schema.sql").read_text() + "\n" + "".join(statements) + + +class PackedSchema(unittest.TestCase): + def setUp(self): + self.db = sqlite3.connect(":memory:") + self.db.executescript(schema()) + self.db.execute("INSERT INTO repository_identity VALUES ('sha1',1,?,'owner',?)", (b'r'*16, b's'*32)) + self.db.execute("INSERT INTO pack_operations VALUES (?,'ingest',?,1,1,'open',0,0)", (b'o'*16,b'i'*16)) + + def tearDown(self): + self.db.close() + + def pack(self, byte=1, sealed=False, fmt='sha1'): + width = 20 if fmt == 'sha1' else 32 + digest = bytes([byte])*32 + cur = self.db.execute("""INSERT INTO packs(operation_id,digest,object_format,git_checksum, + size,manifest_digest,index_size,index_digest,index_manifest_digest,object_count, + inventory_digest,staged_digest) VALUES (?,?,?,?,1,?,1,?,?,1,?,?)""", + (b'o'*16,digest,fmt,b'c'*width,digest,digest,digest,digest,digest)) + pk = cur.lastrowid + if sealed: + self.db.execute("UPDATE packs SET staged_count=1,last_oid=?,state='sealed',sealed_generation=id WHERE id=?",(b'x'*width,pk)) + return pk + + def seal(self, pack): + self.db.execute("UPDATE packs SET staged_count=object_count,last_oid=?,state='sealed',sealed_generation=id WHERE id=?",(b'x'*20,pack)) + + def obj(self, oid, pack, kind='blob', edges=0): + self.db.execute("""INSERT INTO objects(oid,kind,size,digest,pack_id,edge_count,edge_digest) + VALUES (?,?,0,?,?,?,?)""",(oid,kind,b'd'*32,pack,edges,b'e'*32)) + seq = self.db.execute("SELECT sequence FROM objects WHERE oid=?",(oid,)).fetchone()[0] + self.db.execute("INSERT INTO object_pending(sequence,oid,edge_digest) VALUES (?,?,?)",(seq,oid,b'e'*32)) + + def test_composes_with_collaboration_schema(self): + self.assertEqual(self.db.execute("PRAGMA foreign_key_check").fetchall(), []) + tables = {r[0] for r in self.db.execute("SELECT name FROM sqlite_master WHERE type='table'")} + self.assertTrue({'pull_requests','pushes','commit_parents','lfs_objects'} <= tables) + self.assertFalse({'object_uploads','object_chunks','object_locations','git_pack_parts'} & tables) + + def test_format_and_incomplete_seal(self): + with self.assertRaises(sqlite3.IntegrityError): + self.pack(fmt='sha256') + p = self.pack() + with self.assertRaises(sqlite3.IntegrityError): + self.obj(b'x'*32,p) + with self.assertRaises(sqlite3.IntegrityError): + self.db.execute("UPDATE packs SET state='sealed' WHERE id=?",(p,)) + + def test_sha256(self): + self.db.execute("UPDATE repository_identity SET object_format='sha256'") + self.obj(b'x'*32,self.pack(fmt='sha256')) + self.assertEqual(self.db.execute("PRAGMA foreign_key_check").fetchall(), []) + + def test_identity_location_and_retirement(self): + old, new = self.pack(), self.pack(2,sealed=True) + self.obj(b'x'*20,old) + self.seal(old) + with self.assertRaises(sqlite3.IntegrityError): + self.obj(b'x'*20,new) + with self.assertRaises(sqlite3.IntegrityError): + self.db.execute("UPDATE objects SET digest=?",(b'z'*32,)) + with self.assertRaises(sqlite3.IntegrityError): + self.db.execute("UPDATE objects SET pack_id=?",(new,)) + with self.assertRaises(sqlite3.IntegrityError): + self.db.execute("UPDATE packs SET state='retired' WHERE id=?",(old,)) + self.db.execute("UPDATE objects SET pack_id=?,location_version=2 WHERE pack_id=? AND location_version=1",(new,old)) + self.db.execute("UPDATE packs SET state='retired' WHERE id=?",(old,)) + stale = self.db.execute("UPDATE objects SET pack_id=?,location_version=3 WHERE pack_id=? AND location_version=1",(new,old)) + self.assertEqual(stale.rowcount,0) + self.assertEqual(self.db.execute("SELECT sequence, location_version FROM objects").fetchone(),(1,2)) + + def test_reverse_edge_propagation_and_late_parent(self): + p=self.pack() + leaf, parent, late = b'l'*20,b'p'*20,b'q'*20 + self.obj(leaf,p) + for oid in (parent,late): + self.obj(oid,p,'tree',1) + self.seal(p) + self.db.execute("INSERT INTO object_edges VALUES (?,?,'blob',1)",(parent,leaf)) + self.db.execute("UPDATE object_pending SET received_edges=1,last_child=?,remaining_children=1,edges_complete=1 WHERE oid=?",(leaf,parent)) + self.db.execute("INSERT INTO object_closure VALUES (?)",(leaf,)) + for _ in range(2): + changed=self.db.execute("UPDATE object_edges SET waiting=0 WHERE parent=? AND child=? AND waiting=1",(parent,leaf)).rowcount + self.db.execute("UPDATE object_pending SET remaining_children=remaining_children-? WHERE oid=?",(changed,parent)) + self.assertEqual(self.db.execute("SELECT remaining_children FROM object_pending WHERE oid=?",(parent,)).fetchone(),(0,)) + self.db.execute("INSERT INTO object_edges SELECT ?,?,'blob',NOT EXISTS(SELECT 1 FROM object_closure WHERE oid=?)",(late,leaf,leaf)) + self.assertEqual(self.db.execute("SELECT waiting FROM object_edges WHERE parent=?",(late,)).fetchone(),(0,)) + + def test_deleted_artifact_can_be_reimported_with_new_id(self): + old=self.pack(sealed=True) + with self.assertRaises(sqlite3.IntegrityError): + self.pack() + self.db.execute("UPDATE packs SET state='retired' WHERE id=?",(old,)) + self.db.execute("UPDATE packs SET state='deleting' WHERE id=?",(old,)) + self.db.execute("UPDATE packs SET state='deleted' WHERE id=?",(old,)) + new=self.pack() + self.assertGreater(new,old) + + def test_missing_dependencies_rejected(self): + p=self.pack() + self.obj(b'p'*20,p,'tree',1) + with self.assertRaises(sqlite3.IntegrityError): + self.db.execute("INSERT INTO object_edges VALUES (?,?,'blob',1)",(b'p'*20,b'x'*20)) + + +if __name__ == '__main__': + unittest.main(verbosity=2) diff --git a/docs/design/directory-run-coverage.md b/docs/design/directory-run-coverage.md new file mode 100644 index 0000000..df31830 --- /dev/null +++ b/docs/design/directory-run-coverage.md @@ -0,0 +1,58 @@ +# Immutable directory run coverage + +This is the persisted contract for directory-root v3 and range-index v2. It supports bounded compaction windows while retaining the existing native files, canonical headers, source identities, placement versions and persistent range tree. The deployment is a hard cutover: previous directory-root and directory-leaf domains are rejected. Production schema/handler selection and deployment-format installation remain pending. + +## One representation for full files and projections + +Every `StoredRun` contains a complete physical `RunDescriptor`, an authenticated `ArtifactDescriptor`, and a logical `RunCoverage`. A whole-file record uses the physical descriptor's count, endpoints and canonical inventory as its coverage. A projection uses inclusive OID endpoints and the exact count and canonical inventory of the entries within those endpoints. There is no alternate legacy representation or optional projection flag. + +Physical facts continue to identify the entire immutable file: repository, object format, creating operation, count, first/last OID, canonical inventory, byte size and file digest. The artifact additionally binds its authenticated manifest. Coverage never changes these facts or the file's bytes. + +Coverage uses the existing `inventory_seed` and `fold_header` encoding over strictly increasing OIDs. The fold excludes physical placement. A projection starts a fresh canonical fold at ordinal zero; its digest is not a byte-range hash or a substring of the parent's digest. Source and location versions remain in the entries and merge with the existing canonical-conflict and preferred-placement rules. + +Structural validation rejects zero counts, counts larger than the physical count, wrong OID widths, reversed or out-of-file endpoints, and inconsistent singleton endpoints. Full count or both full endpoints require the exact complete physical coverage. Structural checks do not prove a partial inventory; trusted preparation must fold its actual entries before certifying it. + +## Persisted bytes + +Directory snapshots use `canopy.directory-root.v3\0`. Directory range nodes use `canopy.range-index.v2\0`. Source leaves retain their separate existing domain. Native pack/index formats and the canonical header fold are unchanged. + +All integers below use the existing bounded codec's big-endian encoding. Byte arrays are prefixed by a big-endian u32 length. A directory leaf record serializes these fields in order: + +| Field | Encoding | +| --- | --- | +| Physical creating operation | bytes, length 16 | +| Physical object count | u64 | +| Physical first OID | bytes, length 20 or 32 | +| Physical last OID | bytes, length 20 or 32 | +| Physical canonical inventory | bytes, length 32 | +| Artifact byte size | u64 | +| Artifact file digest | bytes, length 32 | +| Artifact manifest digest | bytes, length 32 | +| Coverage object count | u64 | +| Coverage first OID | bytes, same object format | +| Coverage last OID | bytes, same object format | +| Coverage canonical inventory | bytes, length 32 | + +The node context carries repository and object format. Physical descriptor size/digest are reconstructed from the artifact and must agree. Decoder bounds, exact field widths, structural validation and trailing-byte rejection still apply. + +Independent fixed-byte record vectors are [SHA-1](directory-run-v2-sha1.hex) (284 bytes) and [SHA-256](directory-run-v2-sha256.hex) (332 bytes). They are codec fixtures with deliberately synthetic inventory facts; they grant no closure or publication authority. Tests compare exact encoder bytes and decode them independently. + +Directory node fanout is 128, encoded node size is at most 64 KiB, and height is at most seven. Index keys, endpoints and represented object counts use logical coverage. Generic node counts may include overlaps across roots or source shards; they are never a unique canonical-object proof or deletion authority. Snapshot point selection remains bounded by 32 ingress roots plus 16 levels. + +## Verified bounded replacement + +1. Select one catalog-derived source projection and a consecutive prefix of complete target records within the input record/physical-byte budget. Charge a shared physical file once and reject conflicting physical facts under the same identity. +2. Scan the complete source projection in indexed pages of at most 512 entries. Fold parent, moved prefix and retained suffix in the same pass. Reject before either fragment escapes if the parent's count, endpoints or canonical inventory differ. Empty fragments are absent. +3. Merge the prefix and selected targets through the existing admitted builder and partitioner. For disjoint promotion within the output file limit, reuse the verified original physical artifact. The suffix always retains its exact physical descriptor and artifact identity. +4. Revalidate the exact source incarnation and target interval against the selected publication base. Path-copy their replacements, retaining unrelated current files. A replaced source or newly overlapping, missing or changed target rejects reuse. Bind the original source, prefix, suffix and selected targets into the range-compaction v2 input digest. +5. Issue the existing purpose-bound maintenance certificate and publish through admin command 22. The catalog update and immutable outcome are atomic; refs and source roots remain unchanged. Query 23 and existing lease, owner-fence, checkpoint and replay rules continue to apply. + +If an unaffordable first target starts after the source's first OID, the preceding disjoint source prefix may progress. Otherwise a required physical input larger than the job budget rejects. A source spanning many affordable target files progresses through repeated jobs; the scheduler must revisit the retained suffix rather than skip to the original source's last OID. + +## File sharing, retention and cost + +The catalog file cache validates a requested projection, then normalizes its cache descriptor to the complete physical coverage. Distinct projections share one authenticated file, reader admission and disk charge. Different physical descriptors or manifests under the same physical key fail even on a cache hit. Snapshot lookup selects requests by logical coverage before grouping by physical file, so a projection cannot expose entries outside its endpoints. + +Raw descriptors and range-node summaries provide no authorization or closure proof. Only a trusted catalog/preparation context certifies coverage. Reclamation must inventory all retained roots and valid pins, deduplicate physical artifact incarnations, and fence deletion; removal of one projection cannot authorize deleting a file retained by another. Online reclamation is not implemented by this increment. + +Each window scans its complete parent logical projection, including the suffix, to preserve exact canonical proofs. Output runs are bounded at 64 MiB by the current partitioner, but repeated windows can amplify read/CPU work. Geometric selection, continuous fair scheduling, whole-process resource admission and full-history mixed-load measurements must qualify that cost. Current defaults retain 128 input records/256 MiB physical input, a 256 MiB spool ceiling with 192 KiB initial charge and up to 768 MiB reservation, and at most 64 MiB output files. These limits and small native correctness fixtures do not establish capacity for 10,000 engineers. diff --git a/docs/design/directory-run-v2-sha1.hex b/docs/design/directory-run-v2-sha1.hex new file mode 100644 index 0000000..cd95c38 --- /dev/null +++ b/docs/design/directory-run-v2-sha1.hex @@ -0,0 +1 @@ +0000001002020202020202020202020202020202000000000000000a00000014000000000000000100000000000000000000000000000014000000000000001400000000000000000000000000000020040404040404040404040404040404040404040404040404040404040404040400000000000040000000002005050505050505050505050505050505050505050505050505050505050505050000002006060606060606060606060606060606060606060606060606060606060606060000000000000002000000140000000000000004000000000000000000000000000000140000000000000007000000000000000000000000000000200909090909090909090909090909090909090909090909090909090909090909 diff --git a/docs/design/directory-run-v2-sha256.hex b/docs/design/directory-run-v2-sha256.hex new file mode 100644 index 0000000..dc80304 --- /dev/null +++ b/docs/design/directory-run-v2-sha256.hex @@ -0,0 +1 @@ +0000001002020202020202020202020202020202000000000000000a00000020000000000000000100000000000000000000000000000000000000000000000000000020000000000000001400000000000000000000000000000000000000000000000000000020040404040404040404040404040404040404040404040404040404040404040400000000000040000000002005050505050505050505050505050505050505050505050505050505050505050000002006060606060606060606060606060606060606060606060606060606060606060000000000000002000000200000000000000004000000000000000000000000000000000000000000000000000000200000000000000007000000000000000000000000000000000000000000000000000000200909090909090909090909090909090909090909090909090909090909090909 diff --git a/docs/design/durable-native-result.md b/docs/design/durable-native-result.md new file mode 100644 index 0000000..fb10b8e --- /dev/null +++ b/docs/design/durable-native-result.md @@ -0,0 +1,31 @@ +# Durable native completion recovery + +The packed preparation API retains the native completion before Bind so a successor can recover the exact response, versioned ref intent, options and verified signature witness after local cache loss. It reuses PushCompletionRequest, PushPlan, GitHttpResponse, SignedPushAnnotation, ArtifactDescriptor, GitInput, the authenticated artifact transport and the independent checkpoint lease. The production serving path still requires the coordinated hard cutover. + +## Capture and registration + +Register the [original request checkpoint](durable-push-request.md) before native receive. Run native Git and capture its pack/index pairs through the existing native fence. Within the same admitted staging context, pass the exact native PushCompletionRequest to retain_native_result together with the registered predecessor. The factory checks live checkpoint custody, logical request scope, format, response and option bounds, plan shape and native report agreement. A signed result requires the existing opaque VerifiedPushCertificate with the same target, request digest and actor. + +The factory uploads response and optional signature bodies, streams the versioned plan through a charged anonymous spool, and writes a bounded result metadata root. SavedNativeResult privately binds that root to the staging target, full lease check, format, original wire root and exact predecessor digest. append_native_result consumes this owner and appends captured descriptors through the existing path-copy native index. Register the resulting checkpoint through the staging supervisor before Bind. Issued artifacts or a decoded root alone cannot authorize reopening. + +The staged-native-inputs.v3 checkpoint carries the original request, native index and native result roots within the existing 1 KiB MAC envelope. Command 29 compares the exact predecessor and current lease. Once a result is attached, changing the captured native inventory is refused; exact replay and adoption of all unchanged roots remain valid. This prevents later input append from silently changing the native result's meaning. Physical verification, graph closure, ref policy, current write authority and final atomic publication remain independent gates. + +## Stored representation and resource bounds + +InputBody and InputRoot share create-only paths `repos/{repository}/git-inputs/{creating-operation}/{bodies|roots}/{digest}`. Each artifact binds its own content digest and authenticated manifest. The original request root remains capped at 64 KiB. The native result root is capped at 128 KiB and stores descriptors instead of response, plan or signature bytes. It binds the creating operation, exact wire-request root, HTTP status and ordered headers, response descriptor, optional plan descriptor, options and optional signer/body/key annotation. The native-result.v1 domain rejects other metadata purposes. + +The retained-push-plan.v1 stream consists of a length-prefixed header followed by length-prefixed canonical PushPlan chunks. Frames are at most 64 KiB and chunks contain at most 32 existing RefUpdate records. The header carries the validated actor and total update count. Each chunk reuses PushPlan's existing codec, preserving names, expected live/tombstone OIDs, exact versions and new OIDs. Decoding rejects inconsistent actor/count, incomplete or oversized frames and trailing data. Total encoded size is capped at 128 MiB and the existing 100,000-update limit still applies. Encoding borrows bounded ranges rather than cloning the full update vector. A 64 KiB scan computes the spool digest before authenticated upload. + +Responses and signature bodies retain the existing 64 MiB limits and are already owned vectors from native collection. Recovery authenticates every body part before returning the completion. Plan recovery downloads to an admitted disk spool, then uses a blocking reader that owns the spool until its file cursor closes. Observer cancellation cannot release disk or transfer admission while queued or running file work still uses it. These local bounds do not prove whole-operation CPU or RSS containment, provider budgets or full-history performance. + +## Reopening and owner change + +StagingContext and PreparationSession expose reopen_native_result. Both query fresh registered custody, authenticate the bounded result root, require its exact original wire root and current logical scope, reconstruct the framed plan and bodies, validate response/options/report agreement, then require the same checkpoint under a second fresh custody query. Missing registration, corrupt artifacts, changed attempts, expired pins or revoked repository access reject. + +For signed results, recovery also requires the Directory Cell for the repository's exact tenant and application. Its current push_signers query requires an enabled account and enabled write-scoped key with the retained fingerprint. A matching key from another Directory, a disabled account, a read-scoped key or a revoked key cannot reconstruct VerifiedPushCertificate. The constructor is reachable only after these checks and MAC-registered custody of a result captured from the original opaque verified witness. Raw request signature text never enters that constructor. + +Takeover adoption preserves the exact original request, native index and result roots under the successor's independent pin. It authenticates their metadata and logical scope without copying large bodies or walking all input leaves. Once adoption commits, the destination independently retains every borrowed creating namespace after source-pin expiry. Reopening may then download the retained bodies under fresh successor custody. Collection and isolated restore must include these transitive roots and unresolved writers; this API grants no remote deletion authority. + +## Release requirements + +This representation supports durable preparation and recovery for plans larger than the inline command bound. The final publication command still has a 4 MiB envelope and must gain a bounded root interface before large ref batches can be published. The [immutable ref-state data plane](immutable-ref-state.md) supplies conditional versioned roots; final publication must also remove the per-ref SQL loop and atomically commit the catalog, ref and outcome roots under current authority and policy checks. Production HTTP, SSH, mirror and generated-write producers, canonical readers, takeover orchestration and fresh-schema selection must switch together. Full retained-root collection, isolated restore, OS containment, accelerated reads, continuously fair maintenance and full-history mixed-load qualification for large teams remain required. diff --git a/docs/design/durable-push-request.md b/docs/design/durable-push-request.md new file mode 100644 index 0000000..28fe9d8 --- /dev/null +++ b/docs/design/durable-push-request.md @@ -0,0 +1,35 @@ +# Durable encoded push requests + +The packed preparation API can retain the authenticated original request before native receive, register a request-only checkpoint, then append native pack/index descriptors to that same checkpoint. This reuses GitInput, BeginRequest, GitHttpRequest, ArtifactDescriptor, NativeInputIndex and the independent catalog lease row. Production producer selection still requires the hard cutover; the current gateway's selected storage path has not switched. + +## Ownership and artifact format + +EncodedPush computes both the scoped v3 request digest and the unkeyed body digest in one blocking scan with a 64 KiB buffer. The latter is cached privately in the owned preflight. retain consumes that owner, freshly checks its admitted staging target/token/actor/format and uploads the same anonymous spool through the existing pinned-file transport. The authenticated uploader independently checks all bytes against the cached digest. No second body spool, ref-vector layout or additional whole-body hashing pass is required before upload. Queued opens and reads retain the spool's disk and transfer admission through cancellation. Successful upload rewinds and returns the original encoded owner for native decoding. + +Request bodies and metadata roots use the existing create-only authenticated manifest/part transport. Paths are `repos/{repository}/git-inputs/{creating-operation}/{bodies|roots}/{digest}`. Both kinds bind their own digest; a later incarnation never reuses an earlier operation path. Cancellation can leave unregistered staging in the retained creating namespace, which the complete collector must account for before deletion. + +The encoded wire body and native pack can contain much of the same data, but they are different artifacts: gzip, protocol prefixes and thin-pack completion change their bytes. Retaining both is temporary recovery storage, charged alongside native input retention until the operation and independent pins can be retired safely. It is not repository-history duplication on every fetch. Bulk admission, provider-byte budgets and collection qualification must include both artifacts and captured native metadata. + +WireRequest reuses GitHttpRequest with ArtifactDescriptor as its body and the original BeginRequest as its identity. Its `canopy.wire-request.v1\0` metadata binds tenant, application, creating namespace, format, method, path, query, optional content type, gzip/protocol flags and encoded-body descriptor. The root codec is capped at 64 KiB; the request body retains the existing 512 GiB artifact bound and configured request/disk limits. Neither that bound nor a small root proves full-history performance. An authenticated flag in these server-created metadata is a prior transport assertion; current write custody is checked independently at recovery and publication. + +## Checkpoint sequence + +Inside an admitted StagingTicket worker, call EncodedPush::retain, then StagingContext::seal_push_inputs with the opaque SavedPushRequest and an empty native iterator. Register the returned NativeInputCertificate through ticket.register_inputs before launching native Git. The durable checkpoint contains the request root alongside the optional existing native index root. It does not copy encoded bytes into the Cell. + +Decode the returned EncodedPush with the existing gzip and packet parser, run_native_receive, and stage_native_packs. StagingContext::append_native_inputs starts from the exact registered predecessor, inserts new current-namespace descriptors through NativeInputIndex's existing path-copy operation, preserves every old descriptor and the exact request root, rechecks custody, then signs the next checkpoint. Duplicate descriptors return the original proof without another revision. Register that proof before Bind. Physical verification and final catalog/ref/response publication retain their independent gates. + +The checkpoint domain is now `canopy.staged-native-inputs.v3\0`, with no old-domain adapter. Command 29 and query 30 retain their identities. The existing lease row adds only a previous digest and bounded revision counter. Initial registration uses revision zero; append compares the exact previous envelope digest and increments once, with at most 256 revisions per attempt. SQL forbids removal, partial changes, revision jumps, wrong predecessors, updates after Bind and simultaneous Bind/append. The private append issuer establishes preservation of the authenticated tree; the final command rechecks scope, MAC, current permission, owner/attempt and live pin before comparing its predecessor. Racing appends conflict rather than overwrite each other. + +The staging supervisor reuses its one 4 KiB checkpoint slot. It permits replacement only while unbound, after a known successful registration, and when the next proof names the exact completed digest. Pending, uncertain, failed and unrelated proofs cannot replace the slot. Existing observers retain their original receipt. Dropped observers recover the latest accepted command through pending_inputs; recovery retains its exact mutation identity and bytes. There is no additional queue or per-object Cell inventory. + +## Reconstruction and adoption + +StagingContext::reopen_push_request and PreparationSession::reopen_push_request query the registered checkpoint under fresh current custody, authenticate its bounded root and body, spool with the existing DiskBudget/transfer admission, and recompute the complete scoped request identity. A declared size above the caller's limit rejects before body download. A final custody query must return the same proof before the encoded owner escapes. Missing roots, scope mismatches, corrupt parts, changed checkpoints, revoked access, expired pins or changed attempts reject and release incomplete local input. + +Owner takeover adopts both exact immutable roots into the successor's existing independent pin. Source custody is checked at construction and final registration. Adoption copies no request bytes or input nodes. After the destination checkpoint commits, it retains the original creating namespace independently of source-pin expiry. Reopening can reconstruct signed command bytes and options as intent; it does not synthesize VerifiedPushCertificate or authorize an acknowledgement. + +The real receive/publication fixture uses a request-only registration before native work, attaches the retained native result with actual captured inputs, loses its local request/receive/source caches, reopens the original request and native completion, independently verifies physical pairs, publishes, then cold-clones and fscks in both formats. Additional tests cover large plain/gzip bodies, encoded-size rejection, bound recovery, actual restored-owner adoption after source-pin expiry, late-part corruption with partial-spool release, revoked custody, exact uncertain append recovery, old observer receipts, cancellation ownership and SQL revision/phase guards. Synthetic signed text in request tests qualifies byte/intent retention only; native signature and pack validity come from independent checks. + +## Required continuation + +The original encoded request is now durable. The [durable native result API](durable-native-result.md) now preserves the versioned plan, exact response, options and scoped signature witness in the same checkpoint. Final publication still needs a short root command interface and production takeover orchestration. Plans beyond the 4 MiB command envelope require a bounded reference, not a larger inline command. Every production producer/reader and fresh-schema selection must switch together for release. The complete collector/isolated restore must retain this new artifact family and borrowed namespaces, including unresolved writer ownership; these APIs grant no remote deletion authority. Hard containment, continuous fair maintenance and full-history/large-team mixed-load qualification remain mandatory. diff --git a/docs/design/final-publication-lifecycle.md b/docs/design/final-publication-lifecycle.md new file mode 100644 index 0000000..43bfc0b --- /dev/null +++ b/docs/design/final-publication-lifecycle.md @@ -0,0 +1,42 @@ +# Final publication owned by the bound lifecycle + +StagingCoordinator now serializes final push and compaction publication after bound workers/results, due renewal and queued input registration. It uses the existing PublicationCoordinator job, private ready proof, exact SDK command, class/account fair queue and byte reservation. The staging job retains a small ticket; it does not copy the command or create another payload queue. Production handlers and durable owner-loss reconstruction still require integration. + +## Admission and handoff + +PublicationCoordinator::try_reserve synchronously admits a Held job without dispatch. It applies the same target, logical-operation uniqueness, per-class account/operation limits and factory-derived byte reservation as submit. A contended admission lock returns Capacity with the original ready value; the consumer must retry through its bounded scheduling policy. submit still admits directly to Queued. Held jobs do not consume dispatch concurrency or advance the fair queue. + +StagingTicket::publish accepts only a final privately prepared push or compaction sharing the bound session's target, exact lease check, deadline, permanent fence and fixed ceiling. An independently opened session with equal SQL tokens is insufficient. Checkpoint and Claim/Renew factories cannot enter this final handoff. Refusal returns the original ready value. Successful synchronous reservation stores the small ticket and seals new bound worker/checkpoint admission before returning the observation-only StagedPublicationTicket. pending_publication recovers that observer after cancellation. A recorded original bound receipt remains available separately. + +The existing workers and retained typed results must drain before activation. The supervisor keeps renewing while they drain; due renewal and any accepted checkpoint resolve first in the existing exact slot, including uncertain commands recovered after close. Final preflight uses the latest known Bind/Claim, renewal or registration receipt as its minimum query watermark. Successful activation joins the existing class/account fair queue once. The lifecycle stops issuing renewal/checkpoint commands while final dispatch or recovery owns the logical operation. + +Retrieve the producer's StagingTask result before waiting for publication. Awaiting publication inside an owned producer would prevent that producer's own drain. The intended sequence is: + +```rust,ignore +let base = Arc::new(stage.open_base(indexes, files).await?); +let work = stage.spawn_bound(move |_| async move { + // Build/verify with existing admitted native/disk/reader primitives. + // Return a private ready_push or ready_compaction value. + prepare_ready(base).await +})?; +let ready = work.wait().await?; +let observer = stage.publish(&publications, ready)?; +let outcome = observer.wait().await; +// Uncertain: retain the same stage and recover it explicitly. +``` + +This is a composition sketch; prepare_ready is the producer's existing private factory work. The native receive fixture now follows this handoff and reconstructs a cold stock-Git clone from the resulting catalog. A maintenance fixture constructs compaction in the same bound-owned worker and publishes through the reserved maintenance class. + +## Exact outcomes, fencing and shutdown + +PublicationTicket::activate is idempotent for an already activated job. It never retries uncertainty. discard_held serializes with activation and succeeds only while execution has not started. It drops the original proof/resources before returning class/account/byte credits and records Discarded, which cannot acknowledge a push. Activation/discard and exact recovery remain available for previously admitted work after coordinator close. close_and_drain returns held and uncertain tickets still charged; it does not activate or silently discard them. Close the staging lifecycle before the shared publication coordinator in normal shutdown. + +A custody failure or local ceiling before activation discards the held proof and fences/drains the lifecycle's existing worker/result resources before releasing operation admission. An activated command keeps its exact identity/proof and original outcome regardless of later local fencing. Initial transport submission and resubmission after authoritative absence recheck the shared session's local deadline/fence/ceiling. An inactive guard records NotStarted and cannot become an acknowledgement. Known committed results are resolved before that local guard, so an expired/fenced session cannot erase an earlier durable receipt. Unknown, expired, unreachable, malformed or changed-incarnation outcome evidence remains uncertain and charged. + +Final uncertainty appears as StagingState::Uncertain with the original typed PublicationError. StagingCoordinator::recover schedules the same retained PublicationTicket in its fair queue, including after both services close. The lifecycle also observes recovery performed directly by the shared coordinator, preventing a resolved ticket from leaving staging admission stranded. Known final success or rejection becomes Published with its original outcome. The lifecycle permanently fences its session and releases local resources/admission; it does not query the preparation record after completion, since successful publication retires it. The bound receipt, checkpoint receipt and publication result remain distinct. + +The response observer is service-internal access to an already admitted result. Externally requested replay must use the existing authenticated replay_push_response preflight and current read authorization. Compaction results cannot become push responses. A known catalog conflict terminates this local lifecycle; a subsequent Claim and freshly reconciled proof must enter a new admitted lifecycle rather than replacing an ambiguous command. + +## Release gates + +These services retain process-local ownership, not a durable outbox. They do not reconstruct authenticated wire plans/responses after process loss. Staging admission still needs production scheduling, maintenance preparation shares and OS CPU/RSS/file/PID/I/O containment. Fresh-schema selection, HTTP/SSH/mirror/generated producer and reader cutover, hot-root progress, continuous maintenance, physical pack rewriting/read acceleration, complete retained-root collection/drain and source-independent isolated restore remain required. The full-history and 10,000-engineer [mixed-load/recovery gates](../large-team-scalability.md) remain unqualified. No remote deletion authority is added. diff --git a/docs/design/geometric-directory-maintenance.md b/docs/design/geometric-directory-maintenance.md new file mode 100644 index 0000000..56a8c0b --- /dev/null +++ b/docs/design/geometric-directory-maintenance.md @@ -0,0 +1,39 @@ +# Geometric directory maintenance + +`CompactionPlanner::prepare_next` selects one bounded verified directory replacement from a query-derived preparation base. It reuses `DirectorySnapshot`, `NodeRef` summaries, the logical/physical [coverage contract](directory-run-coverage.md), the range builder/partitioner, and command 22/query 23. It adds no authoritative tables, per-object state or compatibility layer. Physical native pack rewriting and production continuous scheduling remain separate required work. + +## Policy and pressure + +The default policy starts level 0 at 262,144 logical objects and multiplies its target by four for each successive nonoverlapping level. Target arithmetic saturates at SQLite's maximum signed 64-bit count, preventing overflow from hiding debt. Object counts are exact within a certified disjoint level; overlapping ingress/levels are not summed into a global unique inventory. + +`CompactionPolicy::pressure` exposes at most 32 ingress roots and two fixed arrays of 16 level counts/targets. Reading this pressure examines only bounded snapshot summaries. Selecting and preparing a job then authenticates indexed paths and verifies exact inventories. Pressure is advisory and cannot authorize publication or deletion. + +Any ingress root remains eligible, including a tail below the urgent watermark. A level is eligible when its logical count exceeds its target; exact equality needs no promotion. The final level has no successor. Pressure above its target returns a capacity error rather than selecting a nonexistent level or reporting a drained backlog. Configure a qualified larger profile or architecture before exceeding that terminal capacity; keep admission bounded. + +Profiles require a positive base count within the SQL count range, ratio 2–16, ingress high water 1–32 and urgent burst 1–32. Defaults use eight roots as the urgent watermark and three as the maximum urgent burst. These values are initial policy, not measured optimal settings. + +## Fair local selection + +There are 16 rotation classes: ingress plus 15 promotable levels. With urgent ingress, dispatch at most three ingress preparations before giving one eligible higher level a turn. Urgent jobs preserve the higher-level rotation position; restarting it at level 0 would starve deeper levels under continuous arrivals. With all 15 levels continuously eligible and successful preparations, every level receives a turn within 60 preparations under the default burst. This is a job-selection bound, not a time, I/O, throughput or publication-progress guarantee. + +Ingress selection rotates through the current bounded root slots. Each promotable level keeps one exclusive last-moved OID. Seek after that OID using the existing range cursor, then wrap to the beginning only upon indexed exhaustion. This revisits lower ranges introduced by intervening publications and follows retained suffixes using the moved prefix's last OID, rather than the original physical file's endpoint. Selection retains no full level inventory or history-sized queue. + +The planner binds repository/object format after its first successful preparation and rejects reuse in another context. Its fixed-size local cursor is advisory. Resetting it after restart changes traversal order but cannot drop work from the authoritative catalog. Failed preparation and cancellation before completion leave rotation unchanged. Successful private preparation advances it without claiming a durable acknowledgement. The caller must retain the prepared inputs and recover any uncertain publication through the existing exact command/outcome rules. + +## Execution contract + +1. Begin an admitted maintenance preparation under the current owner fence and open its query-derived base using existing lease/retention APIs. +2. Call `prepare_next` with a service-owned planner, private workspace, shared disk budget and qualified compaction limits. The API checks live deadlines and current admin access, including when it reports no eligible work. +3. The selected source and target window use the existing `PreparedCompaction::prepare_range`. Verify complete source/projection folds, output inventory and exact path replacement. A required physical input outside the resource profile fails admission; selection never expands the configured limit. +4. Use `ready_compaction` to issue the purpose-bound maintenance certificate and retain exact command 22, then submit through the [shared foreground/maintenance dispatcher](shared-publication-dispatch.md) under the selected catalog CAS. On a changed frontier, reuse `reconcile` only while its exact selected source/target incarnation checks hold. Replaced inputs require a new preparation. +5. Keep unknown outcomes and their inputs admitted until authoritative recovery. After a known outcome, release private scratch and obtain a new queried base before the next job. Do not use a local rotation advancement as evidence that bytes were published or can be collected. + +The selection API does not run a timer, choose maintenance/foreground CPU/I/O resource shares, provide a durable service outbox, or replace owner-loss reconstruction. Those components must integrate it before production schema selection. An unadmittable required file needs a qualified resource profile; repeatedly selecting it cannot prove service progress. + +## Performance and acceptance + +Pressure selection is O(16 + 32) time and fixed space. Range seeking uses bounded-height authenticated paths. Per-job input/run/output bounds and the 48-candidate lookup bound remain unchanged. Every source window folds its complete parent projection, so repeated windows can amplify I/O/CPU. Geometric targets organize debt but do not remove that cost. + +Tests cover both object formats, invalid/overflowing profiles, final-level capacity, exact-threshold idleness, urgent ingress with every higher level continuously pressured, root rotation, tail ingress, and repository/format separation. Native fixtures repeatedly query, prepare and publish until ingress and geometric debt drain, comparing complete effective canonical/source/version entries, unchanged refs and pinned old-root reads after every job. Admission failure retries the same ingress selection, and an empty selection still rechecks current admin access. + +Full-history Kubernetes/Linux/Chromium imports, 100,000 incremental commits, amplification/retention breakdowns, maintenance service exceeding arrivals under foreground load, owner-loss recovery, isolated restore and the large-team working-day/peak/headroom campaigns remain mandatory unpassed gates. This increment is not evidence of capacity for 10,000 engineers. diff --git a/docs/design/immutable-push-outcomes.md b/docs/design/immutable-push-outcomes.md new file mode 100644 index 0000000..6cbb653 --- /dev/null +++ b/docs/design/immutable-push-outcomes.md @@ -0,0 +1,35 @@ +# Immutable push outcome preparation + +`PreparedCatalog::root_push_completion` prepares a joint catalog/ref/outcome input bounded by 8 KiB. It does not execute a publication command or acknowledge a push. The final atomic publisher, authorized completed-response query, streaming replay adapter, typed collector and production hard cutover remain required. + +This factory requires a successful nonempty ref plan. Immutable outcome-only completion for a failed or empty-command receive remains required; the old inline outcome-only API is not a hard-cutover adapter. + +The private factory obtains its native result from the current registered input checkpoint. It authenticates that checkpoint against the prepared catalog's physical input custody, reopens the exact native plan/report/options and uses the existing scoped signing-key lookup for a signed annotation. The native plan must match the private policy guard's original intent and evidence. All ref expectations are checked through the immutable ref transition against the selected joint generation; there is no SQL-ref fallback. For a preparation with no physical inputs, the checkpoint must have no native pack inventory. Its exact digest is still included in the final certificate. + +The factory freezes three outcomes before signing: + +| Choice | Response | +| --- | --- | +| Native | Exact original status, headers and immutable body descriptor | +| Publication rejected | Existing all-successful-ref rejection transformation, preserving native failures and progress | +| Signed certificate replayed | Same transformation with the existing certificate-replay reason | + +An absent report-status response becomes an explicit HTTP 409 rejection. The final command must select a durable refusal when current policy/ACL or signed ownership rejects publication. A moving catalog CAS must instead preserve reconciliation/retry semantics. It must never retain an unselected native success as the client outcome. + +## Representation and namespaces + +`NativeOutcomeRoot` reuses `StoredInputRoot`, the shared immutable metadata representation. Its typed metadata reuses `GitHttpResponse` and links to the original `NativeResultRoot` for plan/options/signed audit information. A body-operation field identifies the response bytes' creating namespace. This avoids copying the large plan, original request, options or signed body into each alternative. + +Every new outcome metadata artifact and rejection body belongs to the current admitted attempt. The native success body can remain in the original native creator namespace after a legitimate owner claim/adoption. Creating replacement artifacts in that old namespace would race its retention lifecycle, so preparation never does so. Registered custody and the adopted independent pin retain borrowed artifacts while preparation remains active or uncertain. + +Completed retention must traverse the selected response body and the original native metadata's plan/options/signed annotation. It must not permanently retain the native metadata's original wire request/body merely because that descriptor remains present. For a rejection, the native success body also needs no completed-response retention unless a separate audit policy selects it. Active/uncertain input pins retain those private-input dependencies independently. The collector must implement and qualify these distinct typed traversals before any deletion is enabled. + +An outcome metadata record cannot decode as native-result metadata. A decoded root is transport data, not a native witness or read capability. The completed-response API must select the root from durable actor/logical-operation/request identity under current read authorization; it must never accept a caller-selected root. That API is not implemented yet. + +## Binding and final obligations + +The completion certificate binds the exact catalog/base/pin/token/actor, policy intent, immutable ref snapshot, response UUID, ref generation, all three outcome descriptors and optional SHA-256 signed-certificate ownership facts. The largest permitted signing-key string is 4,096 bytes. Plans and response bytes are absent from the bounded input. After freezing every artifact, preparation rechecks checkpoint custody and live policy readiness before minting the completion-purpose certificate. The existing inline publisher refuses this purpose. An admitted service must retain this exact prepared input and its mutation identity through an uncertain command result; regenerating a new response UUID is not replay. + +These checks are conditional preparation. The final admitted command still must authenticate the complete MAC and check its actual owner fence, current lease/pin/ACL, guard/epoch/dependencies, selected retained base and root CAS in the same transaction as signed ownership, joint generation and selected outcome writes. It must return the original selected outcome on exact completed replay before requiring new authority, preserve independent pins and roll back every late error. It must perform no remote reads, response rewriting, per-object/per-ref writes or whole-plan decoding. + +Preparation still reopens a `Vec` plan/report and uses the existing report transformer. The 8 KiB bound applies to Cell transport and does not prove whole-operation RSS, CPU or end-to-end throughput. File-backed intent/report processing and hard OS containment remain mandatory, as do full-history Linux/Kubernetes/Chromium and 10,000-developer mixed-load qualification. diff --git a/docs/design/immutable-ref-state.md b/docs/design/immutable-ref-state.md new file mode 100644 index 0000000..7be952b --- /dev/null +++ b/docs/design/immutable-ref-state.md @@ -0,0 +1,65 @@ +# Immutable versioned ref state + +Large atomic ref plans need an immutable data plane so publication can commit a root rather than one SQL mutation per ref. `packs::ref_state` now reuses the authenticated `RangeIndex`, `NodeRef`, artifact transport, `RefExpectation`, `RefUpdate` and canonical `PushPlan` digest. It provides conditional preparation and reads. The serving path still uses SQL refs; selecting this tree requires the coordinated fresh-data hard cutover and final authority protocol described below. + +## Stored records and ordering + +A leaf stores an exact UTF-8 Git ref name and the existing optional OID plus positive version. A deleted ref remains as a tombstone with its version. Recreating a deleted name must match that tombstone; a missing expectation cannot erase an intervening delete and recreate. Every accepted update advances the version, including an identical tip. An expectation at `i64::MAX` rejects before arithmetic or uploads. + +Names use shared `Arc` keys in byte order. Ref-name coordinates may represent slash prefixes; leaf validation separately requires valid Git ref syntax. The new representation limits names to 65,535 encoded UTF-8 bytes. Preparation rejects longer names before artifact creation. This is an explicit new-format limit, rather than an assertion that native Git accepts every name up to that length on every filesystem. + +Ref nodes use the `canopy.ref-state-index.v1` purpose domain, fanout 128, maximum encoded node size 512 KiB and maximum height 16. Splitting considers encoded bytes as well as member count, including both full fence names in each child reference. Existing fixed-width object and source indexes retain their original domains, fanout, byte and height bounds. The generic tree now requires Clone rather than Copy; fixed-width references remain Copy. + +`record_count` includes tombstones. The shared weighted count (`object_count` in the existing reference structure) counts live refs for this record type. Authenticated node decoding recomputes both counts and binds exact child height and fence ranges. The ordinary ordered cursor exposes tombstones; a live cursor skips zero-weight child subtrees. Point reads and seeks use a bounded-height path and the existing 64-node cache. Root and parent authentication are prerequisites for trusting a skipped child's summary. A raw caller-supplied root remains untrusted publication input. + +## Conditional preparation + +`RefStateIndex::prepare` accepts the exact selected base root, creating operation and existing PushPlan. It checks canonical operation identity, actor and update shape, unique names, server-owned ref refusal, OID format, versions and every exact expected state before uploading changes. Namespace checks consider the resulting plan as a whole: deleting a parent while creating its child and deleting a child while recreating its parent are allowed. Two resulting live names in a parent/descendant relationship are refused. Tombstones do not occupy a live namespace. + +Ancestor checks use point reads. Descendant checks seek directly to the slash-prefix interval and skip authenticated zero-live subtrees. Planned deletions remove conflicts, while planned live descendants create conflicts. An exhausted seek branch must advance to later live siblings; otherwise namespace validation could miss a live descendant after many deletions. + +An empty base uses the shared streaming sorted builder. It holds an active block and at most one completed block behind it at each level, validates strict input ordering and persists groups by fanout and encoded bytes. Large initial inventories therefore write tree nodes instead of copying a path for every member. It consumes a sorted iterator over the borrowed plan map and does not materialize another complete inventory of encoded records. + +An existing base uses `RangeIndex::upsert_sorted` to rewrite affected subtrees once. Validation visits the sorted map of borrowed updates for cache locality, while the canonical plan digest retains the caller's original order. The rewriter consumes that map without allocating another complete record inventory. It keeps one pending change and descends only where the next change falls within a child's interval. Gaps route to the next child, and the last child inherits the enclosing interval so insertions can extend the old fence. + +Touched leaves merge their existing records with the sorted changes. An equal key must retain the exact old range; partial range overlaps are refused. Untouched children join the shared builder through their authenticated parent references without loading all descendants. Before appending a reused higher subtree, the builder lifts any pending lower groups into that height. This preserves lexical order and balanced child heights, including partial edge groups. The frontier holds at most two byte/fanout-bounded blocks per level, plus at most two leaf blocks. It balances an underfilled final pair by encoded bytes and fanout before flushing at reuse boundaries or completion. This avoids a full leaf or parent plus a tiny tail on repeated prefix insertions. Working inventory remains bounded by height and node size; it never materializes the full unchanged inventory or a list of all new leaves. + +Every expected version and namespace check still finishes against the exact immutable base before ref preparation uploads nodes. An invalid later structural input or canceled traversal yields no output root; already emitted immutable nodes may remain unpublished. Repeating the same sorted rewrite from the same base and creating operation recreates the same descriptors through create-only transport. These facts confer no publication authority or deletion permission. Whole-operation/provider budgets, sustained batching behavior and hot-root fairness remain capacity gates. + +The private `RefTransition` binds the exact base root, proposed root and existing canonical plan digest. It proves only the conditional structural result. Catalog membership, ancestry, current access, branch rules, required checks, root CAS and exact outcome durability remain separate final-publication obligations. + +## Snapshot metadata + +`RefStateSnapshot` reuses the existing repository ID, ObjectFormat, generation, default-branch string and optional shared tree root. `RefStateSnapshotRoot` stores this metadata through the existing create-only InputRoot transport using the separate `canopy.ref-state-snapshot.v1` purpose domain. The snapshot metadata bound is 256 KiB, enough for the longest default branch and both longest root fence names. The command-facing descriptor remains below 128 bytes. + +The shared transport allows metadata up to 256 KiB, but each typed wrapper keeps its own bound: original wire requests remain 64 KiB and native results remain 128 KiB. Decoding a larger snapshot descriptor as either narrower root refuses it. Reading checks authenticated manifest/body bytes, full framing, purpose domain, canonical creating operation and repository. Generation zero cannot contain a nonempty tree. A snapshot descriptor validates the described root's shape; it does not establish descendant existence or current Cell authority. + +## Query-derived joint generations + +The fresh publication schema now stores an optional bounded ref snapshot descriptor in the existing immutable `catalog_generations` row. `GenerationFact`, preparation leases/frontier queries and catalog certificates carry that same descriptor. The original generation floor retains the joint fact and all later generations. This reuses the catalog generation and pin model instead of introducing an independently advancing ref-root table. Catalog compaction copies the exact ref descriptor into its next generation; it does not change the ref generation stored in the authenticated metadata. + +`PreparedCatalog::prepare_ref_snapshot` accepts only the existing plan. It derives the selected snapshot from its privately constructed, query-derived base and loads it through the prepared catalog's own artifact-store capability. It checks repository, object format and ref generation, reuses the shared coalesced transition, preserves the authenticated default branch and uploads the next metadata root under the held creating namespace and deadline. Its privately constructed output binds the joint base, proposed snapshot and original canonical plan digest. Callers cannot substitute their own store, root or conditional transition into this factory. + +A missing selected snapshot is unavailable, not an empty repository. The reserved generation-zero fact still contains no ref descriptor. The fresh initialization protocol below installs authenticated empty metadata; it must be wired into repository creation during the coordinated cutover. Catalog membership, ancestry and policy/check evidence remain final certificate obligations. This conditional factory does not issue publication authority, select production serving or remove the inline completion protocol. + +The catalog attestation purpose domain is now `canopy.catalog-attestation.v4`; the prior domain is refused, with no decoder adapter. The existing 1 KiB certificate bound remains. The v4 payload encodes the repository/format/creating-operation context once and reconstructs the shared catalog and generation structures on decode; it does not repeat those facts in full nested descriptor domains. A selected ref descriptor is part of the signed joint base. The current inline SQL ref publisher refuses new writes against a selected immutable snapshot, preventing a partial conversion from silently discarding that root; exact already-recorded outcome replay remains available. Full root publication and all producers/readers must switch together before release. Keeping a SQL generation row does not prove that complete remote collection or isolated restore is implemented. + +## Fresh empty initialization + +`PreparedCatalog::empty_ref_initialization` uses the held preparation's own artifact store and private empty catalog. It requires the reserved generation-zero base, zero objects/edges/inputs, no input checkpoint, current admin access and a live held lease. It uploads authenticated ref metadata with ref generation zero, an empty tree and `refs/heads/main`, then issues the existing bounded catalog certificate with a purpose-separated ref descriptor binding. It accepts no caller snapshot, imported SQL ref inventory or signing adapter. The input stays within 2 KiB and the exact shared GenerationFact result within 512 bytes. + +`InitializeCatalogRefs` authenticates the certificate, rechecks the actual owner fence, current admin role, format, attempt/pin/expiry, retention floor and exact base. It refuses any existing ref row, including tombstones, changed default head/ref generation, push/compaction/initialization outcome or positive catalog generation. Its bounded final transaction records the checkpoint, installs catalog generation one and ref metadata generation zero together, persists one immutable initialization outcome and retires the logical preparation. Failures after the first write abort all of these facts together. The independent attempt pin remains retained under its existing lifecycle. + +The singleton outcome stores the existing GenerationFact rather than a new root representation. Exact logical replay returns that original fact without granting a new write, including after actual owner restoration; a pending old attempt must still satisfy the new owner fence. `CheckInitializedCatalog` requires current admin access, the routed Cell's persisted repository identity and exact actor/request identity. It creates no namespace or synthetic mutation receipt. The retained outcome includes the small initial catalog/ref descriptors forever; complete collection and isolated backup/restore must include those roots even after the current generation advances. Remote collection remains unselected. + +Initialization supplies a fresh empty base for ref preparation. Production repository creation, short final catalog/ref/outcome publication and every producer/reader still need to switch together. Selecting initialization alone would leave legacy readers unaware of subsequent immutable ref updates. + +## Final publication and cutover requirements + +Direct-ref preparation now binds current rule/context epochs and exact newest required-run dependencies through [paged policy guards](paged-ref-policy-guards.md). Pages reuse the existing catalog MAC with a separate purpose, validate at most 128 updates/256 KiB and advance their cursor atomically with dependency watches. The private guarded-snapshot factory rechecks the original plan/catalog evidence, conditional transition and fresh readiness before signing a short root proof. This remains conditional input: final admitted root/outcome publication must recheck the live guard, epoch and current authority together with its exact root CAS. Reviewed merges and production conversion remain required. + +The final factory must derive the base from a certified query of the current Cell ref snapshot, authenticate the complete conditional transition and bind it to exact catalog/ref intent and outcome descriptors. Preparation must bind every relevant policy and check fact, with final current-authority checks and CAS against the state those facts describe. A concurrent policy, check or ref change must refuse or trigger safe re-preparation. Recovery must reuse the same retained intent and recorded outcome without authorizing a stale worker. + +The admitted final transaction must atomically switch catalog and ref roots, record the exact response root and advance the durable operation outcome. It must neither carry the full plan in the 4 MiB envelope nor execute an O(update-count) SQL loop inside that transaction. Root transport alone would leave the latter bottleneck intact. + +All HTTP, SSH, mirrors and generated-write producers, ref listing and point readers, default-branch behavior, policies, reviews/checks, recovery and backup paths must switch together under the new schema marker. Retention and collection must traverse these snapshots, tree nodes, tombstones and borrowed creating namespaces. Raw roots and unpublished artifacts cannot authorize remote deletion. Hot-root fairness and sustained ordinary/bulk existing-base workloads remain qualification work alongside the full-history Linux, Kubernetes and Chromium mixed-load gates. diff --git a/docs/design/native-input-capture.md b/docs/design/native-input-capture.md new file mode 100644 index 0000000..d91d0e2 --- /dev/null +++ b/docs/design/native-input-capture.md @@ -0,0 +1,27 @@ +# Native request input capture + +GitHttpBackend::run_native_receive configures receive.unpackLimit=0 so even small staged receives leave native pack/index pairs. It reuses the existing bounded CGI response collector and native process owner. Production producer conversion to this API remains open. Git remains responsible for receive protocol, thin-pack completion and private ref updates. Receiving successfully is preparation; the durable catalog/ref/response transaction remains the acknowledgement boundary. + +## Admitted capture + +GitHttpBackend::stage_native_packs runs inside a StagingTicket producer after the native receive has drained. It takes the service's StagingContext, ArtifactStore and existing PhysicalLimits. A fresh context token supplies the repository, creating artifact namespace and object format. Repository/format mismatches reject before provider access. The creating namespace is never chosen by a client filename or content digest. + +Capture reuses the private GitCache and its disk reservation. It serializes cache selection, reconciles completed native disk writes, then takes an exclusive lock on the same native worker fence used by every Git command. An active native worker causes a refusal rather than concurrent file mutation. The blocking scan retains both the cache and a node read claim; hashing uses fixed 64 KiB buffers. + +Only ordinary pack/info object directories are permitted. Loose or quarantine object directories, nonregular pack/index files, inconsistent names, corrupt index/pack hashes and configured size excess reject. At most 32 input pairs may be captured from one request workspace; this is a request inventory bound, not a repository pack-count limit. PhysicalLimits supplies pack/index byte ceilings, and the node DiskBudget accounts for the existing private files. Capture builds no decoded-body vector or complete OID inventory. + +NativePackDescriptor's local inspector reuses the same streaming pack checksum/BLAKE3 scan as verify_files. Local descriptors remain inside the capture service with unset manifest digests. Each authenticated upload fills the actual manifest descriptor before the input can escape. Pack/index keys reuse the repository/creating-operation/pack-digest layout. All pairs are checked locally before transfer; a descriptor vector is returned only after every pair upload and a final live-custody check. A failed later upload can leave private artifacts in the retained creating namespace, but cannot return a partially complete input set or publish refs. + +## Cancellation and verification + +The existing pinned-file uploader now serves native files as well as immutable metadata. Every queued open/read owns an InputFile, which retains the cache and exclusive fence. Canceling an observer or upload cannot release the cache reservation while its blocking file work still runs. The final CapturePin explicitly unlocks the exclusive fence before its file and cache owners drop. On Unix, an unrelated child can inherit even a CLOEXEC descriptor until exec; closing only the parent copy would leave a completed capture locked. Explicit unlock happens only after every owned InputFile and queued file worker has drained. Remaining owned file pins still exclude native commands. Native workers retain their shared locks through participating descendants; their drain guard is unchanged. Unlock failure leaves admission and cleanup conservative through the existing lock checks. The staging coordinator separately owns producer execution and completed typed results through single handoff. + +Capture establishes authenticated physical bytes, native index validity, header/count agreement and whole pack/index checksums. It does not establish decoded-object CRC validity, self-contained delta decoding, canonical metadata, graph closure or ref authority. PhysicalVerifier must independently reopen the downloaded pair without alternates, decode every physical ordinal and finish its complete metadata partition. CatalogPreparation then checks canonical identity and typed closure against the bound certified base. CompleteCatalogPush rechecks current authority, policy, refs and catalog CAS while saving the exact native response in that same transaction. + +## Executable evidence and remaining scope + +The composition test submits an actual receive-pack CGI request for both OID formats. It first retains/registers the [original encoded request](durable-push-request.md), then receives through native Git, captures the resulting pair through StagingCoordinator and attaches the [durable native result](durable-native-result.md) with that inventory through the service-owned exact dispatcher. After deleting local request/receive/source caches, it reopens the original request and completion, independently downloads/verifies the native pair, binds staging, assembles the catalog and invokes CompleteCatalogPush. A fresh CatalogReader selects the source from the committed root after preparation scratch has drained. A new native cache downloads that source, and stock Git clones and fscks it. The fresh Cell has no legacy objects/object_edges/object_closure/git_packs tables. The test supplies an authorized fixture owner and orchestrates the steps directly; production HTTP/SSH authentication and orchestration are not implemented by this test. + +Failure checks exercise wrong repository, configured size excess, loose inputs, physical byte corruption, active-worker exclusion and exact repeated upload. A queued-upload cancellation test verifies retained disk/cache ownership and exclusion until the queued worker releases its fence. A deterministic fork test pauses an unrelated child before exec, verifies that another owned input still excludes native admission, then drops the final capture pin and verifies admission succeeds while the unrelated child remains paused. + +The production gateway now uses [owned push preflight](owned-push-preflight.md) to bind encoded request identity before replay, then retain normalized input and parsed gzip/signed/options intent together. Actual signature verification remains native. Connecting that owner to this capture/lifecycle path, producer/reader conversion, production integration of [input checkpoints/adoption](native-input-checkpoint.md) and durable native-result recovery orchestration with a short final ref-plan command interface, continuous frontier/maintenance orchestration, whole-operation file/I/O/OS containment and full-history capacity campaigns remain required. No online provider deletion is authorized by this capture API. The fixture's memory provider and small history do not qualify remote durability or 10,000-engineer throughput. diff --git a/docs/design/native-input-checkpoint.md b/docs/design/native-input-checkpoint.md new file mode 100644 index 0000000..3276447 --- /dev/null +++ b/docs/design/native-input-checkpoint.md @@ -0,0 +1,47 @@ +# Native input checkpoints and adoption + +Creating input custody now has a durable descriptor inventory. NativeInputIndex and NativeInputRoot are aliases of the existing RangeIndex and NodeRef, with NativePackDescriptor as the sealed record. The native-input-index.v1 domain distinguishes these nodes from canonical directory and metadata source nodes. Records reuse the operation/pack-digest SegmentKey and existing artifact codec. Fanout is 128, nodes remain within 64 KiB, the cursor retains one bounded-height path, and the index cache retains at most 64 nodes. Physical object counts are summed inventory, not unique canonical coverage. No OID list or per-object SQL row is added. + +StagingContext::seal_native_inputs consumes a descriptor iterator incrementally. Fresh descriptors must belong to its repository, format and allocated creating namespace, and contain manifest digests. Inserting an identical record deduplicates it; an unequal record at the same key rejects. The factory builds the authenticated index before issuing NativeInputCertificate. It queries live staging authority and reuses the private repository secret with the existing 1,024-byte CertificateEnvelope and a distinct staged-native-inputs.v3 payload domain. The envelope binds tenant/application, exact logical request digest, actor, owner/attempt, creating namespace, format, optional native index, original request and native result roots, adoption source and append predecessor. An empty inventory is represented by an absent root. Descriptor syntax and signed custody do not prove that referenced native pair bytes exist or decode; PhysicalVerifier independently establishes those facts before catalog preparation. + +## Durable facts + +RegisterStagedInputs is command 29 in the fresh registry. It checks current write permission, target, admitted owner, exact current attempt, independent pin, live expiry and MAC. A fresh inventory requires an unbound staging phase; an exact borrowed inventory can also attach after a bound preparation Claim. These checks apply before attaching the checkpoint. The existing catalog_leases row stores the bounded envelope and its BLAKE3 digest together. It now supports the [durable request and native append sequence](durable-push-request.md) through an exact predecessor digest and at most 256 revisions while unbound. Immutable roots are preserved by the private path-copy issuer; arbitrary replacement, partial changes, removal, revision jumps and append during/after Bind reject. Attaching a [durable native result](durable-native-result.md) freezes the descriptor inventory; subsequent exact replay and whole-root adoption remain allowed. The existing anti-REPLACE guard applies. Catalog operations, generation facts, refs and outcomes retain their existing structures. The command returns StagingReply as a custody DTO; it does not grant a preparation floor. Exact SDK replay retains its original receipt; a new registration identity rechecks current authority and phase. The checkpoint domain is v2 without a compatibility adapter. + +CheckStagedInputs is query 30. It requires current write access and the exact actor/logical request, independently matches the retained source pin, verifies envelope digest/MAC and target Cell identity, and rejects expired or missing custody. It can read a previous attempt after Claim because its pin is independent of the current operation. This actor query is not an exhaustive collection inventory: revocation can hide a still-retained checkpoint, and None cannot authorize deletion. A qualified collector must enumerate every relevant retained SQL snapshot and pin under its own complete barrier. + +## Successor ownership + +ReadyStaging::claim prepares the existing ClaimStaging command through the same service-owned Exact dispatcher as Begin/Renew/Bind. Observer cancellation cannot discard that command; uncertainty keeps its exact SDK identity and evidence. A fresh post-command query establishes the deadline. Claim allocates a new attempt and creating namespace while preserving the old independent pin. + +StagingContext::adopt_native_inputs and PreparationSession::adopt_native_inputs require an exact registered source checkpoint for the same repository, logical request digest, actor, format and target. The latter supports recovery after a bound ClaimPreparation. Both reuse the exact immutable input-root reference, validate one authenticated root node and sign a new envelope containing the source token and exact checkpoint digest. Adoption performs no historical leaf scan, index-node copy or native upload; original index and native artifact incarnations remain unchanged. Final registration rechecks the source pin, expiry, digest, MAC, authenticated request context and exact root equality in the same short transaction. Expiry between proof issuance and registration refuses adoption without attaching a destination checkpoint. Once committed, the destination's independent pin retains the adopted inventory; identical re-registration no longer depends on the source pin remaining live. Physical reconstruction must subsequently exhaust the authenticated cursor and independently verify native bodies and canonical coverage. + +Custody of an adopted input retains its entire referenced creating namespace, including the original input-index nodes, while the destination pin/writers remain live, as well as the successor's own namespace. Physical metadata reconstruction reuses the original native descriptor context, so retention cannot be reduced to just the two pair keys while reconstruction is allowed. Reaping an old SQL pin does not release namespaces referenced by a live destination checkpoint. The complete retained-root collector must walk these input indexes, retain borrowed namespaces and unregistered output ownership, and prove writer drain/clock barriers before deleting anything. This increment provides no remote deletion authority. + +## Bound checkpoint command ownership + +After ClaimPreparation has established a live bound PreparationSession (now available through [exact service dispatch](bound-preparation-dispatch.md)), its ready_inputs factory accepts only an adopted checkpoint with the same actor/token/target and format. It checks bounded encoding and retains the exact SDK command 29, its mutation identity, checkpoint digest and shared session. Submit ReadyNativeInputs to the repository's existing PublicationCoordinator; there is no separate checkpoint queue or SQL representation. Admission failure returns that original ready value. This factory does not execute or prove native bodies. + +Bound checkpoints use the foreground class and its existing account/operation quotas, FIFO rotation, maintenance reservation, concurrency limits, cancellation ownership and exact uncertainty recovery. Each checkpoint reserves 8 KiB for retained and transport command copies. Pushes still reserve 8 MiB and compactions 8 KiB; the existing job now carries its private factory's reservation, and byte admission/release uses that value. An uncertain checkpoint blocks another admitted command with the same logical request ID until resolved. Close retains its command and credits; recover remains available after closing. + +PublicationOutcome::Inputs returns RegisteredNativeInputs containing the original committed registration and a separate custody observation. A known commit remains a known commit if fresh authority is revoked, expires or changes through Claim. After receiving the original receipt, query CheckStagedInputs at that receipt, verify the exact checkpoint digest, then freshly query CheckPreparation and match the original bound token, floor and format before updating the shared conservative deadline. Failure fences the shared session and returns a custody error alongside the original committed result. Replaying recorded timestamps cannot establish usable custody. ticket.response refuses checkpoint outcomes; registration cannot acknowledge a Git push or update catalog/refs. + +Exact bound Claim/Renew command dispatch now shares the coordinator. Automatic bound renewal, Claim admission, worker/result ownership, checkpoint serialization and local residence now reuse the [staging lifecycle](bound-preparation-lifecycle.md). Final-publication lifecycle serialization and durable reconstruction after process loss remain required. This dispatcher owns one accepted checkpoint command through outcome recovery; it is not the complete preparation lifecycle or an outbox. Source custody and all current authority checks remain enforced by the final registration transaction. Collection and native writer-drain obligations are unchanged. + +## Physical verification and publication + +CatalogPreparation::begin_retained_pack now bridges authenticated checkpoint custody to independently verified physical input. Its private RetainedNativeInput factory queries the current successor checkpoint, matches the entire NativePackDescriptor in the shared authenticated input index, and checks the current bound attempt and original retention floor. Matching only the pack key cannot authorize a different index incarnation. CatalogIndexes shares the existing bounded input index cache across preparation and reconciliation. The ordinary begin_pack API retains its strict creating-namespace guard. + +The private custody proof binds the exact closure context, physical descriptor and live PreparationSession. Each metadata shard must still satisfy the physical partition and original metadata digest; closure copying uses the authenticated original namespace. SourceRecord retains that original native/metadata incarnation, while newly written source-index nodes use the successor namespace. Reconciliation preserves this distinction and the immutable checkpoint digest over a moving, nonempty base. + +Catalog certificates now use catalog-attestation.v3 and carry the optional input-checkpoint digest in the existing 1,024-byte envelope. The issuer freshly checks custody. Both attestation registration and final catalog/ref publication recheck the digest, MAC, exact attempt/actor/scope/format and pin expiry inside the final transaction, alongside the existing current authority and owner checks. Completed exact/logical recovery still returns its original result before fresh-authority checks. The old v2 certificate domain is rejected as part of the hard cutover. This bridge grants no collection authority or production producer selection. + +## Evidence and remaining integration + +Seven tests cover bounded SHA-1/SHA-256 indexes spanning multiple nodes, a sub-1-KiB envelope for 300 synthetic descriptor records, exact checkpoint replay, immutable/paired SQL fields, conflicting inventories, MAC/foreign-scope/current-permission/stale-attempt refusal, source-pin expiry after proof issuance, bound preparation adoption with the exact original root, and observer-independent Claim recovery after absence, lost replies and worker panic. Synthetic descriptors test the index and custody protocol rather than native-byte validity. The actual owner-restoration test captures a real pack, loses its original caches, restores the Cell with a new owner, replays the original checkpoint receipt, claims/adopts, expires the old pin, independently downloads/decodes the native pair through PhysicalVerifier, then binds, verifies closure and atomically publishes the retained input under the restored owner. After completion the active query returns None; the independent pin and exact original checkpoint receipt remain recoverable. The real receive/publication/cold-clone fixture now persists and reads a checkpoint for both OID formats before verification and atomic publication. + +The staging coordinator now supervises one admitted registration through cancellation and exact uncertainty, with renewal-before-registration and registration-before-Bind ordering; see the [service contract](staging-service-lifecycle.md). Bound registration now shares the publication dispatcher. Production producer wiring, service-owned bound Claim/renewal and durable reconstruction, authenticated request preflight, production native-result recovery orchestration and the final short ref-plan command interface, all producer/reader conversion, retained-root enumeration/collection and isolated restore remain required. Registration is an additional durable command when a checkpoint is selected and must be included in bulk-import/lease/durability budgets. Neither these small fixtures nor the bounded index tests qualify full-history import, provider durability or large-team throughput. + +Six additional custody checks cover both object formats, publication and cold Git clone over moving nonempty bases, absent successor custody, unchanged raw namespace rejection, absent inventory membership, exact index incarnation, signed digest substitution and old-domain refusal, and issuer/final-command revocation, expiry and Claim. The maximum actor/ref/completion/checkpoint certificate remains within 1,024 bytes. These are bounded fixtures rather than full-history or large-team qualification. + +Six bound checkpoint checks cover SHA-1/SHA-256 absent/lost/panicked exact recovery, closed recovery and canceled observers, refused ready-value reuse, original committed receipts with failed fresh custody, absence followed by revocation/expiry/Claim without attaching inventory, real physical retained-pair publication after source-pin expiry, and mixed checkpoint/push actor and byte admission. The existing 300-record bound-adoption test now uses the dispatcher and preserves the exact original root. Synthetic descriptors establish bounded index/custody behavior rather than physical validity or capacity. diff --git a/docs/design/native-process-ownership.md b/docs/design/native-process-ownership.md new file mode 100644 index 0000000..a879179 --- /dev/null +++ b/docs/design/native-process-ownership.md @@ -0,0 +1,52 @@ +# Native process ownership and drain + +The shared GitProcess guard now lives in native_git::process and serves HTTP/SSH transport, isolated pack validation, cache maintenance, native blob extraction, candidate preparation, ref listing and decoded-object batch/history workers. The decoded-object path previously used direct-child kill-on-drop; transport guards previously released owners after sending a process-group kill signal. Signaling a process is not evidence that its work has stopped. + +The guard reuses its existing generic owner: cache generations, input files and account/transfer permits stay together. It does not create another object inventory, compatibility store or publication certificate. The guard now also requires a private permit from the [shared native resource pool](native-resource-admission.md). Actual CPU/RSS/file/process containment and the production packed-catalog cutover remain separate deliverables. + +## Spawn and supported fence + +On Unix, create a private process group and an anonymous socket pair before spawn. Duplicate the child end into a descriptor of at least three, outside the standard streams that spawning replaces. Keep both parent descriptors close-on-exec by default. Clear close-on-exec only for the duplicated child end inside its pre-exec callback, using async-signal-safe fcntl. The cache's existing inherited workspace lock remains independent and continues to fence filesystem cleanup. + +Drop the Command immediately after spawn, including failed spawn and completion-fence setup failure, before dropping the caller's owners. This releases parent copies of inherited descriptors; the running child and its descendants retain their ends. One completion read descriptor remains in the guard. A native/helper process that preserves the inherited end prevents EOF until it exits or closes that end, including after it closes standard streams or escapes the initial process group. + +The child wrapper exposes only stdin/stdout/stderr and read-only PID inspection. The actual Child and wait/try_wait stay private. Callers cannot reap the leader before the guard's drain check. This matters because an unreaped leader's PID cannot be reused for an unrelated process group while cancellation is preparing its signal. + +## Completion and cancellation + +Callers retain their existing total-work/idle deadlines, drain standard output and bounded diagnostics, then invoke GitProcess.wait. On Unix, wait for EOF on the private completion descriptor before reaping the leader. Unexpected completion-channel data or an I/O error rejects completion. A closed stdout/stderr pair alone cannot complete the operation while a participating descendant is still alive. Once the leader is reaped, disable later group signals. + +On cancellation/error/drop, signal the original private group while its leader is still unreaped. Transfer the Child, completion descriptor and generic owner into a service-owned reaper. Reap the leader and await inherited-end EOF independently; only successful completion of both releases the owner. The reaper is independent of the requesting future. A helper that escapes the process group still retains its owner while it holds the descriptor; the guard does not pretend that the initial kill signal terminated it. + +A successful completed wait uses an immediate drop path without spawning a reaper task. Other cases use at most 512 shared reaper slots. Missing runtime, saturation, failed drain or runtime shutdown conservatively quarantine the owner through restart. These conditions emit an error and do not return account/disk/cache credits for potentially live work. The node supervisor now closes and drains the shared native pool before Cell shutdown, workspace release and heartbeat withdrawal. Quarantined claims keep that boundary pending; quarantine is the failure behavior, not an operational substitute for drain. Escaped or stuck workers can consume admission indefinitely and must be visible to operations. + +The guard preserves native exit-status handling. Group/fence completion does not certify object identity, graph closure, current authorization or durable publication; those still use the existing verification and fenced Cell APIs. The ref-list helper now uses the same guard, a one-hour work deadline and bounded stderr; its existing whole-ref-list collection remains and needs the namespace/resource profile before cutover. + +```mermaid +flowchart LR + Spawn[Spawn with owner] --> Work[Private group and inherited end] + Work --> EOF[Await completion EOF] + EOF --> Reap[Reap leader] + Reap --> Drop[Drop guard and release owner] + Work --> Cancel[Cancellation or error] + Cancel --> Signal[Signal original unreaped group] + Signal --> Supervise[Bounded independent reaper] + Supervise --> Both[Leader reaped and inherited EOF] + Both --> Drop + Supervise --> Unknown[Failed drain or unavailable runtime] + Unknown --> Hold[Retain owner through restart] +``` + +## Required containment and qualification + +This descriptor fence covers descendants that preserve the inherited completion end. A process can explicitly close it while continuing to run. The selected Git version, allowed helpers and command profile must qualify descriptor preservation. The fence is not a universal process-membership oracle, a hard RSS limit or authority to delete remote artifacts. Production OS/container containment must bound memory, CPU, process/file counts and escaped descendants, and its drain must be qualified before resource release or remote reclamation depends on stronger guarantees. Complete retained-root inventory and renewable serving/backup pins remain required for deletion. + +On non-Unix platforms the guard retains owners through direct-child reaping, but has no Unix process-group or inherited-descriptor proof. Equivalent job containment and descendant drain remain unimplemented; the packed-storage release cannot claim that platform's process-tree safety from these checks. No production format marker or registry is selected by this change. + +Decoded-object actors use the shared process guard and the existing cache workspace fence. They and ref-list helpers now require explicit native-resource permits from the owning node scope. A () generic owner still supplies no account admission; the native permit independently retains the node claims through drain. Profile qualification, fair preparation admission and hard OS limits remain required. Physical input, metadata, closure, edge spools and detached SQL/I/O jobs keep their own admitted ownership. + +## Evidence and remaining gates + +Unix fault fixtures cover a leader that exits after a helper closes all standard streams, cancellation with a helper that creates a new session, and a separately spawned daemon-style fixture with closed stdin/stdout. They confirm pending completion, retained transfer credit, safe descriptor placement and release after actual inherited-end drain. The escaped-session helper uses the same test executable and a bounded lifetime; it introduces no Python dependency or mutation of the parent test process's standard descriptors/environment. + +Existing HTTP backpressure/disconnect/deadline/spawn-failure checks, decoded-object cancellation/stream-integrity checks, native SHA-1/SHA-256 isolated verification, ancestry and maintenance fixtures use the new guard. Production network and owner-loss/restore suites remain required regression evidence. These checks establish this ownership protocol, not 10,000-engineer throughput, hard native memory bounds, full input adoption or completed production hard cutover. diff --git a/docs/design/native-resource-admission.md b/docs/design/native-resource-admission.md new file mode 100644 index 0000000..a8a4b35 --- /dev/null +++ b/docs/design/native-resource-admission.md @@ -0,0 +1,50 @@ +# Native resource admission + +Every shared GitProcess spawn now requires a private NativePermit. One NativeResources pool belongs to the node; the server constructs it from required native_limits before opening the workspace or probing storage. RepositoryManager clones that pool into every gateway. Gateways pass the same foreground scope through their pack readers, bare caches, cache generations, decoded-object/history readers, candidates, HTTP/SSH transports and ref listing. PhysicalVerifier and CanonicalVerifier require an explicit scope from their preparation service. There is no implicit unadmitted production constructor or process-global path registry. + +This reuses GitProcess ownership, cache generations and the existing bounded staging/transfer/publication services. NativeCapacity is a fixed four-field vector: process slots, CPU admission units, memory bytes and descriptor claims. A single short synchronous mutex admits the complete vector or returns typed exhaustion without changing any dimension. No partial semaphore acquisition, retained native waiting queue or per-object admission inventory is created. Independent foreground and maintenance counters have disjoint caps; maintenance cannot consume foreground's share and foreground cannot consume maintenance's reservation. Unused reserved capacity is not borrowed. + +## Configuration and profiles + +native_limits is a required node JSON and ServerConfig field. Unknown profile fields reject, and unsigned integer decoding rejects negative or out-of-range values. Startup validates profile minima, one process slot per native guard, checked aggregate arithmetic and room for a simultaneous read/pack pipeline in each share. A library service creates NativeResources once, passes cloned scopes and chooses the work kind at each launch. Creating a separate pool for every operation would defeat node admission and is outside this API's service contract. + +| Dimension | Default node total | Maintenance reserved | Read profile | Pack profile | +|---|---:|---:|---:|---:| +| Process slots | 32 | 2 | 1 | 1 | +| CPU admission units | 48 | 8 | 1 | 4 | +| Memory claims | 16 GiB | 2 GiB | 256 MiB | 1 GiB | +| Descriptor claims | 2,048 | 128 | 32 | 64 | + +The foreground cap is total minus reservation in every dimension. Defaults allow ten foreground pack claims before CPU admission exhausts; a concurrent reader consumes its own units. These defaults are initial estimates, not measurements, actual RSS limits or a 10,000-engineer throughput result. Provision against the selected hardware, workload and OS containment. Configured read/pack estimates may increase. Minimum read claims are one process slot, one CPU unit, 128 MiB and 32 descriptors; minimum pack claims are one process slot, four CPU units, 512 MiB and 64 descriptors. Minima are conservative admission floors, not evidence that every repository fits those amounts. + +Pack work includes HTTP/SSH Git trees, generated candidates and index/pack validation. Persistent decoded readers, history/ref listing and native blob extraction charge read claims. Read work can still have a large native heap on full histories. The sanitized Git policy limits pack search to two threads and now explicitly supplies --threads=2 to isolated and maintenance index-pack verification. Pack caches/mappings keep their existing bounds. Helpers, traversal, decoded object heaps and inherited descendants need qualification and OS enforcement; Git cache configuration alone cannot bound them. See the [Git index-pack thread contract](https://git-scm.com/docs/git-index-pack) and [Git pack.threads configuration](https://git-scm.com/docs/git-config#Documentation/git-config.txt-packthreads). + +The small evaluation scripts share an explicit fixture profile from scripts/native_limits.py: two GiB total native memory claims, a 768 MiB maintenance reservation, 128 MiB readers and 512 MiB pack workers. They keep room for a 1.5 GiB tmpfs and server in the existing four-GiB container. The example production JSON uses the larger default profile. No profile is silently selected for a missing native_limits field. This is a configuration hard cutover; all checked-in node config producers and the deployment example supply the field. Fleet readiness and large-repository reports record the selected limits. Source fingerprinting now includes the workspace crates and manifests rather than the removed src directory, including the new native policy. The repository release source digest includes native admission, command policy, process ownership and the Unix completion fence. + +scripts/check_container.py reads the configured vectors alongside the kernel's effective cgroup values. It rejects invalid shares, below-minimum profiles, descriptor headroom that ignores native claims, or native claims plus admitted tmpfs plus 256 MiB of server headroom exceeding memory.max. That minimum server headroom still needs measurement. CPU units are admission weights, not a CPU bandwidth guarantee; the existing fixture cgroup remains limited to two CPUs. Vector checks do not prove process containment or actual peak memory. + +## Lifetime and failures + +The process guard's existing generic cache/input/account owner and native permit travel together. Their field order drops generic ownership before native credits return. Successful guard completion/drop releases the claim after inherited completion-descriptor drain and leader reaping. Cancellation, timeout or error transfers it to the existing bounded independent reaper. Uncertain, saturated or shutdown cleanup quarantines the entire owner and native claim through restart; signaling the process group cannot release capacity by itself. See the [native ownership protocol](native-process-ownership.md) for the participating-descendant and platform limits. + +Cache maintenance uses a maintenance scope for its simultaneous listing and packing workers. Both guards are explicitly dropped after successful drain before acquiring a validation claim. The returned cache generation preserves its original scope, so a repack cannot move foreground serving into the maintenance quota. The isolated physical-file binding job also retains a read claim in its blocking closure; canceling its observer cannot free that claim while native-index authentication is still running. Physical verification then charges pack validation and its persistent decoded reader separately. + +Native exhaustion has a typed I/O source and becomes HTTP 503 through gateway I/O, Git HTTP, cache I/O and decoded-object wrappers. A generic WouldBlock error is not reclassified as capacity. Callers receive failure before spawn rather than creating an unbounded wait. Per-account transfer/staging admission remains separate; this node pool does not by itself implement native per-account vector shares or a fair preparation waiting scheduler. SSH retains its existing failed-command/error behavior. NativeUsage provides bounded current class counters, without repository or OID label cardinality. The node supervisor now closes and drains this same pool before releasing Cell authority, workspace exclusion or lease renewal. + +A poisoned admission mutex rejects new claims. Dropping an existing claim against poisoned state leaves its counter charged. Counter underflow permanently faults drain proof, even if a corrupted counter is already zero. Admission overflow/denial never changes counters, and there is no public way to clone or forge a permit. Failed spawn releases parent inherited descriptors before the native and generic owners drop, preserving the existing cache cleanup fence. + +## Shutdown boundary + +NativeResources::close permanently rejects admission through every existing or future scope of that pool. Closure and admission use the same short mutex, so a concurrent admission either retains a complete pre-close claim or fails. A previously admitted owner can still finish its work; closure does not revoke claims. Closed admission has a typed source and uses the same HTTP 503 classification as capacity exhaustion. + +NativeResources::drain closes admission and waits for both complete class vectors to reach zero under healthy accounting. It registers notification before inspecting state; the final release wakes all registered observers, and a late observer reads zero without needing an old notification. Canceling any observer leaves closure and claims unchanged. Poison, underflow and quarantined native owners cannot return a successful drain. No timeout converts uncertain native cleanup into release permission. + +RunningServer owns the same pool cloned into RepositoryManager. Normal shutdown closes native admission, cancels maintenance/ingress, joins HTTP/SSH and tracked work, drains native owners, then shuts down Cellule. Only confirmed Cell drain clears workspace exclusion; after these drains the supervisor stops heartbeat renewal and withdraws the advertisement. Startup failure after enrollment uses the same tracked-work/native-before-Cell ordering. Detached native reapers and admitted blocking verification retain the live supervisor, workspace and renewal task while drain is pending. Canceling the caller's shutdown wait does not abort this supervisor. A quarantined claim can keep shutdown pending until process restart; deployment supervisors must treat this as unproven drain. Provider or lease failure can still fence the node; local drain cannot promise successful renewal against unavailable authority storage. The inherited-descriptor and OS containment limits remain unchanged. + +## Evidence and remaining work + +Resource tests exercise independent exhaustion of all four dimensions, unchanged counters after repeated denial, disjoint class shares, simultaneous admission through cloned scopes, invalid/overflowing configuration and poisoned-state quarantine. Existing native descendant fixtures now assert the complete resource claim remains charged after closed standard streams and escaped-session cancellation, and returns after actual drain. Failed-spawn coverage asserts no leaked native claim. Typed HTTP classification distinguishes native exhaustion from unrelated WouldBlock errors. Two backends sharing one pool reject launch while it is exhausted and advertise stock Git again after release. + +Shutdown checks cover permanent closure across cloned scopes, admission/closure races, simultaneous class owners, multiple observers, canceled observers, final-release registration races and refusal to prove drain under poisoned or underflowed accounting. The native descendant fixtures wait on the pool's drain boundary after releasing escaped/closed-standard-stream helpers. A real server test holds foreground and maintenance claims past one node lease, checks live advertisement and workspace exclusion before Cell shutdown, cancels a shutdown wait, then verifies Cell drain, advertisement withdrawal and workspace reuse after both owners release. + +These checks establish accounting and ownership, not full-history resource estimates or process isolation. Production preparation services must use the same node pool and retain account admission through all native work. OS/container limits must cover escaped helpers, CPU, memory, files and processes; non-Unix job containment and qualifying helpers that may close inherited completion descriptors remain open. Full-history imports, persistent native indexes/graphs/bitmaps, staging adoption, continuous maintenance, retained-root collection, isolated restore, deployment-format selection and mixed-load capacity remain required parts of the hard cutover. diff --git a/docs/design/outcome-only-completion.md b/docs/design/outcome-only-completion.md new file mode 100644 index 0000000..38197e0 --- /dev/null +++ b/docs/design/outcome-only-completion.md @@ -0,0 +1,27 @@ +# Outcome-only push completion + +Refused and empty native pushes must save their exact response without rebuilding or uploading a catalog. PreparationSession owns the existing authoritative lease capability separately from PreparationBaseResolver's catalog reader. Both share the same monotonic deadline and irreversible renewal-failure fence. Opening a session validates the canonical repository target and fresh CheckPreparation result; it does not accept a caller-supplied trusted token alone. + +## API and stored structures + +After BeginPreparation, open PreparationSession at the committed receipt. Call push_outcome to prepare a CatalogPushCompletion, complete_outcome for direct final invocation, or ready_outcome to retain the exact SDK command in PublicationCoordinator. Native ref-success reports and any Some(plan), including an empty plan, reject on this path. Checked ref plans continue through PreparedCatalog and its membership/ancestry proof. + +The session factory takes no artifact loader, provider, scratch path, DiskBudget or native scope. It queries current preparation authority and the repository secret. It reuses the 1 KiB certificate envelope, existing lease/token and GenerationFact structures, command 19, pushes/response/options/signed-certificate tables and foreground dispatch queues. No new authoritative SQL row or storage layout is introduced. The envelope body is bounded at 960 bytes. The ordinary path remains Begin plus final completion, with read-only issuance queries. + +OutcomeCertificate uses the purpose domain canopy.push-outcome.v1. It binds tenant/application, repository operation and attempt, admitted owner fence, actor, object format, original immutable generation floor, and the existing payload digest covering response identity/status/ordered headers/body/options and signed-witness bytes. The typed codecs reject cross-purpose catalog certificates. Native signed witnesses remain opaque, verified gateway outputs; the factory checks their target/request/actor context. + +## Final authority and response recovery + +CompleteCatalogPush authenticates the proof and exact payload before using it. A new outcome checks the actual owner fence, exact operation/pin/expiry, current write ACL and immutable floor. It saves the response in the same transaction and removes the operation, while the independent retention pin remains until expiry. It increments no catalog generation and changes no refs. A concurrent catalog advance does not invalidate a response-only proof, because it asserts no new object visibility or ref identity. + +An already completed identical payload replays its original result before live-owner checks; this is recovery authority, not new write authority. Exact SDK replay preserves the original committed receipt. A fresh logical completion identity returns the stored logical result. A different actor, request digest or completion payload cannot replace the result. New unfinished proofs from a prior owner reject after actual restoration. + +The same admitted invocation can load its own exact response at the committed receipt, even if permissions changed while it ran. This grants no repository read capability. Reconnect/preflight uses replay_push_response and CheckCompletedPush, requiring current read ACL and exact actor/request context before loading stored bytes. + +After admission, there is no local lease timeout around the final command. Unknown acceptance retains the exact SDK identity, command bytes, session and queue reservation. Observer cancellation cannot drop them or synthesize a refusal. Recovery uses the existing authoritative resolve/absence/execute path. Terminal ownership releases before queue credits. + +## Scope and validation + +Tests cover SHA-1/SHA-256 refusal, native error and empty response persistence without catalog artifacts, a moving floor with an unavailable old artifact store, payload tampering and cross-purpose rejection, current write/read revocation, expiry, canceled observers, absent/lost-reply/panicked dispatch recovery, and genuine owner restoration with original completed receipts and stale unfinished rejection. Trusted fixtures supply native signed witnesses and immutable roots; these tests do not qualify production signatures or full-history throughput. + +The fresh schema is still not selected by production handlers. HTTP/SSH producers must perform completed-request preflight before Begin, route failed/empty requests to this session API, retain exact uncertainty after admission and use the same atomic completion for ref publication. Production reader conversion, service reconstruction, large inline payload roots, full-history resource bounds and mixed-load qualification remain required. diff --git a/docs/design/owned-push-preflight.md b/docs/design/owned-push-preflight.md new file mode 100644 index 0000000..da37366 --- /dev/null +++ b/docs/design/owned-push-preflight.md @@ -0,0 +1,32 @@ +# Owned push request preflight + +The production Git gateway now retains one owned handoff from authenticated encoded input to normalized native input and parsed command intent. It reuses GitInput's disk-accounted spool, the existing BeginRequest identity and the existing command parser. The push handler consumes PushPreflight rather than accepting independent input, actor, operation and digest arguments. This is a prerequisite for routing production receive-pack into the packed lifecycle; the selected production schema and publication path still require conversion. + +## Authentication and identity + +The gateway checks current repository access, an account identity and transport authentication before spooling a push. EncodedPush then validates POST receive-pack, the actor component and the canonical target derived from tenant, application and repository UUID. Its authentication flag is an internal transport assertion after those gateway checks, not an independently verifiable authorization proof. Authorization must still be rechecked at final publication. + +The request digest is BLAKE3 with domain `canopy.git.push-request.v3\0`. Fields appear in this order: + +1. Tenant, application, namespace, partition, repository UUID, logical operation ID and actor, each prefixed with its byte length as a little-endian u64. +2. Four one-byte values: object-format byte width (20 or 32), protocol-v2 flag, content-type-presence flag and gzip flag. +3. Method, path, query and content type (empty if absent), each prefixed with a little-endian u64 byte length. +4. The little-endian u64 encoded body length followed by the exact encoded body bytes. + +GitInput hashes the body with a fixed 64 KiB buffer and rewinds the spool. It computes the unkeyed artifact digest in that same scan for [durable request retention](durable-push-request.md); the authenticated uploader independently verifies the bytes. Hashing precedes decoding and is independent of upload chunk boundaries. Distinct gzip representations have distinct identities even when they expand to identical commands. This replaces the earlier v2 request domain directly; there is no legacy-digest replay adapter. Old persisted identities are not supported across the required fresh-data hard cutover. + +The immutable BeginRequest is available before decoding so completed replay can return the original saved response without allocating another decoded spool or invoking native Git. Normalization consumes EncodedPush once and carries the same identity into PushPreflight. BeginRequest's lease duration remains a preparation policy value, outside request identity. + +## Native input and intent + +Gzip decoding reuses GitInput's admitted multi-member decoder and cancellation ownership. Blocking work retains its input/output files, disk reservations and transfer admission through drain. The command parser inspects only the bounded packet prefix and optional push-option group, preserving the entire normalized spool and rewinding it for native Git. It retains the existing 40 MiB command-prefix, 32 KiB option-prefix and 100,000-update ceilings. These bounds do not establish a whole-operation RSS or full-history latency qualification. + +Old and new OIDs must match the repository format before zero IDs become create/delete intent. The same format check applies to signed certificate commands and shallow OIDs. Parsed ref versions remain placeholders for intent; current durable ref versions must supply the final CAS plan. Parsed certificate bytes are not a signature witness. Existing native signature/nonce verification and registered-key authorization still produce the private verified certificate. Signed options must match the separate option group; unsupported options retain the existing durable refusal path. + +SSH's boundary detector still parses enough packets to locate a complete request before entering the common gateway. The common preflight performs normalization and policy parsing once per unreplayed request. Other media types and command limits continue through the existing native refusal path. + +## Evidence and required continuation + +Five focused tests cover context/actor/operation/format/metadata/body identity changes, both-format gzip normalization with immutable identity and spool rewind, mismatched signed/unsigned/shallow formats, signed option preservation/mismatch, invalid scope/authentication and invalid or oversized gzip with admission release. The production smart-HTTP replay regression reserves all disk except the encoded request size, proving a completed gzip retry can return its original response without expansion. Changing only gzip header metadata under the completed operation ID must conflict without changing refs. + +The [durable request API](durable-push-request.md) now checkpoints the original encoded input before native receive and appends native inventory under exact predecessor CAS. Reopening reconstructs normalized input and parsed intent with the same identity. The [durable native-result API](durable-native-result.md) now retains the versioned plan and exact response/options/signature witness in the same checkpoint, and reopens them under fresh custody and current scoped signer authority. Production process/owner-loss orchestration remains required. Plans exceeding the 4 MiB final-command envelope need a bounded root reference rather than copied inline ref vectors. HTTP/SSH packed producer orchestration, every other producer and canonical reader, fresh-schema selection, containment and full mixed-load qualification remain mandatory. This preflight adds no alternate publication format or remote deletion authority. diff --git a/docs/design/packed-repository-schema.sql b/docs/design/packed-repository-schema.sql new file mode 100644 index 0000000..dda0314 --- /dev/null +++ b/docs/design/packed-repository-schema.sql @@ -0,0 +1,143 @@ +-- Proposed fresh-format storage schema. Not a migration and not loaded by Canopy. +-- Earlier per-object Cell design fixture. Superseded for the large-team release +-- by docs/large-team-scalability.md; do not install this as that release schema. +-- Compose with the unchanged product tables using check_packed_repository_schema.py. +PRAGMA foreign_keys = ON; + +CREATE TABLE pack_operations ( + id BLOB PRIMARY KEY CHECK(length(id) = 16), + kind TEXT NOT NULL CHECK(kind IN ('ingest', 'compact')), + incarnation BLOB NOT NULL CHECK(length(incarnation) = 16), + owner_epoch INTEGER NOT NULL CHECK(owner_epoch >= 0), + attempt INTEGER NOT NULL CHECK(attempt > 0), + state TEXT NOT NULL CHECK(state IN ('open', 'ready', 'complete')), + created_ms INTEGER NOT NULL CHECK(created_ms >= 0), + updated_ms INTEGER NOT NULL CHECK(updated_ms >= created_ms) +) WITHOUT ROWID; + +CREATE INDEX pack_operations_by_state ON pack_operations(state, kind, id); + +CREATE TABLE packs ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + operation_id BLOB NOT NULL REFERENCES pack_operations(id), + digest BLOB NOT NULL CHECK(length(digest) = 32), + object_format TEXT NOT NULL CHECK(object_format IN ('sha1', 'sha256')), + git_checksum BLOB NOT NULL, + size INTEGER NOT NULL CHECK(size BETWEEN 1 AND 549755813888), + manifest_digest BLOB NOT NULL CHECK(length(manifest_digest) = 32), + index_size INTEGER NOT NULL CHECK(index_size BETWEEN 1 AND 549755813888), + index_digest BLOB NOT NULL CHECK(length(index_digest) = 32), + index_manifest_digest BLOB NOT NULL CHECK(length(index_manifest_digest) = 32), + object_count INTEGER NOT NULL CHECK(object_count BETWEEN 1 AND 4294967295), + inventory_digest BLOB NOT NULL CHECK(length(inventory_digest) = 32), + staged_count INTEGER NOT NULL DEFAULT 0 CHECK(staged_count >= 0 AND staged_count <= object_count), + staged_digest BLOB NOT NULL CHECK(length(staged_digest) = 32), + last_oid BLOB, + sealed_generation INTEGER UNIQUE CHECK(sealed_generation IS NULL OR sealed_generation > 0), + state TEXT NOT NULL DEFAULT 'staging' + CHECK(state IN ('staging', 'sealed', 'retired', 'deleting', 'deleted')), + CHECK(length(git_checksum) = CASE object_format WHEN 'sha1' THEN 20 ELSE 32 END), + CHECK(last_oid IS NULL OR length(last_oid) = length(git_checksum)), + CHECK((staged_count = 0) = (last_oid IS NULL)), + CHECK((state = 'staging') = (sealed_generation IS NULL)), + CHECK(state = 'staging' OR (staged_count = object_count AND staged_digest = inventory_digest)) +); +CREATE UNIQUE INDEX live_pack_digest ON packs(digest) WHERE state != 'deleted'; +CREATE INDEX packs_by_operation ON packs(operation_id, id); +CREATE INDEX packs_by_state ON packs(state, id); +CREATE INDEX packs_by_generation ON packs(sealed_generation, id); + +CREATE TABLE pack_operation_inputs ( + operation_id BLOB NOT NULL REFERENCES pack_operations(id), + pack_id INTEGER NOT NULL REFERENCES packs(id), + PRIMARY KEY(operation_id, pack_id) +) WITHOUT ROWID; +CREATE INDEX pack_inputs_by_pack ON pack_operation_inputs(pack_id, operation_id); + +CREATE TABLE objects ( + sequence INTEGER PRIMARY KEY AUTOINCREMENT, + oid BLOB NOT NULL UNIQUE CHECK(length(oid) IN (20, 32)), + kind TEXT NOT NULL CHECK(kind IN ('blob', 'tree', 'commit', 'tag')), + size INTEGER NOT NULL CHECK(size >= 0), + digest BLOB NOT NULL CHECK(length(digest) = 32), + pack_id INTEGER NOT NULL REFERENCES packs(id), + location_version INTEGER NOT NULL DEFAULT 1 CHECK(location_version > 0), + edge_count INTEGER NOT NULL CHECK(edge_count >= 0), + edge_digest BLOB NOT NULL CHECK(length(edge_digest) = 32), + CHECK(kind != 'blob' OR edge_count = 0) +); +CREATE INDEX objects_by_pack ON objects(pack_id, sequence); + +CREATE TABLE object_edges ( + parent BLOB NOT NULL REFERENCES objects(oid), + child BLOB NOT NULL REFERENCES objects(oid), + expected_kind TEXT NOT NULL CHECK(expected_kind IN ('blob', 'tree', 'commit', 'tag')), + waiting INTEGER NOT NULL CHECK(waiting IN (0, 1)), + PRIMARY KEY(parent, child) +) WITHOUT ROWID; +CREATE INDEX object_edges_by_child ON object_edges(child, waiting, parent); + +CREATE TABLE object_closure ( + oid BLOB PRIMARY KEY REFERENCES objects(oid) CHECK(length(oid) IN (20, 32)) +) WITHOUT ROWID; + +-- Keep a certified child here until all waiting reverse edges are propagated. +CREATE TABLE object_pending ( + sequence INTEGER PRIMARY KEY REFERENCES objects(sequence), + oid BLOB NOT NULL UNIQUE REFERENCES objects(oid), + received_edges INTEGER NOT NULL DEFAULT 0 CHECK(received_edges >= 0), + edge_digest BLOB NOT NULL CHECK(length(edge_digest) = 32), + last_child BLOB CHECK(last_child IS NULL OR length(last_child) IN (20, 32)), + remaining_children INTEGER NOT NULL DEFAULT 0 CHECK(remaining_children >= 0), + edges_complete INTEGER NOT NULL DEFAULT 0 CHECK(edges_complete IN (0, 1)), + CHECK((received_edges = 0) = (last_child IS NULL)), + CHECK(remaining_children <= received_edges) +); +CREATE INDEX pending_ready ON object_pending(edges_complete, remaining_children, sequence); + +CREATE TABLE refs ( + name TEXT PRIMARY KEY, + oid BLOB CHECK(oid IS NULL OR length(oid) IN (20, 32)), + version INTEGER NOT NULL CHECK(version > 0) +) WITHOUT ROWID; +CREATE INDEX refs_by_oid ON refs(oid) WHERE oid IS NOT NULL; + +CREATE TABLE ref_generation ( + singleton INTEGER PRIMARY KEY CHECK(singleton = 1), + generation INTEGER NOT NULL CHECK(typeof(generation) = 'integer' AND generation >= 0), + pack_generation INTEGER NOT NULL DEFAULT 0 + CHECK(typeof(pack_generation) = 'integer' AND pack_generation >= 0), + default_branch TEXT NOT NULL, + visibility TEXT NOT NULL DEFAULT 'private' CHECK(visibility IN ('private', 'public')) +) WITHOUT ROWID; +INSERT INTO ref_generation(singleton, generation, default_branch) + VALUES (1, 0, 'refs/heads/main'); + +-- Identity and format protection belongs in SQL as well as typed commands. +-- repository_identity is supplied by the unchanged product schema. +CREATE TRIGGER packs_format BEFORE INSERT ON packs BEGIN + SELECT CASE WHEN NOT EXISTS ( + SELECT 1 FROM repository_identity WHERE singleton = 1 AND object_format = NEW.object_format + ) THEN RAISE(ABORT, 'repository object format mismatch') END; +END; +CREATE TRIGGER objects_format BEFORE INSERT ON objects BEGIN + SELECT CASE WHEN NOT EXISTS ( + SELECT 1 FROM packs p JOIN repository_identity r ON r.singleton = 1 + WHERE p.id = NEW.pack_id AND p.state = 'staging' + AND p.object_format = r.object_format + AND length(NEW.oid) = CASE r.object_format WHEN 'sha1' THEN 20 ELSE 32 END + ) THEN RAISE(ABORT, 'invalid object placement') END; +END; +CREATE TRIGGER objects_identity BEFORE UPDATE OF sequence, oid, kind, size, digest, edge_count, edge_digest ON objects BEGIN + SELECT RAISE(ABORT, 'canonical object identity is immutable'); +END; +CREATE TRIGGER objects_move BEFORE UPDATE OF pack_id, location_version ON objects BEGIN + SELECT CASE WHEN NEW.location_version != OLD.location_version + 1 OR NOT EXISTS ( + SELECT 1 FROM packs WHERE id = NEW.pack_id AND state = 'sealed' + ) THEN RAISE(ABORT, 'invalid object location switch') END; +END; +CREATE TRIGGER packs_retire BEFORE UPDATE OF state ON packs +WHEN NEW.state IN ('retired', 'deleting', 'deleted') BEGIN + SELECT CASE WHEN EXISTS (SELECT 1 FROM objects WHERE pack_id = OLD.id) + THEN RAISE(ABORT, 'pack still owns objects') END; +END; diff --git a/docs/design/paged-ref-policy-guards.md b/docs/design/paged-ref-policy-guards.md new file mode 100644 index 0000000..92e9956 --- /dev/null +++ b/docs/design/paged-ref-policy-guards.md @@ -0,0 +1,76 @@ +# Paged ref policy guards + +The fresh packed-publication module prepares current direct-push policy predicates in bounded transactions. This is a prerequisite for short atomic catalog/ref/native-outcome publication. The final publisher and production hard cutover remain required. A guard, a page receipt, a conditional snapshot or a decoded certificate alone grants no publication authority. + +## Intent and custody + +`PreparedCatalog::ref_policy_preparation` owns the original `PushPlan` and catalog-verified ancestry evidence. It captures the current policy configuration epoch before verification. The intent contains a fresh UUID, update count, canonical plan digest and evidence digest. Its persisted scope also binds the existing preparation token, actor and object format. The token includes the actual owner fence and creating artifact namespace. + +The guard is independent of the selected catalog/ref generation. Unrelated catalog advancement can therefore leave policy readiness intact. A rebased prepared catalog must still verify membership/ancestry, the original plan/evidence digests and all conditional ref expectations against its own selected root. A catalog certificate binds that exact selected generation and proposal separately. + +`RefPolicyPreparation::page` copies at most 128 updates and encodes at most 256 KiB including certificate and framing. It counts encoded update bytes before copying, so long ref names reduce page length. Each page remaps ancestry bits from the original offset, including offsets within a byte. The existing catalog MAC uses a separate page binding over intent, offset, exact page plan and ancestry; another certificate purpose cannot register a page. + +## Transactional registration + +Command 33, `RegisterRefPolicyPage`, checks shape and purpose MAC, scoped repository, actual admitted owner fence, current write access, format, live matching operation, expiry and independent retention pin. It checks that the certificate's selected generation remains an authenticated retained fact and that the original floor/input custody still match. The selected generation need not be current; registration does not publish roots. + +Pages extend a contiguous cursor. A new guard starts at offset zero. A signed future page, a partially overlapping page, a changed scope or an invalid guard refuses before writes. An already covered exact issued page returns fresh current progress without advancing the cursor or adding watches. The original SDK mutation identity also preserves its exact receipt; that receipt may describe historical readiness and must not authorize later publication. + +The command checks enabled branch rules with MAC-verified ancestry and current check-context versions/reporters. Direct pushes cannot bypass a required pull request. It installs dependencies for exactly the newest passing attempt of every required context. Watch identity is `(guard, OID, context, context version, run number)`; identical dependencies within/across pages are deduplicated. + +All policy checks, dependency selection, capacity checks, watch installation, budget accounting and cursor advancement execute in one Repository Cell transaction. There is no observation gap between checking a run and watching it. Every rejection precedes the first write. A later SQL failure aborts the guard, watches, scalar budget and cursor together at the existing Cell durability boundary. + +## Freshness without a global check-report epoch + +Branch-rule, required-context and check-context insert/update/delete operations advance one monotonic configuration epoch. These changes are rare relative to ordinary CI reports. Readiness requires exact equality with the captured epoch; returning a rule/context to its old values does not restore an old guard. + +Ordinary run reports invalidate exact indexed dependencies: + +- A newer attempt of the same OID/context/version invalidates an older watched attempt even when the new run is queued or failing. +- Updating or deleting a watched run invalidates it, including a successful-to-successful edit, reporter change or move to another tuple. +- Updating a different run into the watched tuple invalidates whenever its run number is at least the watched number. +- `INSERT OR REPLACE` observes either existing unique key before insertion. This catches replacement that moves the run's OID/context/version even when SQLite suppresses replacement DELETE triggers. +- Reports for another OID/context version and updates/deletes of older attempts do not invalidate the newest watched dependency. + +The secondary watch index begins with `(OID, context, context version, run number)`. Guard-local lookup and cleanup use the existing watch primary key. Reports do not scan every page or advance a global report epoch. + +Query 34, `CheckRefPolicyGuard`, rechecks current write access, persisted repository identity, matching live operation/expiry and its pin, scope and configuration epoch. `ready` requires a valid guard and cursor equal to the original count. A read query is advisory and cannot establish a new write's actual owner fence. The final admitted publisher must check the live guard and epoch in the same transaction as its current authority, root CAS and outcome writes. + +## Conditional ref root + +`PreparedCatalog::guarded_ref_snapshot` accepts only a private ready guard from the matching preparation. It re-verifies the exact original plan against its own prepared catalog, requires identical catalog evidence, prepares all conditional ref changes against its selected immutable snapshot and performs a fresh readiness query before issuing a separate root-purpose MAC. The existing coalesced `RefStateIndex` transition, `RefStateSnapshotRoot` and bounded catalog certificate are reused. Missing immutable ref state never falls back to SQL refs or an empty tree. + +The returned `RefRootPublicationProof` contains the certificate, intent and ref snapshot descriptor; it does not carry the entire ref plan. It is conditional transport, not the final root/outcome command. The inline SQL ref publisher refuses this purpose. Native response custody must still be linked to the original plan and included in the atomic final command. + +Final completion preparation now reuses the existing immutable native-result metadata and artifact descriptors through [typed immutable outcomes](immutable-push-outcomes.md). Its private factory authenticates registered custody and compares that result's original plan with the guard before signing the catalog/ref/result binding. Completed outcome lookup must derive the selected result from the durable actor/logical-operation/request identity, never a caller root. Completed response reads and retention must preserve exact response, plan, options and signed-body artifacts without requiring the original wire request/body forever. Private input checkpoints and their independent pins continue to retain that input chain while preparation or uncertainty needs it. The same metadata bytes can therefore have distinct typed private-input and completed-outcome traversal obligations; these obligations and final atomic publication still need implementation and fault qualification. + +The private completion factory also prepares deterministic all-ref rejection response descriptors outside the final transaction, bound to the same exact original native report/plan and signed custody. Current policy/ACL or signed-certificate-replay refusal must atomically select the appropriate durable rejection, rather than retain a cached successful response or assemble an unbounded report inside the Cell. A changed catalog CAS must preserve the existing reconciliation/retry distinction; it cannot publish an older prepared root. Completed replay must return the originally selected outcome even when later checks or owners change. + +## Limits and cleanup + +| Resource | Bound | +| --- | --- | +| Ref updates in original intent | 100,000 | +| Updates per policy page | 128 | +| Encoded page | 256 KiB | +| Catalog certificate | 1 KiB | +| Guards per Repository Cell | 4,096 | +| Required contexts per branch | 16 | +| Watches per Repository Cell | 2,097,152 | +| Watches deleted per cleanup command | 512 | + +The watch budget is one checked integer updated from exact inserted/deleted row counts. Registration checks it before writes, rather than counting the entire watch table per page. Guard admission counts at most the fixed guard cap. Epochs, cursors, totals, run numbers and context versions require integer storage; guard identity is immutable, cursors cannot move backward and validity cannot resurrect. Live watches cannot be deleted or replaced. + +Catalog ancestry policy lookups independently limit the exact existing SQL wire encoding to 256 KiB and 128 statements. A count bound alone is insufficient for long ref names. The production SQL transport remains at its existing 1 MiB limit; the fixture derives that descriptor rather than assigning a smaller artificial limit. Bit positions advance by the actual selected page length. + +Command 35, `ReapRefPolicyGuard`, requires the current admitted fence and admin access. A live valid guard at the current epoch cannot be reaped. Expired, inactive, old-owner or invalid guards can release at most 512 indexed watches per command, with transactional budget decrement. A late budget failure restores all deleted rows. There is no cascading whole-guard delete. + +An invalid guard whose creating operation is still live retains its tiny tombstone even after all watches are gone. Deleting it early would allow an old signed first page to recreate the guard. Once that operation is inactive, or the actual admitted owner has changed, the empty guard can be removed. An old page must then fail operation/fence checks before recreating any state. Exact original mutation replay can return its historical receipt after restore but grants no new write. + +These records retain no remote artifact deletion authority. Cleanup scheduling, enumeration, account fairness and process-loss service reconstruction remain required. + +## Verification and open gates + +The focused native SHA-1/SHA-256 fixtures exercise private certification, independent rebase, exact framing/truncation, long-name byte paging, update-count paging, nonaligned ancestry remapping, ordering/MAC refusals, exact mutation replay, late cursor rollback, guard/watch admission, watched mutation/replacement/deletion, configuration invalidation, immutable state, bounded cleanup, late cleanup rollback and actual owner restore. Synthetic quota/cleanup fixtures exercise fixed boundaries; they do not qualify large-team capacity. + +The original plan remains a `Vec`, ancestry remains a bounded bit vector and conditional tree preparation still owns its changed-ref map. File-backed whole-operation intent/RSS bounds remain required. Final short root/native-outcome publication, reviewed merges, production producer/reader conversion, fair hot-root progress, continuous maintenance and full-history mixed-load qualification remain open. The number and cost of page commands must be included in the mandatory 10,000-developer workload measurements. diff --git a/docs/design/shared-publication-dispatch.md b/docs/design/shared-publication-dispatch.md new file mode 100644 index 0000000..f16a872 --- /dev/null +++ b/docs/design/shared-publication-dispatch.md @@ -0,0 +1,57 @@ +# Shared foreground and maintenance publication dispatch + +`PublicationCoordinator` now dispatches privately prepared push completions, catalog compactions, bound native input checkpoints and bound Claim/Renew commands through the same bounded admission, fair queues, durability waits and uncertainty recovery. It reuses Cellule's exact `PreparedCommand`, preparation leases, purpose-bound certificates, catalog CAS and immutable outcomes. Production HTTP/SSH integration, continuous maintenance preparation, durable service reconstruction and capacity qualification remain open. + +## Typed factories and outcomes + +`PreparedCatalog::ready_push` retains the verified catalog and exact command 19 when publishing refs. `PreparationSession::ready_outcome` retains only the admitted session and exact command 19 for refused/empty outcomes; it requires no catalog artifacts. The prepared-catalog wrapper delegates those outcomes to the same session factory and drops catalog ownership from the ready value. See the [outcome-only contract](outcome-only-completion.md). `PreparedCompaction::ready_compaction` retains the verified compaction and exact command 22; it issues the existing maintenance certificate, checks a 4 KiB input envelope and checks the live lease before and after SDK preparation. Both factories perform verification/certification before admission. Raw descriptors and a caller-selected class cannot construct either ready object. + +PreparationSession::ready_inputs retains the existing command 29 and shared bound session for an adopted native input checkpoint. It checks exact scope/format/adoption context and bounded encoding before SDK preparation; the final command still checks MAC, source custody, current permission, owner, pin and expiry. These checkpoints use the foreground queue. See the [checkpoint contract](native-input-checkpoint.md). They establish descriptor retention rather than physical, canonical or ref authority. + +ReadyPreparation::claim and PreparationSession::ready_renew retain exact commands 12/13 with bounded requests and fresh post-commit session observations; see the [bound preparation contract](bound-preparation-dispatch.md). They share foreground admission with an 8 KiB reservation. The [bound lifecycle](bound-preparation-lifecycle.md) now schedules renewal automatically; durable takeover reconstruction remains required. + +`ReadyPublication` wraps those private factory outputs. `submit` accepts any factory output and returns the same `PublicationTicket`. Admission failure returns the original typed ready value, preserving its mutation identity and wire bytes. Logical IDs are unique across both classes in one coordinator. + +`PublicationOutcome` distinguishes committed push, compaction, input checkpoint and bound preparation results. RegisteredNativeInputs preserves the original registration outcome and separately reports fresh checkpoint/bound-session custody; a failed observation fences the shared session without erasing a commit. `PublicationError` preserves the corresponding typed Cellule invocation error, evidence and rejected receipt. `PublicationState` includes held, queued, running, uncertain, finished and proven unexecuted discarded states. `ticket.class()` identifies the class. `ticket.response()` accepts only a completed push outcome; compactions, input checkpoints and bound preparation commands never become HTTP push responses. These APIs replace the previous push-only outcome shape; there is no compatibility adapter. + +## Bounded class and account admission + +| Default | Bound | +| --- | --- | +| Total admitted operations | 32, including uncertain work | +| Maintenance operations | Four reserved slots | +| Foreground operations | Remaining 28 slots | +| Per actor | Eight operations per class | +| Encoded command reservation | 8 MiB per push; 8 KiB per compaction, input checkpoint or bound Claim/Renew command | +| Total command-byte budget | 256 MiB | +| Concurrent durability waits | Eight | +| Maintenance durability waits | At most two | +| Foreground dispatch burst | At most three before eligible maintenance | + +Maintenance reserves both operation slots and its maximum encoded-command credits. Foreground cannot consume those reservations; maintenance cannot consume foreground reservations. Counts for the same actor are separate by class, so foreground activity by the administrative actor does not consume its maintenance quota. Account rotation remains FIFO within each class. All maps, queues, actor strings and tickets are bounded by admitted operation counts. + +Each job records its private factory's reservation; mixed foreground checkpoint/push byte accounting uses the actual reservation rather than charging every foreground job as a push. The command reservation covers two bounded encoded copies: the retained command and the dispatch copy. Dispatch consumes its copy, rather than cloning a third payload. Private verification/native scratch keeps its separate disk/process admission. These credits do not claim to account for the entire service's heap, CPU, native descendants or provider traffic. + +Configuration requires room for another foreground account, nonzero reserved maintenance slots, checked byte headroom, a burst in 1–32, and a valid maintenance concurrency bound. With multiple durability waits, maintenance cannot use every slot. A one-wait profile permits one maintenance wait; fair class starts then share that serialized dispatch slot. Invalid profiles reject before a coordinator is created. + +## Held ownership and fair starts + +try_reserve admits a charged Held job synchronously without execution. It returns the original ready value on capacity, contention, duplicate, target or closure refusal. activate joins the existing fair queue once; discard_held succeeds only before activation, dropping resources before credits. Both remain usable after close for existing admission. close_and_drain returns held and uncertain jobs still charged. See the [final lifecycle handoff](final-publication-lifecycle.md) for worker/renewal/checkpoint ordering and observation-only final tickets. + +## Fair starts and exact recovery + +The shared queue contains two instances of the existing account-fair queue. With both classes ready and maintenance concurrency available, dispatch at most the configured foreground burst before a maintenance start. At its concurrency cap, maintenance stays queued while ready foreground work proceeds. The burst controls starts, not CPU time, I/O shares, network arrival order, SQL order or publication success. + +Catalog/ref CAS and current policy/ACL checks remain in the authoritative command. Two preparations against one old catalog can conflict even when both dispatch fairly. Uploads, native decoding, canonical verification and reconciliation never run inside this queue. A known durable catalog conflict may reenter only with a newly prepared command identity and a properly reconciled certificate. + +Dropping an observer does not cancel admitted execution. Pending, malformed published and panicked-task outcomes retain the original ready value and reservation. `pending`, `recover` and `close_and_drain` handle both classes. Recovery joins the same class/account queues. Staging Begin/Renew/Bind now reuse this same exact invocation/resolution implementation with a 4 KiB decoded-result bound; push and compaction results retain their 128-byte bound. Bound input checkpoints and Claim/Renew commands share this foreground dispatcher with a 4 KiB decoded-result bound and fresh post-commit custody queries. Staging has its own long-input admission/lifecycle rather than entering the final-command fair queues; see the [service contract](staging-service-lifecycle.md). Push and compaction dispatch check local session custody before initial submission and after authoritative absence. Resolve a known committed outcome before that guard; decode a committed result with its original receipt without rerunning its handler. Unknown, expired, unreachable or changed-incarnation evidence remains uncertain. Never replace its proof or mutation identity while acceptance is unknown. + +A terminal result drops dispatch/retained proof ownership before releasing class/account/byte credits. Resolved tickets retain only bounded result/read context. Recovery remains possible after closing admission. The bound lifecycle now owns automatic renewal and bound Claim; accepted input registration does not renew a lease or extend the original generation floor. The coordinator is service-owned local state, not a durable outbox or permission to delete remote inputs. + +## Integration and evidence + +Keep one coordinator and geometric planner per repository. Obtain a fresh admitted query-derived maintenance base, call the [geometric planner](geometric-directory-maintenance.md), wrap the verified result in an Arc and call `ready_compaction`, then `submit`. Observe or recover the exact ticket before releasing uncertain inputs. Obtain a fresh frontier for the next preparation. Integrate process admission, fair CPU/I/O shares, renewal/reaping, owner-loss reconstruction and complete retained-root inventory before selecting production handlers. + +Tests exercise class/account admission, retained failure values, duplicate logical IDs, maintenance concurrency while foreground completes, canceled observers, current admin revocation, and SHA-1/SHA-256 absent/lost-acknowledgement/panic recovery with original receipts and exactly one logical outcome. The existing push dispatcher tests remain in place with typed-result assertions. The geometric native fixture now prepares and publishes repeatedly through this shared dispatcher until ingress and level debt drain, checking canonical/source/version identity, unchanged refs and old-reader access. + +These fixtures establish bounded dispatch and recovery. They do not establish stable maintenance service under 35 pushes/s, full-history amplification, durability grouping, source-independent restore or capacity for 10,000 engineers. The mandatory workload and recovery campaigns remain release gates. diff --git a/docs/design/staged-input-retention.md b/docs/design/staged-input-retention.md new file mode 100644 index 0000000..244bb8e --- /dev/null +++ b/docs/design/staged-input-retention.md @@ -0,0 +1,56 @@ +# Staged input retention and late catalog binding + +Long imports now acquire artifact custody without retaining every catalog generation published during upload and physical verification. This uses the existing preparation token, namespace allocator, operation rows and independent lease rows. A one-way bind adds the current catalog floor only when catalog-dependent preparation starts. A [service-owned staging coordinator](staging-service-lifecycle.md) now supplies bounded admission, automatic renewal, retained typed input results, drain before Bind and exact uncertainty recovery. Production producer wiring, authenticated takeover reconstruction and full-history qualification remain required. + +## Stored phases and exact identity + +`catalog_operations.generation` and `catalog_leases.generation` are nullable in the fresh schema. NULL means input staging; zero still means a bound, certified empty catalog. A staging pin retains its complete creating namespace until its independently recorded expiry and reaping. It does not retain historical catalog facts. No new per-object table or separate input ownership service is introduced. + +Both tables store a generated `binding_generation = coalesce(generation, -1)`. The deferred composite foreign key uses that non-null value alongside incarnation, admission sequence, logical operation, owner epoch, creating namespace and expiry. Using a nullable generation directly in the composite foreign key would allow SQLite to skip all the other binding checks. The generated value prevents that escape while retaining the existing exact binding structure. + +Pin identity stays immutable. Its generation can change once from NULL to a non-null floor, with no attestation present. A bound floor cannot advance, reset to NULL or change to another generation. A staging operation or pin cannot contain a publication attestation. Both sides of the binding transition commit atomically; a pin-only update fails the deferred foreign key. + +## Typed protocol + +| Operation | ID | Result and authority | +| --- | --- | --- | +| BeginStaging | 24 | Allocates one current-owner attempt and creating namespace, without reading a catalog floor | +| RenewStaging | 25 | Extends an unexpired matching staging lease; preserves namespace and NULL generation | +| BindStaging | 26 | Reads the current certified generation and binds it once; preserves token, namespace and expiry | +| CheckStaging | 27 query | Reads an exact authorized unexpired staging lease; gives no catalog or publication proof | +| ClaimStaging | 28 | Allocates a new admitted attempt and namespace, preserving the previous independent pin | + +Begin and Claim reuse `BeginRequest` and `LeaseRequest`. Check and Bind reuse `LeaseCheck`. A distinct bounded `StagingLease` carries the token, object format and observed/expiry times, with no base fact. Decoded lease data cannot construct `PreparationBaseResolver`. The existing Abort and bounded Reap commands serve both phases. + +Repeated Begin with the same actor and request digest returns the same live staging identity; a conflicting or completed request is refused. Exact Cellule replay returns the original result and receipt. Repeated Bind returns its original floor even after the current catalog advances. It does not allocate another namespace or pin and can succeed when the pin quota is full. Read-only frontier refresh remains query 21 after binding. + +Regular Begin, Renew and Claim preparation paths reject staging rows. Staging Begin, Renew and Claim reject bound rows. CheckPreparation and CheckPreparationFrontier return no base for staging. Final publication and attestation validation explicitly reject an absent floor. Bind grants no canonical, dependency, policy or physical-verification authority; the existing private preparation and certificate factories remain mandatory. + +## Producer sequence + +1. Obtain durable BeginStaging acceptance before external upload. Keep its exact SDK mutation identity and receipt for uncertain-outcome recovery. For ordinary short pushes, the direct BeginPreparation path remains available. +2. Upload, normalize and physically verify inputs under the returned creating namespace. Apply existing process, reader, disk and artifact admission. RenewStaging before expiry using current authority; renewal never resurrects expiry or acquires a catalog floor. A thin input needing published bases must use a separate admitted base reader and retain those exact bases during normalization. Staging itself supplies no base reader. +3. Finish the input phase, preserving private physical witnesses, canonical metadata segments and admitted workspace ownership. Physical verification does not prove dependency closure or grant publication. +4. BindStaging once, then open PreparationBaseResolver through the authoritative query at the binding receipt. The creating namespace is unchanged, so the existing CatalogPreparation accepts its verified physical witnesses and segments without decoding the pack again. Resolve closure and canonical overlaps against this late selected base. +5. Complete private verification and certification, enqueue the exact ready command, and retain inputs through ambiguous outcomes. Reconcile through the existing frontier protocol; never rebind a live floor or substitute another mutation identity while acceptance is unknown. +6. On owner loss, resolve exact and logical outcomes first. ClaimStaging stamps the new admitted owner and a fresh namespace. The old pin remains intact. Staging and bound preparation recovery now have authenticated descriptor checkpoint/adoption APIs that retain the exact original input root without copying nodes; production reconstruction orchestration and physical-input custody enforcement remain open; raw descriptors cannot bypass the assembler's creating-namespace check. + +The input phase must be service-owned and renewed while workers or detached readers remain active. A DTO or canceled observer is not sufficient lifecycle management. The service lifecycle primitive now exists, including observer-independent input execution, retained results and exact command recovery; it is not selected in production yet. + +## Expiry and capacity + +Binding neither renews nor shortens artifact custody. Shortening the existing lease would invalidate previously admitted uploads or physical readers that borrowed its deadline. The bound floor therefore lasts for the remaining input lease, normally at most the default 60-second renewal interval. Configurations allowing five-minute staging renewals must account for a five-minute remaining floor when binding. This change separates an hours-long import from an hours-long catalog floor; it does not eliminate bound-floor capacity limits. + +Both phases share the 1,024-operation and 4,096-independent-pin caps. Expired and detached pins count until reaped. Claim consumes another pin; Bind does not. The existing indexed minimum ignores NULL staging generations and retains all facts at or above the oldest bound floor, including expired floors until their pins are reaped. Generation zero and the current root remain protected. Reaping removes at most 512 rows from each class in one admitted command. + +A bulk operation adds BeginStaging and BindStaging before final publication, plus renewals, any recovery claims and each selected durable input checkpoint registration (command 29). Account for this measured durable-command load separately from the ordinary two-command path. At the provisional 35 publications/s, a remaining 60-second floor contributes roughly 2,100 generations plus maintenance and reaping lag; an indefinitely renewed bound floor still exhausts the 8,192-fact cap. The 70/s headroom profile still requires qualified lease/pin/admission configuration. + +A staging pin must be included as namespace custody in any future complete retained-root collector even though it has no catalog fact. SQL reaping does not authorize remote deletion. Recovery roots, backups, active readers, unexpired custody and uncertain publication outcomes remain separate retention obligations. + +## Evidence and remaining release work + +Nine tests cover normalized nullable-phase binding and immutable floors, bounded codecs, exact replay and one-way phase transition, revocation/exact identity/expiry, namespace separation after Claim/Abort, actual owner restoration and original receipts, shared operation/pin quotas and no-allocation Bind, and pre-bind native physical verification feeding the existing private catalog proof and durable attestation for SHA-1/SHA-256. + +The retention fixture injects 10,240 intervening immutable generation facts in bounded batches with the production reaper between batches, keeping only generation zero and current. It uses trusted tiny catalog facts to test retention independently of publication and throughput. It does not simulate an hours-long import, prove provider durability or qualify full-history memory/CPU/I/O. + +Required release work includes production HTTP/SSH/mirror/generated producers integrating the service-owned renewal/drain and uncertainty handling; production integration of [durable input checkpoints and adoption](native-input-checkpoint.md), including production staging checkpoint wiring, exact bound-preparation registration supervision and physical-input custody; larger normalized input/metadata limits; capacity-aware scheduling of the remaining floor; complete retained-root collection and isolated restore; and the mandatory full-history hot-repository mixed-load campaigns. The full implementation goal remains open. diff --git a/docs/design/staging-service-lifecycle.md b/docs/design/staging-service-lifecycle.md new file mode 100644 index 0000000..f92971f --- /dev/null +++ b/docs/design/staging-service-lifecycle.md @@ -0,0 +1,71 @@ +# Service owned staging lifecycle + +`StagingCoordinator` now owns admitted Begin/Claim/Renew/RegisterStagedInputs/Bind commands, input and bound tasks and completed results through observer cancellation and exact outcome recovery. It composes the [stored staging phases](staged-input-retention.md) with fresh deadline observations and the existing private catalog pipeline. Production HTTP/SSH/mirror/generated producer selection, durable takeover reconstruction, full process admission and large-team qualification remain required. + +## Admission and ownership + +Keep one service-owned coordinator per repository. `ReadyStaging::new` prepares the exact SDK BeginStaging command without executing it; `submit` performs synchronous local admission and starts service-owned supervision. Failure returns the original ready command and reason, so retry preserves mutation identity and bytes. Foreign targets, incompatible lease duration, duplicate logical IDs, closed admission and capacity reject before execution. + +| Default bound | Value | +| --- | --- | +| Admitted operations, including uncertain work | 32 | +| Operations per actor | Eight | +| Input workers and retained completed results | 64 | +| Workers and retained results per actor | Eight | +| Encoded command envelope | 4 KiB | +| Command reservation per operation | 8 KiB for retained and transport copies; another 4 KiB after checkpoint admission | +| Checkpoint slots per operation | One bounded request and retained result | +| Renewed input lease | 60 seconds | +| Renewal lead time | 30 seconds | +| Overall session lifetime | Four hours | +| Bound local residence ceiling | 60 seconds; at most MAX_LEASE_MS | + +Limits require room for another actor, checked operation/worker maxima, a nonzero renewal lead shorter than the lease, and a lifetime from the lease duration through 24 hours. These are bounded initial profiles, not capacity results. SQL operation/pin quotas and existing process/disk/file admission remain independent. The shared actor worker semaphore spans all that actor's operations; one actor cannot take every default worker slot. Admission rejects overload rather than creating an unbounded worker queue. This component does not claim account-fair CPU or I/O service. + +The service map retains each admitted job. Dropping StagingTicket, StagingTask or an awaiting future does not cancel admitted command or producer execution. `pending(operation)` retrieves the control ticket internally. Completed worker results remain in a bounded, typed service slot and retain worker/actor admission until handoff. `pending_task(id)` retrieves such a result after a lost observer; a wrong result type rejects. This is trusted service plumbing, not an externally authorized product query. + +## Execution and deadlines + +After Begin or Claim succeeds, query CheckStaging at its receipt. Derive the local monotonic deadline from a timestamp sampled before the query and the queried remaining lease, capped by MAX_LEASE_MS. Queue and transport time shorten usable custody. A saved or replayed success never establishes a fresh deadline. + +The service periodically prepares and executes RenewStaging before that deadline. Each renewal has a fresh mutation identity; an ambiguous renewal retains its original command instead of allocating another. After a known success, another authoritative query checks live identity, format, expiry and current access before advancing the shared deadline. Producers receive StagingContext, which supplies the checked namespace token and format, and observes the shared conservative deadline and lifetime. + +`ticket.spawn` owns and supervises the producer future. Its returned StagingTask observes the result. A producer error, panic, expired custody or lost access fences the job. Cancellation aborts and joins the producer before releasing its worker credit. Completed retained inputs drop before their credit on fencing, including when an external observer remains alive. Results transfer once through `wait`; resources move to the caller before that slot releases. Retained results count toward both global and actor worker caps, preventing an unbounded completed-result backlog. + +Use existing admitted native-process, workspace, reader and disk primitives inside producers. The callback counter does not account for arbitrary heap allocation, unjoined descendants or detached physical readers. Their independent admission and retention obligations remain in force; production integration must preserve them through work and handoff. + +## Durable input checkpoint registration + +After sealing a NativeInputCertificate, call `ticket.register_inputs(proof, identity)` before seal or stop. Admission synchronously transfers that bounded envelope and exact mutation identity into one service-owned checkpoint slot. Local checks require an active live stage and matching actor/token/target. A completed successful slot can be replaced while unbound only by a proof naming that exact checkpoint digest, enabling the [request-before-native append sequence](durable-push-request.md). Pending/uncertain/failed slots and unrelated proofs reject; failure returns the original proof without executing. Existing observers retain their original receipt. The slot remains charged while the job is admitted, including uncertainty and retained completion. It adds 4 KiB to the existing 8 KiB command reservation; it cannot form an unbounded queue. + +The same supervisor prepares command 29 and stores its exact SDK command before dispatch. Due renewal precedes queued registration; accepted registration precedes Bind or graceful stop. `StagedInputsTicket::wait` observes the original durable receipt or uncertainty/error. Dropping it never discards the queued or executing command; `pending_inputs` retrieves the observer. `recover(ticket)` resolves the exact registration without replacing identity or bytes. Closing returns uncertain registrations with their existing reservations. + +A known registration stores its original receipt before a fresh CheckStaging query. That query alone can establish usable custody. Revocation after a lost acknowledgement preserves the original committed receipt while fencing the stage; authoritative absence followed by revocation or expiry rejects registration without attaching the checkpoint. Bind can start only after accepted registration resolves successfully and the fresh live probe succeeds. A retained receipt and completed replay never resurrect permission, expiry or a preparation floor. + +## Seal and catalog handoff + +Call seal when the input phase should finish. It prevents new producer admission and enters Draining. Existing producers and retained completed results continue under renewed staging custody. Bind does not begin until all input slots have drained through handoff or failure. This prevents a canceled observer from silently losing a physical witness while the service advances to catalog preparation. + +BindStaging uses a freshly prepared exact SDK command. Known binding preserves the token, creating namespace and artifact expiry, and adds only the current catalog floor. Bound records that durable result and its original receipt; its recorded timestamps are not a fresh live-lease observation. Stage contexts become inactive after handoff. The operation remains admitted through bound preparation. `ticket.open_base` refreshes at the binding receipt and uses the existing PreparationBaseResolver with the supervisor's shared session, validating current access and expiry while inheriting automatic renewal, shutdown fencing and the bound residence ceiling. + +A producer can physically verify a native pack and return its private PhysicalPackWitness and sealed metadata segments. Take that result, seal, observe Bound, open the base, and feed the witness/segments to CatalogPreparation. The existing assembler rechecks store, namespace, partition completeness, canonical overlap and closure. Its private factories issue the publication proof. Bind and a generic producer result do not grant canonical or publication authority. + +Bound preparation is now automatically renewed by this service, and spawn_bound reuses its worker/result ownership; see the [bound lifecycle contract](bound-preparation-lifecycle.md). After a bound Claim, PreparationSession::ready_inputs can transfer an adopted checkpoint into the existing PublicationCoordinator for exact registration recovery; see the [checkpoint contract](native-input-checkpoint.md). The shared dispatcher now owns exact bound Claim/Renew commands through private factories; see the [bound preparation contract](bound-preparation-dispatch.md). Final publication now uses the [serialized lifecycle handoff](final-publication-lifecycle.md); durable takeover orchestration remains required. Configurations allowing a five-minute staging lease can leave that much remaining catalog-floor retention; the floor-cap and hot-repository progress requirements are unchanged. + +## Exact uncertainty and shutdown + +Begin, Claim, Renew, RegisterStagedInputs and Bind share the same exact invocation/resolution implementation with push and compaction dispatch. Resolution of authoritative absence permits execution of the retained exact command. A committed outcome decodes with its original receipt; it never reruns the handler. Unknown, expired, unreachable, changed-incarnation and malformed published results retain evidence and reservation. + +Uncertain stops new producer admission. Existing work can continue only through its previously established deadline. `recover(ticket)` resumes the exact retained command; it cannot replace its identity or bytes. No new renewal, registration or bind is issued while an earlier command remains ambiguous. Panicked command tasks retain pending evidence. Unexpected service-worker failure fences local work and requires explicit exact recovery before restarting supervision. + +`stop` prevents new workers and waits for accepted input tasks/results to drain while renewal continues. It does not retract an independent SQL pin. `close_and_drain` closes all admission, stops jobs and returns still-charged uncertain tickets once running commands and input slots have drained. Service consumers must take retained completed results before a graceful stop can finish; retrieve lost observers through pending_task. Recovery remains possible after closing. A reached lifetime or lost authority fences and discards untransferred results conservatively. + +This service is process-local ownership, not a durable outbox or authenticated input inventory after process loss. Owner takeover must resolve exact/logical outcomes and reconstruct or adopt retained physical inputs under the new admitted namespace through the [authenticated input checkpoint protocol](native-input-checkpoint.md). ReadyStaging::claim now retains/resolves the exact Claim command and supplies a fresh staging context. Staging checkpoint supervision now exists; exact checkpoint supervision after bound Claim now uses the publication dispatcher. Production producer wiring, durable takeover reconstruction and complete wire-plan/response recovery remain required. Final publication now uses the existing fair coordinator through an observation-only lifecycle ticket; accepted final intent continues through close, while a pre-activation fence discards only proven unexecuted work. Neither local completion nor SQL reaping authorizes remote deletion. + +## Evidence and remaining work + +Nine service tests cover canceled observers and single typed handoff; operation/account/global and actor worker admission; rejected ready-command reuse; automatic renewal without a waiter or manual tick; Begin/Renew/Bind absent, lost-acknowledgement and post-execution panic recovery with original bind receipts; close/drain retaining uncertainty; fresh renewal queries after revocation; producer error/panic fencing; completed-resource drop before credit release with a live observer; and native SHA-1/SHA-256 physical verification followed by late binding, the existing private catalog proof and durable attestation. + +Five additional checkpoint service tests cover canceled observers; absent, lost-reply and panicked exact dispatch in both OID formats; registration-before-Bind ordering and original receipt replay; one-slot and foreign/duplicate/closed refusal without execution; committed recovery followed by current-access revocation; and authoritative expiry after absence. The real receive/publication/cold-clone fixture uses service-owned registration in both formats. + +These tests establish protocol composition and ownership on small fixtures. They do not establish four-hour full-history throughput, stable maintenance under peak traffic, source-independent restore or capacity for 10,000 engineers. Production producer wiring, complete resource/descendant admission, authenticated durable input inventories and owner-loss adoption, remaining-floor configuration, complete retained-root reclamation, accelerated readers and mandatory mixed-load/recovery campaigns remain release gates. diff --git a/docs/design/verified-edge-spool.md b/docs/design/verified-edge-spool.md new file mode 100644 index 0000000..d97bd37 --- /dev/null +++ b/docs/design/verified-edge-spool.md @@ -0,0 +1,31 @@ +# Shared dependency spools for native verification + +PhysicalVerifier now retains one dependency file per metadata page, rather than one file per structural object. A page still contains at most 512 privately constructed VerifiedObject witnesses. This reduces dependency file descriptors from as many as 512 to one for the physical-verification page. Native Git processes, pack/index files, SQLite connections and other concurrent operations have separate resource costs; this change does not establish a whole-process file-descriptor limit. + +The implementation reuses VerifiedObject, DiskSink, AdmittedFile, the existing Cellule DiskBudget and the fixed-width typed-edge encoding. It changes private witness ownership, not persisted catalog or metadata bytes. Structural dependencies remain occurrence records containing the child OID followed by its expected-kind byte. The maximum encoding buffer remains 512 records (16,896 bytes for SHA-256). Blobs and other objects with no emitted dependencies allocate no file or retained range. + +## Producer and range ownership + +Create one private EdgeSpool at the start of each physical metadata page. Acquire one append permit for the entire inspected object, including between its edge pages. Reject overlapping object writers; a file mutex alone would allow two objects' pages to interleave. The permit stays in the object's shared state, so queued blocking writes retain exclusivity after the requesting future is dropped. Complete native frame/hash verification remains the only way to construct a decoded witness. + +On the first dependency page, assign the object's start offset from the admitted file length. Subsequent pages must begin at that object's previous end and the file's current end. Checked arithmetic enforces both the per-object dependency-byte ceiling and the accumulated file length. Before any write, reserve or grow disk credit. Seek explicitly to the admitted end: a prior witness replay may have moved the shared file cursor. Update file length, object length and digest only after the entire page is written. + +Mark the storage failed before allocation, growth, length checking, seeking or writing. An unrecovered failure leaves it poisoned; retained witnesses reject replay and later appends reject. This prevents a partially written tail or a denied growth attempt from being silently adopted by a subsequent object. The operation fails closed and is discarded. An observer-canceled queued write may finish and leave an orphan range. Those bytes remain admitted but cannot produce the canceled object's witness. The writer permit releases only after all of that object's workers drain. + +A successful witness owns the storage Arc, start offset, exact length and complete occurrence digest. Completion releases the object writer permit while preserving the immutable range. PhysicalVerifier drops the producer handle before transferring the whole page to the blocking metadata worker. Every witness and queued write keeps the complete admitted file alive independently of the producer. Dropping some witnesses never releases partial credit: their bytes still occupy the same file. Cleanup removes the file before releasing its entire reservation, reusing AdmittedFile's conservative credit retention on cleanup failure. + +## Replay and SQLite retries + +Replay holds the storage mutex for its range. Reject poisoned storage, zero or misaligned range lengths, overflowing/out-of-file ranges and any actual file length differing from the admitted append length. Seek to the private range offset and read exactly its declared bytes in pages of at most 512 dependencies. Validate every OID and kind and compare the complete range digest before the metadata transaction commits. Appended bytes from later objects are outside the range and cannot enter its graph. + +Every SQLite growth retry repeats the complete range validation and digest calculation from the same start. MetadataBuilder keeps all witnesses through transaction rollback/replay and poisons sealing on an unrecovered error. No committed header/edge prefix survives a late digest failure. Shared file cursors, repeated replay and reversed witness order cannot change canonical dependency identity. + +There is no global spool or repository-wide verification lock. Separate admitted physical operations use separate page files. A slow worker retains its file and admission until it exits. Whole-operation/native RSS, CPU/I/O and global file/process admission still need the shared production resource profile, and production handlers must adopt the verified pipeline before the hard cutover. + +## Verification + +A native SHA-1/SHA-256 fixture creates 512 distinct commits and retains all decoded witnesses in one dependency file under a 64 KiB disk budget. It interleaves earlier-range replay with later appends, checks exact offsets and parent/tree identities, replays in reverse order twice, and verifies full-file credit retention until the last witness drops. + +The wide-tree metadata fixture uses the same page spool through admitted SQLite growth under a 2 MiB shared budget. A late corruption check modifies the second range at a nonzero offset after a valid 1,600-edge tree range; the entire metadata batch rolls back and sealing remains poisoned. A deterministic blocked-worker fixture confirms that cancellation prevents completion and new-writer admission until queued work drains, then checks that orphan bytes cannot enter a later witness. Additional checks reject denied growth, a real write failure after credit growth, later appends to poisoned storage, appended/truncated files and premature credit release. Existing isolated physical SHA-1/SHA-256, artifact-integrity and queued-assembly fixtures exercise the production PhysicalVerifier wiring. + +These tests qualify dependency ownership, integrity and bounded page file count. They do not prove full-history import throughput, native descendant limits, large-team capacity or completed production cutover. diff --git a/docs/large-repository-implementation-plan.md b/docs/large-repository-implementation-plan.md new file mode 100644 index 0000000..fa6268b --- /dev/null +++ b/docs/large-repository-implementation-plan.md @@ -0,0 +1,311 @@ +# Canopy packed repository implementation plan + +**Mandatory scale amendment:** [Large-team scalability requirements](large-team-scalability.md) supersedes the per-object Cell schema, globally serialized preparation and normal globally drained deletion in this plan for the >10,000-engineer workload. Its dependency-ordered deliverables and mixed-load gates are part of the implementation scope. Do not land the earlier DDL as the fresh-format release schema and defer bulk metadata/catalogs to a later migration. + +Implement the [packed repository design](large-repository-storage-design.md) as one incompatible deployment format. Every Git object uses a native pack; existing object, graph, ref, push and collaboration structures are reused. Start on fresh data. Do not build migration tooling, dual writers or a compatibility `ObjectStorage` variant. + +**Deliverable status:** the design, executable schema fragment, schema checks and native Git smoke fixture are delivered. Runtime changes listed below are not implemented by these documents. Existing working-tree cache/benchmark work is separate and should be integrated, not overwritten. No Kubernetes, Linux or Chromium capacity claim is closed by the small fixtures. + +Track actual primitive integration and remaining full scope in [implementation status](large-repository-implementation-status.md). The [final publication lifecycle](design/final-publication-lifecycle.md) now connects private final factories, staging worker/result drain, exact renewal/checkpoint recovery and account-fair dispatch without another command payload layout. It remains a process-local composition requiring production wiring and durable takeover. The earlier schema fixture is superseded for the large-team release; its passing checks do not validate the new catalog architecture. + +The [durable original-request checkpoint](design/durable-push-request.md) now reuses the authenticated artifact transport and native input index: register encoded bytes before receive, then append captured native descriptors under exact predecessor CAS. Staging/bound reconstruction recomputes the scoped request identity, and restored-owner adoption preserves the original roots without copying. The [durable native-result checkpoint](design/durable-native-result.md) now preserves the exact native response/options/signature witness and versioned plan, freezes the descriptor inventory, and supports current-custody reconstruction and whole-root adoption. The final short ref-plan command interface, production takeover orchestration and every producer/reader conversion remain required in F/G/J; these APIs do not complete those packages. + +## Implementation sequence + +Each package has one reviewable outcome. The schema cutover and all producer/consumer changes land together in the release; intermediate development commits are allowed to be unreleasable. Avoid temporary production fallbacks merely to make an intermediate commit deployable. + +```mermaid +flowchart LR + A[A. Format and metrics] --> B[B. Cellule fence] + A --> C[C. Artifact transport] + B --> D[D. Metadata and closure] + C --> E[E. Pack verifier] + D --> F[F. Push and generated objects] + E --> F + F --> G[G. Readers and cache] + G --> H[H. Product and ancestry] + G --> I[I. Compaction] + C --> J[J. Backup and retained roots] + D --> J + I --> K[K. Online retained-root collection] + J --> K + H --> L[L. Qualification and cutover] + K --> L +``` + +Suggested ownership is a Canopy storage implementer for C–G/I, a Canopy runtime/product implementer for A/H/J/K, and the Cellule maintainer for B and the retained-root portion of J. These are code ownership boundaries, not a requirement for multiple agents or a particular staffing model. With limited staffing, follow dependency order. + +## A. Establish the new deployment format and measurements + +**Files:** `crates/canopy-server/src/deployment/root.rs`, `crates/canopy-server/src/deployment/mod.rs`, `crates/canopy-server/src/server/mod.rs`, `crates/canopy-server/src/lib.rs`, `crates/canopy-server/src/main.rs`, `docs/contracts.md`, `docs/operations.md`; preserve existing evaluation scripts and cache edits. + +1. Wrap `RootPurpose` in the `canopy-pack-v1` envelope at the existing marker key. Make startup, maintenance, backup and restore require it before opening Cells. Reject missing, old or unknown format; test old decoder rejection of the new envelope. +2. Version local workspace/cache directories with the same format. Reserve a new provider prefix/application identity for every qualification deployment. Keep SQL bootstrap version 1 in that fresh format; update module source digest coverage when new files are added. +3. Introduce one shared `PackLimits` configuration covering transfer concurrency, native threads/window, command row/byte bounds, scratch budget, artifact part limit and work/idle deadlines. Follow existing config parsing conventions; no separate Git-store service config file. +4. Add stage timers and physical/logical byte counters from the design before optimizing. Log operation IDs and stage transitions. Keep metric labels bounded; full OIDs and repository UUIDs belong in diagnostic logs. +5. Remove stale hard-coded 120-second whole-worker assumptions in docs/tests. The working tree already uses a one-hour deadline; implement distinct idle/work budgets and an explicit four-hour bulk-import class. + +**Native ownership implemented:** the [shared guard](design/native-process-ownership.md) retains existing owners through Unix inherited-end drain and leader reaping, with bounded supervised cancellation cleanup and conservative quarantine. Decoded-object actors reuse that guard. The [native resource pool](design/native-resource-admission.md) now requires a four-dimension permit at every shared spawn, with one configured node pool and disjoint foreground/maintenance shares. Required native_limits is wired through production gateways/caches and explicit preparation scopes. The supervisor now closes and drains that same pool before Cell shutdown, workspace release, heartbeat stop and advertisement withdrawal; poisoned/underflowed accounting and quarantined claims keep drain unproven. OS containment, account/fair preparation scheduling, profile/descriptor-preservation qualification and non-Unix descendant support remain required. + +**Acceptance:** fresh-format startup succeeds; both directions of format mismatch fail before serving/writing; old prefixes are never adopted; metrics distinguish verification, artifact transfer, Cell publication and cache work; cancellation reaps all native descendants. + +**Review artifact:** format decision in persisted contracts, config defaults table, startup rejection tests, one tiny import stage report. + +## B. Expose Cellule's existing owner fence + +**Runtime delivered:** [Cellule PR #38](https://github.com/crabbuild/cellule/pull/38) merged the accessor using activation-specific admission capabilities. After merging Canopy main, all Canopy pins consume revision `0f4ca0919b0dfe20a3dcd964d21da03135e42eed`, which includes that API. The original draft revision's workspace tests, lints, API docs and contract checks passed; those historical results do not qualify this newer dependency revision. The Canopy prepared-operation acceptance scenario below remains part of package D; exposing the runtime accessor does not implement it. + +**Repository:** Cellule. **Files:** `crates/cellule-runtime/src/registry/handlers.rs`, registry execution construction sites, `cell/actor/` admission/execution paths and tests; reuse `control` authority and `identity::IncarnationId`. Update all Canopy Cellule dependency revisions together after the dependency change passes. + +Add: + +```rust +#[derive(Clone, Copy, Eq, PartialEq)] +pub struct OwnerFence { pub incarnation: IncarnationId, pub epoch: u64 } +impl CommandContext<'_, '_> { + pub fn owner_fence(&self) -> OwnerFence; +} +``` + +Populate it from the admitted actor authority. Audit replay and retry execution so a handler sees the correct admitted execution fence; receipt replay must return the old result without re-executing against a guessed current fence. Keep caller-supplied fence bytes separate from runtime context. Canopy command handlers compare both explicitly. + +**Acceptance:** a worker begins under A; authority moves to B; A's delayed mutation submitted through B is rejected by the Canopy token check; B claims attempt 2 and resumes; an exact already-committed request replay remains stable. Test eviction/reacquisition with unchanged owner and genuine epoch/incarnation changes separately. + +**Scope limit:** do not redesign Cell roles, Blob routing, queues or workflow composition. Packs initially use Canopy's external artifact transport. No Git concepts enter the Cellule API. + +## C. Generalize authenticated artifact transport + +**Files:** `crates/canopy-object-storage/src/external.rs`, `crates/canopy-object-storage/src/artifact.rs`, `crates/canopy-server/src/lfs/mod.rs` and its existing helper callers, `crates/canopy-server/src/deployment/backup/bodies.rs`. + +1. Extract/reuse `ArtifactDescriptor` and generalize `publish_lfs`/hashed reads into the common `CANOPY02` path. Preserve LFS SHA-256 identity and body behavior. Delete `CANOPY01` support when its Git callers are removed in F/G. +2. Stream BLAKE3 file/part hashes. Enforce exact manifest length, part count, fixed part lengths except the final part, checked integer arithmetic and the 512 GiB per-artifact limit. +3. Implement derived repository/creating-operation-scoped pack/index paths. A retired path is never reused, including for an identical later upload. On conditional-create collision, verify existing manifest and part content against the descriptor; do not accept `AlreadyExists` as proof. +4. Upload parts before manifests. Add bounded concurrent downloads and per-artifact single-flight within a node; canceled waiters do not prematurely destroy an active shared producer. +5. Reuse the helper in backup copying. No SQL `pack_parts` table and no separate manifest codec for each artifact kind. + +**Acceptance:** round trips across zero-length LFS, part boundary, multi-part pack/index; corrupt/missing/reordered parts; conflicting existing manifest; lost upload reply; restart before manifest publication; provider timeout and disk exhaustion. A complete descriptor cannot be returned for incomplete bytes. + +**Review artifact:** exact manifest golden vectors, failure-injection tests and provider request counts for a representative pack. A local memory store result alone does not qualify remote object-store semantics. + +## D. Publish the bulk catalog and verified closure under a short Cell command + +**Files:** `crates/canopy-server/src/schema.sql`, `crates/canopy-server/src/lib.rs`, `crates/canopy-server/src/packs/{metadata,directory,sources,catalog,verification,closure}/`, new operation/publication commands, existing ref/push/product commands and repository Cell tests. **Dependency:** B's actual admitted owner fence; C's authenticated artifacts; E's complete isolated physical verification. + +The earlier [SQL fixture](design/packed-repository-schema.sql) is not the release definition. Deliver fresh release DDL together with the new deployment marker. Preserve product tables unless a demonstrated requirement changes them; keep push response/certificate chunks. Remove historical Git bodies, mutable per-object placement and graph rows from the Cell. Store current catalog descriptor/generation, root certification, fenced operation attempts/outcomes and retention/reader-pin facts. Immutable metadata segments and directory/source indexes hold canonical object/edge inventories. Do not implement the superseded `PutObjects`, `CertifyObjects` or `SwitchPackLocations` Cell commands as a compatibility stage. + +1. Reuse the implemented immutable artifacts, canonical `ObjectHeader`/typed edges, native index ordinal partitions and bounded catalog codecs. Operation records bind repository, format, identity, admitted owner fence/attempt, input catalog/generation, artifact descriptors and exact durable response. Register large input/output sets through a bounded immutable root; never serialize every historical descriptor into Begin or Complete. +2. Implement a trusted base resolver from the retained, certified input catalog and authoritative root facts. Keep a valid generation lease throughout preparation. Reuse `CatalogFiles` for admitted, authenticated SQLite files and `CatalogReader::headers` for grouped file reads. Resolve only requested OIDs in ordered batches of at most 512; presence in a native index or raw catalog is insufficient for closure certification. Compare the complete incoming canonical header against every matching base header, including body and graph digests. +3. Compose `PhysicalVerifier` and `ClosureVerifier`. The assembler now seals its admitted incoming spool, verifies its full canonical inventory against the closure witness, and streams disjoint bounded output files into one indexed level-zero root. Output runs are at most 64 MiB; `new_with_run_limits` configures their size independently of the verification spool. Directory-root v3 stores the same range-index references at every level; range-index v2 always stores physical file facts plus logical coverage. Both reject older layouts. Full files and projections reuse the same StoredRun representation and canonical header fold; see the [coverage contract](design/directory-run-coverage.md). Exhaustion verifies complete output coverage before a PreparedCatalog can escape; reconciliation reuses that private incoming root. This implements output partitioning. Metadata/directory/closure/ancestry spools now use [admitted geometric growth](design/admitted-sqlite-growth.md) and indexed bounded closure setup transactions; full-history profiles, native/file/process resource admission and production input retention remain required. Physical verification now reuses one admitted dependency file per at-most-512-object page, with exact witness ranges and complete retry hashing; see the [edge-spool contract](design/verified-edge-spool.md). Whole-process file/native admission remains required. Stream exact physical partitions one shard at a time; discard incomplete/canceled/failed preparations. Bind the closure witness's unique canonical inventory to the incoming directory and its physical-input digest set to complete source coverage. Authenticate and verify output catalog descendants before issuing a trusted publication certificate. Raw descriptors and caller-controlled booleans cannot produce it. Carry the bounded certificate inline in the ordinary final command; durable operation-row registration is an optional checkpoint, whose extra command must be counted. The implemented factory and registration reuse the repository secret with a separate MAC domain. `PreparedCatalog::ref_proof` now derives catalog-certified target membership and admitted disk-backed ancestry, binding the exact existing PushPlan wire bytes and evidence into the certificate. Production selection, native commit-graph acceleration and large-history qualification remain open. +4. On publication against a changed generation, inspect only intervening changed runs/ranges for incoming OID intersections, reject canonical conflicts and rebind the verified facts to the actual CAS base. Recompute affected closure/policy facts if dependency retention or refs changed. Use a bounded fair ready-operation coordinator; physical verification and uploads remain concurrent outside it. The implemented read-only CheckPreparationFrontier query now returns the original attempt/floor and current root without another durable Claim or namespace. Independent pins retain every later generation until reaped; indexed bounded cleanup and a DELETE guard enforce that range. PreparedCatalog::reconcile now retains private incoming roots and admitted closure scratch, checks incoming intersections and exact external anchors against the selected current catalog, reuses physical/DAG verification, and reconstructs roots from current plus incoming. Certificates and final/checkpoint validation bind original floor and actual selected base separately. PreparedCatalog::ready_push and PublicationCoordinator now supply bounded account-fair final-command admission, concurrent durability waits and retained exact-command uncertainty recovery. Integrate those primitives with frontier/maintenance orchestration and qualify changed-run skipping and pipelined/grouped root work under continuous publication; API correctness alone does not prove hot-repository progress. The existing operation/pin rows now support an unbound staging phase through commands 24–26/28 and query 27; generated non-null phase binding preserves exact deferred FK checks. Bind once after physical verification, preserving namespace and input expiry, then reuse the existing private assembler and frontier. The service-owned StagingCoordinator now supplies bounded operation/actor/worker/result ownership, automatic fresh-query renewal, single typed result handoff, drain before Bind and exact uncertainty recovery; see the [service contract](design/staging-service-lifecycle.md). NativeInputIndex/NativeInputCertificate, immutable pin checkpoints (command 29/query 30), ReadyStaging::claim and authenticated descriptor adoption through staging or a bound PreparationSession now exist, reusing the exact immutable input root without copying index nodes; see the [checkpoint contract](design/native-input-checkpoint.md). StagingTicket::register_inputs now owns one bounded registration through cancellation and exact recovery, serializing renewal before registration and registration before Bind; CatalogPreparation::begin_retained_pack now checks exact authenticated checkpoint membership and carries its digest through certificate v3, reconciliation and final publication. PreparationSession::ready_inputs now supervises exact bound checkpoint registration through the same PublicationCoordinator, using an 8 KiB foreground reservation and fresh post-commit custody checks. ReadyPreparation::claim and PreparationSession::ready_renew now retain exact commands 12/13 through the same service dispatcher and return original receipts alongside fresh custody; see the [bound preparation contract](design/bound-preparation-dispatch.md). The staging supervisor now retains Bound admission, accepts claim_bound, renews its shared session automatically, reuses typed worker/result/checkpoint slots and enforces a separate local residence ceiling; see the [bound lifecycle contract](design/bound-preparation-lifecycle.md). Integrate those APIs, final-publication lifecycle serialization and full wire-plan/response recovery into production producers; qualify larger input limits and remaining-floor capacity. See the [staged input contract](design/staged-input-retention.md). An indefinitely renewed bound floor still exhausts the 8,192-fact cap under traffic. Verify progress under continuous independent-branch traffic and maintenance; do not blindly advance a generation, rescan history or restart full preparation on each CAS loss. +5. Keep the final Cell mutation short: validate admitted owner fence/attempt, actual base generation, trusted attestation, current ACL/policy/check bindings and expected ref identities/versions. Reuse the private `refs::validate_refs` result for format/ACL/CAS/tombstone/namespace validation and its command-bound application. Add catalog-certified target membership and independently verified ancestry; the legacy `graph::certified_roots` and policy ancestry SQL cannot run against the fresh schema. Evaluate current required checks, reporters and merge authorization in the final transaction. Atomically publish the catalog/root facts, refs and independent exact request outcome through the selected Cellule durability gate. Failed CAS changes no published root; lost replies replay the original outcome without reexecuting preparation. `PublishCatalogRefs` now implements the typed catalog/ref/outcome transaction, with current-policy checks and all rejection paths before writes. CompleteCatalogPush now invokes that same core and persists the exact native response, options and signed-certificate bytes/ownership within that transaction, with durable final-policy refusals and receipt-bound exact reads. Its factory checks native report/plan agreement and opaque signed-witness target/request binding. Route actual HTTP/SSH producers through the API, invoke implemented CheckCompletedPush/replay_push_response before native preparation, and integrate reviewed merge/candidate bindings before selecting this path. PreparationSession now issues purpose-bound outcome-only proofs directly from the admitted lease, without loading or uploading a catalog, and ready_outcome uses the existing exact final-command dispatcher. Wire actual failed/empty HTTP/SSH producers through this session path. Publishing the typed ref result and saving the HTTP response in another command is insufficient. Deliver an immutable ref-plan root for inputs beyond the current 4 MiB inline envelope. The typed publisher now accepts a privately reconciled selected base while preserving the original attempt floor. Frontier/maintenance scheduling and mixed-load qualification in step 4 remain mandatory capacity prerequisites. +6. Allow concurrent private preparation only after overlap/ref/policy races pass. Bound operation records, node scratch and retained generations; maintenance has separate admission. Reconstruct private scratch from authenticated inputs after a crash. The closure scratch's `synchronous=OFF` is not the durability policy for artifacts or the Cell. +7. Separate the logical request ID from the creating artifact attempt. Reuse existing artifact descriptor operation fields with a unique admitted-attempt identity. The implemented Begin/Claim allocator uses the persistent repository watermark and preserves it through operation pruning and normal owner recovery; independent pins retain namespaces and optional certificates after Claim/Abort. Integrate these facts into complete recovery/output retention so collection cannot discard old-worker ownership, and ensure isolated rollback restores use a new provider namespace. Qualify a delayed delete from an old attempt against a recreated logical push with identical content, including owner succession. No online artifact deletion is allowed until that test and complete retained-root inventory pass. + +**Acceptance:** matching/conflicting cross-pack OIDs; changed base generation; incomplete/mismatched physical partitions; cross-format/repository input; stale owner/attempt; premature publication; wide tree/reverse fanout and deep chain; missing/wrong-kind dependencies; cycles including base-overlapping incoming vertices; failed last batch; pinned old-root reads; ref ABA/policy revocation; exact durable replay after takeover. Query-plan and resource checks must show bounded indexed work rather than a historical table scan per new object. + +**Review artifact:** fresh DDL/codec and deployment-marker diff, trusted attestation boundary, publication/failover tests and phase timings on synthetic 100k-object preparations followed by full-history incremental pushes. The old Python DDL checker and primitive tests cannot close this package. + +## E. Build the streaming pack verifier + +**Files:** `crates/canopy-server/src/git_objects/mod.rs`, `crates/canopy-server/src/native_git.rs`, `crates/canopy-server/src/git_http/mod.rs`, new `crates/canopy-server/src/packs/verify.rs` and `crates/canopy-server/src/packs/spool.rs`; native-resource tests. + +1. The new GitHttpBackend::run_native_receive API retains pack/index pairs even for small staged requests. GitHttpBackend::stage_native_packs captures request-private pairs using the admitted StagingContext namespace, the existing pinned-file uploader and an exclusive native file fence; it returns authenticated inputs without legacy body/placement rows or an OID inventory. The native receive/staging/verification/publication/cold-clone composition fixture executes those APIs for SHA-1/SHA-256. See the [capture contract](design/native-input-capture.md). Authenticated descriptor inventories/checkpoints and retained-input adoption now exist using the shared range index and lease row; see the [checkpoint contract](design/native-input-checkpoint.md). The real receive fixture now registers through the staging supervisor. Bound checkpoint supervision now uses the publication dispatcher. Exact bound Claim/Renew dispatch now uses private factories and the publication coordinator. Automatic bound renewal, Claim admission, worker/result/checkpoint ownership and a local residence ceiling now reuse the staging supervisor. Final-publication lifecycle serialization now exists. The production gateway also uses [owned push preflight](design/owned-push-preflight.md), binding encoded request identity before replay and retaining normalized input with parsed intent. Production producer orchestration connecting that owner to the packed lifecycle, durable takeover reconstruction and complete wire-plan/response recovery remain open. Turn loose generated objects into packs. Repair thin packs with native Git. Use Git 2.50.1-compatible commands initially and record exact Git version in reports. +2. Validate native `.idx` v2 and pair checksums. Reopen each output pack in an isolated object directory without alternates and decode every physical entry. Graph references may point outside the pack; delta bases may not. +3. Stream canonical hashes and structural parsing. Extract parser logic from graph verification so there is one tree/commit/tag semantics implementation, with both bounded stream and small-fixture adapters as useful. Include signed commits/tags, multiline headers, binary names and gitlinks. +4. Sort/deduplicate inventory and edges in scratch SQLite with a 32 MiB page cache. Add a direct, pinned `rusqlite` dependency for this disposable spool if no existing supported scratch interface suffices; never bypass Cellule for authoritative SQL. +5. Produce the canonical inventory folds, artifacts and per-parent edge streams. On recovery, rebuild private spool from retained authenticated input artifacts and compare full descriptors. Reuse completed output artifacts only after re-verification; incomplete private progress is disposable. Owner succession acquires a new admitted attempt and creating namespace while preserving old retention pins. Do not depend on the removed per-object SQL ingest cursor. + +**Acceptance:** independent Git/body hashes agree for SHA-1 and SHA-256, delta chains, thin packs, empty blob, huge single object and oversized tree. Tampered body/index/trailer, wrong external delta base, malformed structural object and conflicting typed child fail. Memory does not scale with largest decoded body or full OID set; account separately for native index memory. Cancellation removes/reclaims scratch without losing committed progress. + +**Review artifact:** native version/command qualification, streaming-parser adversarial tests, subprocess/scratch peak measurements. Reuse [the native smoke fixture](design/check_native_pack_contract.py) but add large/fault fixtures in Rust integration tests. + +## F. Route every object producer through packed publication + +**Files:** `crates/canopy-server/src/git_gateway/mod.rs`, `crates/canopy-server/src/git_gateway/push.rs`, `crates/canopy-server/src/push/mod.rs`, `crates/canopy-server/src/refs.rs`, `crates/canopy-server/src/pulls/candidates/mod.rs`, merge/rebase producer modules and server file-edit handlers found by their old `StoredObject` call sites. + +1. Replace body enumeration/publication with E's verified artifacts and D's metadata/edge pipeline. Seal outputs and establish closure before the converted final push command. Reuse existing push identities/options/certificates and exact response semantics while publishing catalog facts and refs in that same final transaction; calling the unchanged legacy completion handler is insufficient. +2. Preserve ordinary and atomic push behavior, hooks, push options/certificates, ref CAS, branch rules and durable response replay. Recheck ACL/policy at completion, not just on initial admission. +3. Submit network pushes, mirror imports and every generated commit/tree/blob through the same service. Small generated batches produce native packs too. +4. Begin an operation before external upload so it owns the artifact namespace. Retain authenticated inputs through uncertain outcomes and process/owner loss, then reconstruct conditional preparation under a newly admitted attempt if appropriate. A disconnected client or rejected ref decision does not authorize deleting another worker's inputs. Keep public reachability based on published refs/candidates, and distinguish retained private artifacts from published catalog membership. +5. Remove durable inline/chunked/external Git-body paths and old fixtures testing those layouts. Replace them with equivalent behavior tests over packs; preserve large LFS tests. + +**Acceptance:** HTTP and SSH push/clone for both object formats; no new raw Git body rows or per-blob external objects; exact replay after lost response; owner death at upload/register/header/edge/seal/ref/cache boundaries; denied actor cannot stage through public endpoints; racing branch update does not publish the losing plan. Generated merge/rebase/file-edit objects survive cache loss. + +**Review artifact:** first complete fresh-format vertical slice: import a small repository, inspect canonical metadata and pack descriptors, delete the disposable cache, then clone/fsck and replay a push outcome. + +## G. Unify readers and replace heap inventory with native indexes + +**Files:** `crates/canopy-server/src/object_reads/mod.rs`, `crates/canopy-server/src/git_read/mod.rs`, `crates/canopy-server/src/git_cache/mod.rs`, `crates/canopy-server/src/git_cache/maintenance.rs`, `crates/canopy-server/src/git_gateway/{hydration,fetch,maintenance}.rs`, `crates/canopy-server/src/server/mod.rs`, HTTP/SSH service construction; cache tests. + +1. Inject `GitObjectReader` into all consumers. Resolve canonical object headers and typed graph metadata from the selected certified catalog's immutable files; the Cell supplies authoritative ref/product snapshots and catalog facts. Part downloads, native lookup and verified streams remain in the service layer. Do not retain per-object SQL metadata queries as a hidden dependency of this cutover. +2. Reuse current immutable generation/pinning mechanics. Install complete verified pack/index pairs; coalesce downloads; keep private request refs. Replace per-pack/full-repository heap `HashSet` membership with native batch/index/MIDX lookup. +3. Cold installation resolves bounded directory/source pages from one pinned certified catalog snapshot. Warm refresh compares certified catalog generations and reconciles their changed runs/source bindings; certify exact installed canonical coverage before advancing the cache. A numeric high-water mark or matching native entry count cannot establish coverage. Test an object introduced physically before its later catalog publication, an overlapping source replacement and concurrent installation against a changing catalog. The selected fresh schema has no legacy object-sequence or `sealed_generation` cursor. +4. Distinguish full and structural-only coverage. Preserve shallow/partial clone semantics, hidden-ref isolation and exact-want authorization even when extra objects are present physically. Mixed-pack cold transfer amplification is measured and reported. +5. Build MIDX/commit-graph per complete declared generation. Generate bitmaps only with a complete supported inventory and version-tested invocation; otherwise native traversal remains correct. Qualify exact subcommands as well as the format: current [Git MIDX documentation](https://git-scm.com/docs/git-multi-pack-index) states that `compact` writes version 2, unreadable before Git 2.54, and that `expire`/`repack` are incompatible with incremental MIDX files. The initial Git 2.50.1 profile cannot depend on that newer compaction path. Choose and test its supported maintenance sequence, or explicitly qualify a newer native Git deployment before using newer commands. Never invoke a destructive cache maintenance command against pinned pack files. +6. Reconcile reservations with actual disk usage, including pinned old generations and scratch. Stop all native children before releasing pins. Cache failure after commit remains recoverable. + +**Acceptance:** repeated warm fetch performs no historical-body hydration; cold cache reconstructs from artifacts; simultaneous readers and local cache maintenance cannot lose files; cache eviction and disk exhaustion are bounded; visibility change invalidates authorization despite cache hits; filtered cache cannot masquerade as full. MIDX/commit-graph verification passes for both OID formats. + +**Review artifact:** traces showing one cold installation and subsequent delta refresh, plus a per-request memory/disk breakdown. Do not introduce generated-fetch-output caching until these invariants pass. + +## H. Complete browser, candidate and large-history behavior + +**Files:** `crates/canopy-server/src/git_read/mod.rs`, `crates/canopy-server/src/ancestry.rs`, `crates/canopy-server/src/pulls/candidates/mod.rs`, merge/rebase/patch/history readers; `crates/canopy-server/tests/multi_server/{browse,comparison,candidates,merge,rebase,partial_clone,sha256}.rs`. + +1. Remove SQL-body dependencies from browsing, diffs, patches and merge input readers. Retain bounded preview/API response semantics with explicit errors; storage's streaming support is independent. +2. For generated candidate validation, compare expected canonical OID/body digest/size and certified metadata. Preserve ordered commit parents and exact policy inputs; unordered `commit_parents` is not an order proof. +3. Replace ancestry's in-memory discovered-commit limit and permanent mutable SQL ancestry projection with certified immutable commit metadata/native commit graphs plus admitted disk-backed traversal scratch when needed. Reuse existing OID/typed parent meanings and verify every selected path against the pinned certified catalog. Bind a bounded ancestry certificate to the exact old/new OIDs, canonical inventories and publication context; authenticate it in the final policy transaction. Do not submit the legacy parent-proof command to tables removed by the fresh schema. Keep cancellation and admission; incomplete traversal is an error, not a negative result. The fallback now reuses admitted SQLite growth, exclusive per-walker traversal, exact StoredCatalog memo binding and permanent failure/cancellation fencing. Queue resets page at most 512 keys and preserve bounded exact-catalog answers. Native commit-graph acceleration, serving/candidate integration and native histories exceeding 100k commits still require implementation/qualification. +4. Audit minimum receipts and ref snapshots across product reads after owner movement. Do not read a locally cached newer/older branch in place of the selected authoritative snapshot. + +**Acceptance:** a synthetic history exceeding 100k commits can check positive/negative ancestry, prepare merge/rebase candidates and exercise branch protection within configured budgets; wrong parent order/body digest is rejected; wide tree pagination and binary paths remain correct. Signed commits/tags and SHA-256 candidates pass existing semantics. + +**Review artifact:** API regression matrix and large-history policy tests. Capacity reports must distinguish a budget failure from unsupported Git history. + +## I. Implement online bounded durable compaction + +**Implemented increment:** `PreparedCompaction` selects 2–32 exact level-zero roots from an admitted query-derived catalog, streams verified input runs through the existing merge/partition/range-index structures and replaces them with one root. Defaults are 128 input runs/256 MiB, a 256 MiB spool ceiling with 192 KiB initial charge and up to 768 MiB reservation, and 64 MiB output files. Purpose-bound certificate v3, admin command 22 and recovery query 23 reuse preparation leases, selected-generation CAS, immutable outcomes and optional checkpoint rebinding. Concurrent ingress is retained; replacement of any selected input rejects reconciliation. No refs, native pack bytes or source roots are rewritten. Seven correctness tests are included in the 289-test workspace run. The steps and acceptance below remain required for geometric levels, physical packs, ongoing resource-admitted scheduling, retention and takeover reconstruction; this package is not complete. + +**Further implemented:** `prepare_range` streams one source projection and a bounded consecutive prefix of target overlaps into the adjacent level. It verifies the complete parent projection and both split inventories in one scan, moves the verified prefix and retains an exact suffix descriptor referencing the same physical file. Physical input bytes deduplicate shared files; record limits count projections. Disjoint promotion reuses the authenticated artifact; merges reuse the existing builder and output partitioner. Exact source/target replacement path-copies the shared range index; current-range revalidation preserves unrelated updates and rejects changed or newly overlapping inputs. Repeated jobs now progress across over-budget overlap sets when each required physical input fits the profile. Geometric advisory policy is implemented below; fair continuous service scheduling, repeated-parent scan amplification, large-input retention and physical rewriting remain required. + +**Geometric selection implemented:** `CompactionPlanner::prepare_next` reuses queried snapshot/NodeRef logical counts and the existing range job. Defaults use 262,144 objects, ratio four, eight-root urgency and a maximum three urgent preparations before a rotated higher-level turn. Ingress slots rotate; one indexed last-moved OID per level wraps after exhaustion. Invalid profiles, terminal-level over-capacity and foreign repository/format reuse reject. `CompactionPressure` exposes fixed-size count/target diagnostics. Preparation failure leaves local rotation unchanged; successful private preparation advances advisory traversal, not durable progress. Production admission, publication/recovery dispatch, maintenance/foreground resource shares and owner-loss reconstruction remain required. See the [execution contract](design/geometric-directory-maintenance.md). + +**Shared publication dispatch implemented:** `ready_compaction` issues the existing maintenance certificate and retains the exact SDK command. `PublicationCoordinator` admits push/compaction variants with typed outcomes, reserved class counts/encoded bytes, per-class actor counts, bounded foreground bursts and maintenance concurrency. Both classes share cancellation-safe retention, supervision, pending lookup, original-receipt resolution and close/drain. Defaults reserve four maintenance operations and two of eight durability waits; the geometric native fixture now uses this dispatcher for every publication. Continuous preparation, whole-process CPU/I/O shares, renewal/reaping, durable reconstruction and production invocation remain required. See the [shared dispatch contract](design/shared-publication-dispatch.md). + +**Files:** new `crates/canopy-server/src/packs/maintenance.rs`, D's commands, existing `crates/canopy-server/src/git_gateway/maintenance.rs`, node maintenance/admission scheduling. + +1. Begin a durable compaction operation and pin its selected catalog generation and exact input descriptors. Initial selection: at most 32 packs or 8 GiB compressed; prioritize small/duplicate packs, leave large stable history alone. A single over-budget pack requires a separately admitted job, not an unbounded default repack. Account for directory-run compaction separately from physical pack rewriting. +2. Stream selected preferred objects into structural/blob OID spools. Pack each with native settings from the design, verify and upload outputs. Preserve all canonical objects, including unreachable ones. +3. Seal outputs and construct immutable directory/source replacements in admitted batches of at most 512 objects. Reuse each entry's expected canonical header/source/placement version for private replacement validation; compare all overlaps and preserve the canonical inventory. Publish the changed catalog through generation CAS under the actual owner fence. Reconcile intervening generations before a retry so an older replacement cannot overwrite a newer preferred placement. Do not introduce per-object authoritative location updates or a whole-history heap OID map. +4. Retire inputs only after certified effective-source resolution shows they are unused in the current root and the complete retained-root inventory shows no pinned old catalog, preparation, serving reader, backup or uncertain outcome still depends on them. Raw directory row counts do not establish this. Perform remote deletion only through the qualified reclamation state machine; keep cache deletion under its independent process/file pins. Reuse current scheduler admission, but remove redundant per-node full repacking when durable outputs already exist. +5. Resume after takeover by re-verifying output inventory. Bound total active/staged/retired bytes and stop new compactions when retention headroom is exhausted. + +**Acceptance:** crash at every private replacement batch and final catalog CAS; concurrent push; concurrent claimed/stale worker; old pinned reader during swap; matching canonical inventory before/after; no ref-generation change; compaction alone grants no remote deletion authority. At least 100 incremental-push/compaction cycles produce an explainable physical-byte inventory. + +**Review artifact:** compaction operation trace, canonical inventory equivalence and pre/post storage breakdown. Merely repacking disposable cache files does not complete this package. + +## J. Make backup and recovery inventory complete + +**Canopy files:** `crates/canopy-server/src/deployment/backup/mod.rs`, `crates/canopy-server/src/deployment/backup/bodies.rs`, `crates/canopy-server/src/deployment/root.rs`, `crates/canopy-server/src/deployment/recovery.rs`, `crates/canopy-server/tests/multi_server/backup.rs`, deployment tests. + +**Cellule files:** `crates/cellule-runtime/src/recovery/{backup,retention}/`, control/catalog publication and pin APIs; relevant LTX retained-root APIs after tracing their existing recovery use. + +1. Add a read-only retained-root enumeration API to Cellule. Reuse existing root/pin types. Return paginated roots with retention reason, control/pin revision bindings and a completion indicator. A caller may treat it as exhaustive only while its existing maintenance barrier prevents root/pin changes; detect revisions changing during enumeration and fail/restart. +2. Audit every recovery selector, pin and unfinished backup/restore path. Test that each possible selected root appears in the enumeration. If any cannot be enumerated, return an explicit incomplete result; the collector must refuse deletion. +3. Canopy walks each retained SQL snapshot's current catalog, immutable generation facts, independent preparation pins/checkpoint certificates, unfinished/uncertain operation outcomes and LFS facts. Resolve each retained catalog's complete authenticated directory/source dependencies, including immutable index nodes and all required metadata/pack/index artifacts. Preserve private attempt namespaces and unregistered output ownership until their writers are proven drained; never infer their deletion eligibility solely from catalog membership. The fresh schema has no `objects.pack_id` inventory. Retired catalog tombstones and completed operations alone do not retain bytes. Include all certified canonical objects, not only currently referenced tips. Restore preserves logical IDs, descriptors and the artifact allocation watermark; an isolated rollback restore uses a new provider namespace so delayed old deletes cannot target its bytes. +4. Copy and verify all required artifacts before setting backup complete. Preserve/restore product tables and push replay state as in existing backup behavior; no old-format migration is added. + +**Acceptance:** backup while compaction changes locations retains the old snapshot's packs; source pins survive incomplete copy; corrupted/missing index blocks completion; isolated restore with source access revoked succeeds for Git, LFS, product data and push replay. Collection candidate inventory includes roots recoverable after an uncertain publication. + +**Review artifact:** retained-root API tests and a source-independent restore report. An API that enumerates only current roots does not satisfy the collector prerequisite. + +## K. Add qualified repository-scoped durable collection + +**Files:** `crates/canopy-server/src/deployment/mod.rs`, `crates/canopy-server/src/deployment/recovery.rs`, new `crates/canopy-server/src/packs/collection.rs`, new bounded owner-fenced publication/retention commands, existing CLI maintenance command group. + +The large-team amendment replaces the original global-drain requirement for routine collection. Reuse maintenance identities and owner fencing, with complete repository/catalog-generation retention accounting from J. Serving and backup readers must register renewable generation pins before artifact access; detached native/I/O workers keep their ownership until they actually stop. Expiry, elapsed grace time or a missing heartbeat alone does not prove that those workers have drained. Qualify suspended-reader behavior and clock bounds before allowing deletion. + +Deliver five stages: enumerate and bind the complete retained-root inventory; derive candidates by repository and creating namespace; atomically recheck revisions/fence/ownership and mark eligible incarnations deleting; perform idempotent authenticated descriptor-based deletion while persisting bounded progress; finalize bookkeeping after confirming deletion. The mark transaction and subsequent admission must prevent any new publication, recovery selection or reader pin from depending on a deleting incarnation. Concurrent operations on retained/live generations continue; remote deletes do not hold the publication coordinator or require deployment-wide downtime. A changed or incomplete retained-root inventory invalidates the decision before any new delete is issued. + +Offer a dry-run reporting retained reasons and eligible bytes. Execution binds those candidates to the rechecked inventory and maintenance fence; the manifest is operational progress, not a second object-location database. Keep globally drained collection as a recovery tool. Until repository-scoped retention, admission exclusion and detached-reader safety pass, no production online deletion is enabled. + +**Acceptance:** a remotely held reader or pinned old backup prevents its artifact deletion while unrelated publications proceed; suspended or dead nodes require proven worker fencing/drain; incomplete/changed inventory prevents marking; pin acquisition and publication racing deletion fail safely; collector/owner loss at each mark/part/finalize boundary recovers; an already-issued delayed DELETE cannot damage a recreated logical push with identical content because the new attempt owns a fresh namespace. After completion, retained packs clone/fsck after total local cache loss. Mixed-load qualification shows bounded collector resources and no publication-coordinator hold across remote deletion. + +**Review artifact:** dry-run and executed manifests from the same synthetic lifecycle, crash recovery log, and before/after retained-byte totals. Logical unreachable-object pruning remains excluded. + +## L. Qualify and perform the hard cutover + +Use the existing `scripts/local_eval.py` and `scripts/benchmark_large_repository.py` as the harness foundation. They currently cover only part of this matrix; extend their reports rather than claiming unimplemented checks ran. Benchmark only on a dedicated disposable deployment because the harness restarts its node. + +Pin source refs and OIDs before each run. Save Canopy source-tree digest, Cellule revision, Git version, provider/version, CPU/RAM/disk/network limits, concurrency and every failure. Separate mainline Linux, stable histories and `chromium/src`; do not label a shallow snapshot as full history. + +| Corpus | Required operations | Gate | +| --- | --- | --- | +| Synthetic adversarial | SHA-1/SHA-256; wide tree; large single blob; deep deltas; >100k commits; many refs; shared-child fanout | Correctness, bounded work and policy preservation | +| Kubernetes full history | Import, warm/cold full clone, incremental push/fetch, v0/v2, shallow and `blob:none`, browse, owner loss | First end-to-end large-repository release gate | +| Linux mainline, then stable | Same operations; deep ancestry and merge-base; compaction/collection | History/graph scale beyond the first corpus | +| Chromium `src` | Same operations; metadata/edge/LTX footprint and cold preview amplification | Chromium support claim only if this corpus passes | +| Mixed fleet | Large import/compaction beside the existing small-repository density workload | Small metadata/tree p95 <=2x unloaded baseline; report queue/rejection rate | +| Isolated restored deployment | Revoke original prefix and all old local caches; full clone/fsck, LFS and hosting state | Durability gate for each claimed provider | + +Repeat performance trials at least five times after fixing warm/cold conditions; record raw samples and p50/p95 with the sample count, not a precision claim from one run. Use longer repeated traces for meaningful latency percentiles. Warm incremental fetch p95 target <=1.25x matched native Git; live pack/index bytes after compaction/collection <=2x matched canonical-inventory native baseline. Report current SQL, recovery roots/LTX, staged/retired artifacts, backups, peak RSS and local scratch separately, plus total bytes versus native Git. A full-repository rewrite on each one-file push fails the incremental-work gate even if its small fixture is fast. + +The first baseline run sets explicit absolute import, cold-clone and restore targets for the chosen hardware/provider. Record those targets before optimization runs; do not silently raise them to pass. All claimed corpora must fit the same declared resource policy or carry distinct published profiles. A Kubernetes-only pass is a Kubernetes-only claim. + +**Cutover deliverable:** new release plus provider-qualified report, runbook, format rejection evidence and restore drill. Create/import repositories under the new identity; switch traffic only after checks pass. Keep any old deployment isolated for its owner's archival policy. There is no automatic transfer of old hosting metadata and no rollback into the old format after new writes. + +## Test and execution commands + +Available now, non-destructive design checks: + +```sh +python3 docs/design/check_packed_repository_schema.py +python3 docs/design/check_native_pack_contract.py +``` + +Required implementation checks after the relevant packages land: + +```sh +cargo fmt --check +cargo test --lib +cargo test --test repository_cell +cargo test --test smart_http +cargo test --test git_http +cargo test --test multi_server +cargo test --test owner_restart +cargo clippy --all-targets -- -D warnings +``` + +Run touched-module tests during development, then the complete release suite once integration is ready. Test names added for pack failures must be discoverable in the existing test targets or a documented new target. Run Cellule's own runtime/control/recovery tests in its checkout before updating the pin; do not assume Canopy tests cover the dependency's authority behavior. + +Example existing large-corpus harness invocation, after a fresh local-evaluation deployment has been created according to [the local evaluation runbook](../deploy/local-evaluation.md): + +```sh +python3 scripts/benchmark_large_repository.py \ + --state-dir /absolute/path/to/disposable-eval-state \ + --source /absolute/path/to/pinned-kubernetes.git \ + --work-dir /absolute/path/to/new-result-directory \ + --name kubernetes --mode full --timeout 14400 +``` + +The state/source must already exist and work directory must be new. The harness restarts the evaluation node. This command is a template with explicit local paths, not a claim that qualification ran. Set the server's bulk-import budget too; increasing the client timeout alone cannot fix server cancellation. + +## Required fault matrix + +Each cell below is a test family, with SHA-1/SHA-256 coverage on a representative subset and at least one remote provider before release. + +| Boundary | Inject | Assert | +| --- | --- | --- | +| Pack/manifest write | Missing part, conflicting create, lost response, provider outage | No seal/ref publication for incomplete artifacts; safe retry | +| Header/edge commands | Kill/restart, stale cursor, last-row conflict, repeated request | Atomic batch and exact resume/replay | +| Authority | Owner transfer between upload and mutation; old attempt after claim | Old worker cannot stage, seal, move or publish | +| Ref completion | ACL revocation, racing ref CAS, branch-rule update, lost response | Final policy holds; exact committed outcome replay | +| Native cache | Disk full, corrupted index, canceled downloader, reader during replacement | No unchecked bytes, use-after-eviction or ref drift | +| Compaction | Crash after some switches, new push, repeated job | Canonical identity preserved; both generations retained safely | +| Backup | Pin before location move, copy interruption, source disappearance | Correct retained root or explicit failed backup | +| Collection | Remote reader, incomplete root inventory, lost drain fence, partial delete | No deletion until proven safe; resumable maintenance | +| Product | Parent reordering, hidden candidate object, large-history budget exhaustion | Policy/reachability preserved; resource error distinguished | + +## Ref root publication implementation sequence + +The [immutable ref state](design/immutable-ref-state.md) supplies conditional versioned roots and streaming initial construction. Complete the final publication change in this order: + +1. The shared streaming rewrite now coalesces existing-base batches by affected subtree and preserves untouched roots. Qualify sustained ordinary and bulk preparation separately, including long-name byte splits, retained tombstones, provider budgets and hot-root fairness. +2. The fresh immutable catalog generation now carries the ref snapshot through the same query-derived base and retention floor. The private preparation factory loads and rewrites that exact root; compaction carries it forward and the inline publisher refuses selected roots. Fresh empty initialization now authenticates a private empty preparation and atomically installs joint roots with one durable outcome; wire it into repository creation at cutover. Bind membership/ancestry and exact policy/check facts to the privately issued transition certificate, and qualify their current-state CAS/fairness semantics. +3. Direct-push [paged policy guards](design/paged-ref-policy-guards.md) now bind rare configuration epochs and indexed exact check dependencies, with bounded transactional registration/cleanup and private conditional root signing. The private [immutable completion factory](design/immutable-push-outcomes.md) now binds registered native custody and freezes success/refusal descriptors in an 8 KiB input. Publish those exact catalog/ref/response descriptors atomically with a bounded command that checks the live guard/epoch in that same transaction. Keep current authorization, owner/lease fencing and exact recorded replay; remove full-plan transport and per-ref final SQL mutation. Implement reviewed-merge bindings and service-owned exact page dispatch/recovery; include page costs in hot-repository capacity qualification. +4. Convert every producer, reader, default-branch, policy/check, review and recovery path together. Delete the old ref/body schema and adapters for the fresh-data cutover. +5. Include snapshots and their transitive immutable nodes in complete retention, collection and isolated restore, including the immutable initialization outcome's retained empty catalog/ref roots. Qualify hot-root fairness and full-history mixed workloads against the mandatory large-team gates. + +## Release completion checklist + +- [ ] One new-format schema/codec set; old Git body structures and callers removed. +- [ ] All object producers and readers use packs; stock HTTP/SSH and LFS behavior passes. +- [ ] Trusted-verifier boundary and fencing tests pass; no client-controlled verification path. +- [ ] Metadata and graph work is paginated, replayable and within SQL/wire limits. +- [ ] Warm refresh handles late certification and does not scan/rebuild full history. +- [ ] Candidate parent order and large-history ancestry checks pass. +- [ ] Online compaction survives crashes; durable deletion requires complete retained-root inventory and valid reader fencing. +- [ ] Isolated backup restore works without source artifacts or local caches. +- [ ] Required corpus/provider reports meet declared performance and resource gates. +- [ ] `docs/contracts.md`, `docs/git-compatibility.md`, `docs/performance-plan.md`, `docs/operations.md`, `docs/delivery-plan.md` and `ROADMAP.md` reflect actual evidence. + +Authorized output caching, concurrent preparation, bounded OID lookup, online retained-root collection and the large-team gates are mandatory for this release under the scale amendment. Bundle/CDN bootstrap, fork-family deduplication, range-based decoding and additional logical reachability pruning may follow after those gates pass. diff --git a/docs/large-repository-implementation-status.md b/docs/large-repository-implementation-status.md new file mode 100644 index 0000000..b7b1dc8 --- /dev/null +++ b/docs/large-repository-implementation-status.md @@ -0,0 +1,284 @@ +# Large-repository implementation status + +Updated during implementation on 2026-10-02. **The full implementation and capacity goal remains open.** The [large-team amendment](large-team-scalability.md) is mandatory scope alongside the original storage design. Passing primitive tests is not completion of the hard cutover or proof of capacity. + +Implementation is isolated in the PR worktree. The original checkout contains an unrelated, extensive staged workspace merge; its workspace, benchmark and runtime work has been preserved. Canopy is now split into Git-format, object-storage and server crates. Cellule dependencies and their lockfile entries are pinned together to `0f4ca0919b0dfe20a3dcd964d21da03135e42eed`, matching the merged Canopy main branch. This revision includes the admitted command owner fence from merged [Cellule PR #38](https://github.com/crabbuild/cellule/pull/38). Historical validation below remains attributed to its original source revisions. Canopy now consumes it in preparation operations and the typed catalog/ref publisher. Trusted ref-plan certification, atomic typed publication and the exact-response completion API exist; production HTTP/SSH producer/reader conversion and frontier/maintenance scheduling remain open. A bounded class/account-fair dispatcher now serves pushes, compactions, input checkpoints and bound Claim/Renew commands. + +## Immutable completion preparation + +The private `PreparedCatalog::root_push_completion` factory now authenticates the exact registered native checkpoint, reopens its original plan/report/options and prepares joint immutable refs against its selected generation. It freezes original success, generic publication rejection and signed-certificate-replay rejection before minting a certificate binding all descriptors and ownership facts. The final input is bounded by 8 KiB and contains no ref plan or response bytes. New metadata/failure bodies use the current admitted namespace; legitimately adopted native success bytes can retain their original creating namespace. Shared metadata roots, response descriptors and the original native result are reused without duplicating the request, plan or signed body. See the [immutable outcome contract](design/immutable-push-outcomes.md). + +The original compile caught private helper/test-module visibility and a fixture SQL-state argument mismatch; these were corrected without changing production limits. The first focused run failed all four cases. Three synthetic descriptor fixtures incorrectly used UUIDs where creating namespaces require `CANOPY01` plus a positive admitted sequence. After correcting those fixtures, the multi-ref receive exposed a test-only 4 KiB packet-group assertion; its exact generated group now supplies the read length, with an explicit assertion that the old smaller read rejects it. The custody probe then found the composition using `begin_pack`, which leaves the registered checkpoint unbound. The rooted path now uses the actual `begin_retained_pack` API and asserts the exact checkpoint digest. The last fixture failure was an expected wrong-purpose rejection incorrectly consumed with SDK `?`; it now asserts the typed `InvocationError::Rejected` and precise Unauthorized result. The final focused run passed all four tests (3.16 seconds), including native SHA-1/SHA-256 with 257 refs over three policy pages, exact frozen outcomes, eight tamper cases per format, stale policy refusal, maximum signing-key framing, explicit no-report-status rejection and authenticated late-part corruption. The composed unsigned input remains below 2 KiB. A mistaken short-name `--exact` probe selected zero tests and is excluded from runtime evidence. Earlier failed logs remain preserved. + +The complete workspace library run passed all 473 unique tests with four threads: six Git-format (1.40 seconds), 14 object-storage (5.66 seconds) and 453 server (162.61 seconds). This includes the new preparation checks, existing native receive/publication/cold-clone composition and owner/lease/checkpoint/ref-policy recovery. Child-process test summaries are not counted as additional unique tests. + +The first all-target Clippy run rejected a nested policy error whose enum variant occupied at least 128 bytes. Boxing that error preserves its source and behavior; the final all-target workspace Clippy pass succeeded with warnings denied (37.02 seconds). The library run above preceded this representation-only correction; the final focused rerun passed all four original checks (2.01 seconds). All three retained-catalog startup/refusal integration tests passed (0.83 seconds). Formatting/diff checks passed, and every local link in the changed design/status/plan documents resolves. The 63,965-byte historical PR-body archive matches its source byte-for-byte (SHA-256 `c7494d679abed5e1e55a5b2d605d80e786cb4de86406d77f0c7a37c71c79437e`). + +This is conditional preparation, not atomic outcome selection or an acknowledgement. Final root publication, authorized durable replay, typed audit/retention traversal, supervised recovery, file-backed report/intent handling, production hard cutover and full capacity qualification remain mandatory. Historical PR progress and validation have been [archived verbatim](archive/pr20-progress-through-8bb0ee7.md); the archive preserves earlier failures and their original scope. Both CI runs at `8bb0ee7` completed successfully: [37081670293](https://github.com/crabbuild/canopy/actions/runs/37081670293) and [37081666281](https://github.com/crabbuild/canopy/actions/runs/37081666281). Those runs precede the new completion preparation changes. + +## Main branch merge + +PR #20 merges Canopy main at `64db462`, then integrates its newly merged pack-storage prerequisite at `877dc33`. All five Cellule dependency pins and six lockfile entries now agree on `0f4ca0919b0dfe20a3dcd964d21da03135e42eed`, including the merged owner-fence accessor. Both documentation navigation entries are retained. The lifecycle test requests normal listener binding through the new optional-listener argument. The inherited-fork regression retains the live-fence disk charge, then verifies deferred reclamation before admission is released; it also uses the existing native resource permit. + +The second main merge preserves the already validated Rust production sources byte-for-byte. It keeps native admission, bounded pack indexes, weak gateway-owned pack-reader lifetime and the stronger gateway-only teardown assertion. Main's additional predecessor-repository rejection check and benchmark source identity test are retained, along with the source digest helper that resolves every workspace crate from the script location. The updated Python harness passes all 96 tests (105.31 seconds). Both Directory/repository compatibility checks passed (0.37 seconds), and the final all-target workspace Clippy pass after this second merge succeeded with warnings denied (3 minutes 43 seconds). + +The initial library compile exposed one lifecycle test call using the old two-argument startup signature. Adding `None` for normal listener binding corrected it. The complete workspace library invocation then passed all 458 unique tests (6 Git-format, 14 object-storage, 438 server; server 280.77 seconds). All-target workspace Clippy passed with warnings denied (3 minutes 17 seconds). All 95 Python harness tests passed (57.35 seconds). The first retained-catalog integration run failed all three cases before startup: the historical full-application fixture also selected obsolete Repository code, which the hard cutover does not retain. A shared owned fixture now keeps other modules current while selecting the exact historical Directory descriptor. The original golden bytes and digest checks remain unchanged, and both contract and startup fixtures explicitly assert refusal of the full historical application. The Directory contract check passed; all three original retained-catalog startup scenarios then passed (1.40 seconds), preserving immutable catalog identity and unsupported-control refusal assertions. Four listener-handoff checks passed (0.81 seconds). Signed SHA-256 SSH push and fresh-disk restore passed (3.61 seconds). Both HTTP backend checks passed (0.21 seconds), and the full stock-Git push/clone workflow passed (225.54 seconds). Final all-target workspace Clippy after the fixture correction passed with warnings denied (39.39 seconds). Formatting, diff checks and all 86 local documentation links passed. Earlier failed runs remain recorded below. This merge does not complete producer/reader cutover or large-team capacity qualification. + +## Paged current ref policy preparation + +The fresh packed-publication module now prepares direct-push policies in pages of at most 128 updates/256 KiB. It reuses the existing plan, catalog certificate, preparation token, immutable ref transition and root descriptor. Rare rule/context configuration mutations advance a monotonic epoch; ordinary CI reports instead invalidate indexed dependencies on the exact newest required run, including tuple-moving SQLite replacements. Guard preparation survives unrelated catalog/ref generation advancement, while final signing rechecks catalog membership/ancestry and conditional ref expectations. Registration, watches, scalar budget and cursor updates share one transaction. Cleanup deletes at most 512 watches and retains live invalid tombstones to prevent old page MACs from recreating readiness. See the [paged policy contract](design/paged-ref-policy-guards.md). + +The first compile found an unconverted ref-shape error and a SQL result field named `rows_affected`, both corrected. The original focused run passed the root/rebase check and failed its unrelated-report assertion because the native fixture's `other` OID equals its tip. The probe now uses an explicitly distinct blob OID. The expanded run passed eight cases and failed long-name preparation at a SQL codec limit. Investigation confirmed a generic 4 KiB fixture input where production already permits 1 MiB, plus a real count-only ancestry batching bug that could exceed production's limit. The fixture now derives the existing production descriptor, while ancestry lookups count the exact SQL wire bytes and stay within 256 KiB/128 rows; no production limit changed. All ten focused guard tests passed (3.66 seconds), and the separate query-bound regression passed (0.06 seconds). It demonstrates that the old 128-long-name batch exceeds the unchanged production limit, then verifies bounded exact encode/decode for every new page. No page receipt or guarded snapshot grants final publication authority. The short joint root/native-outcome command, reviewed merge binding, production cutover, file-backed intent and large-team capacity remain required. + +CI run [37077052177](https://github.com/crabbuild/canopy/actions/runs/37077052177) at main-merge head `a286214` passed its harness but failed one of 104 multi-server tests: ref discovery assumed three new repositories necessarily evict the original. The other run [37077055118](https://github.com/crabbuild/canopy/actions/runs/37077055118) at that same head completed successfully. The unchanged failing test passed a focused local reproduction (3.29 seconds), confirming an intermittent setup precondition. Existing residency code can temporarily exclude a just-published Cell, and other cold-restore tests already establish actual cold state through a bounded shared helper. Commit `7ff1fe1` reuses that helper and preserves directory absence, the original two-second discovery assertions and complete clone/fsck integrity checks. The corrected focused test passed (3.63 seconds). The failed CI invocation remains failed; new-head CI is still required. + +The complete workspace library invocation passed all 469 unique tests with four test threads: six Git-format (2.49 seconds), 14 object-storage (43.90 seconds) and 449 server (623.06 seconds). This includes all ten guard cases, the ancestry byte-bound regression, existing generation/lease/checkpoint recovery and native capture/ref/pack verification. Preparation deadlines and resource caps remain unchanged. The first all-target Clippy run found one parity assertion that should use `is_multiple_of`; after that equivalent test-only change, all-target workspace Clippy passed with warnings denied (1 minute 56 seconds). All three retained-catalog startup/refusal integration tests passed (0.94 seconds). Formatting/diff checks and all 55 local links in the four updated design/status documents passed. These tests do not qualify whole-operation RSS, continuous hot-root progress or full-history/10,000-developer mixed load. + +## Fresh empty catalog/ref initialization + +The private empty preparation now mints a purpose-bound ref initialization proof through its own artifact store. It requires generation zero, zero catalog objects/edges/inputs, no input checkpoint, a live lease and current admin access. The final command authenticates the existing certificate, rechecks current owner/admin/attempt/expiry/pin/retention/base and refuses any prior ref history, changed default head, push/compaction outcome or positive catalog generation. It atomically records the checkpoint, installs catalog generation one with authenticated empty ref metadata generation zero and retains one immutable exact logical outcome. The shared GenerationFact is reused; no SQL-ref conversion or decoded-root signing adapter exists. Exact recorded replay and the original mutation receipt survive actual owner restore; a pending old attempt still requires the current fence. Complete collection/isolated backup must retain the tiny initial catalog/ref roots stored in that outcome. See the [initialization contract](design/immutable-ref-state.md). + +The first focused compilation failed because the pinned runtime's read-only QueryContext does not expose a target accessor. The lookup now uses the existing routed Cell capability and verifies its persisted repository identity plus current admin access, as the existing compaction query does. No assertion, deadline or capacity limit was widened. All four focused initialization tests passed (3.02 seconds). They cover both object formats, deterministic private preparation, input/reply framing and truncation, first ref preparation from the installed root, exact mutation/logical replay, current identity/role checks, history/head/expiry/MAC/scope refusal, immutable outcome guards, late-write rollback, concurrent attempts with exactly one winner and actual restored-owner recovery with stale pending-worker refusal. The factory future retains a Send assertion. Full-workspace and integration validation follows below. + +The first broad library run passed 446 unique tests and failed five existing native/catalog cases (426 server passed, five failed; server 324.74 seconds). Four returned Base(Inactive): complete physical partitions/base reuse, ancestry pair reuse, full ingress compaction and final current policies. The denied-ancestry-growth case failed its expected budget-error assertion. All four new initialization tests passed in that run. Before probing, the ranked possibilities were preparation lease expiry under long native work, shared clock/fence interference and a schema/query regression. These fixtures use the unchanged 60-second default preparation lease. All five unchanged cases then passed in separate sequential reproductions: physical partitions/base reuse 95.77 seconds, ancestry pair reuse 92.99 seconds, denied ancestry growth 39.35 seconds, full ingress compaction 199.90 seconds and final current policies 79.90 seconds. No source, assertion, deadline or capacity change separated the failed broad run from these reproductions. This supports contention/lease expiry as an explanation; the broad run remains failed and does not qualify concurrency or capacity. Across the broad run and unchanged reproductions, all 451 unique library cases have passed, without claiming a passing full-workspace invocation. All-target workspace Clippy passed with warnings denied (1 minute 32 seconds). Formatting/diff checks and documentation link checks passed. Rebuilt stock-Git smart-HTTP integration passed (129.34 seconds). Rebuilt signed SHA-256 SSH push with registered key, audit assertions and fresh-disk restore passed (3.11 seconds). Final formatting/diff checks, whitespace for all 12 changed/new files and 52 local documentation links passed. CI [37067561966](https://github.com/crabbuild/canopy/actions/runs/37067561966) completed successfully at `ecc5c6c`. This validates the initialization increment before the main-branch merge; the merged dependency and source need their own checks. + +This protocol is not yet wired into production repository creation. Short catalog/ref/outcome publication with current policy/check binding, every producer/reader, fresh-schema selection, complete retained-root collection/isolated restore and full-history/large-team qualification remain required. It does not establish production capacity or complete the hard cutover. + +## Joint catalog/ref generation preparation + +The fresh publication schema stores a bounded optional ref snapshot in the existing immutable catalog generation. Shared GenerationFact codecs, preparation queries and catalog certificates carry the exact descriptor. The original generation floor retains this joint fact and later generations; compaction copies the descriptor rather than resetting or advancing the ref metadata generation. The new privately constructed preparation value reads through the prepared catalog's own query-derived base/store, reuses the coalesced ref transition, binds the canonical plan digest and preserves default-branch metadata. It accepts no external store/root/transition. Missing snapshots refuse rather than fall back to an empty tree. The inline SQL ref publisher refuses new writes when an immutable snapshot is selected. Catalog attestations use purpose domain v4, refuse v3 and retain their 1 KiB bound; no old-domain adapter or schema migration is added. + +The first all-target check and focused compile attempt failed on a missing Arc import; that import is corrected. The first focused run passed two tests and failed one (0.72 seconds): a mechanical fixture edit added a fourth checkpoint decoder column to a three-column query. Restoring only that fixture projection corrects the mismatch; the whole-state rejection oracle includes the new ref root separately. + +The first full library run passed 446 unique tests and failed the new maximum-shape certificate check (426 server passed/one failed; 183.10 seconds). The isolated check reproduced CodecError::Limit in 1.16 seconds. Repeated repository/format/creating-operation fields in the catalog descriptors overflowed the unchanged certificate bound; the new v4 payload encodes those already-bound facts once and reconstructs the existing typed structures on decode. The same maximum-shape assertion and 1 KiB production cap are retained. All three focused query-derived ref tests passed after the compact encoding change (1.63 seconds), including the unchanged maximum-shape assertion. + +The final library suite passed all 447 unique tests (6 Git-format, 14 object-storage and 427 server; server 160.29 seconds). Four new tests cover SHA-1/SHA-256 exact query-derived bases, canonical plan digests, old roots and deterministic retries, tombstone expectations, stale-base and actor refusal, cross-format/future/missing metadata, full joint-fact framing, MAC binding, maximum populated certificate fields, generation-floor retention, inline-publisher refusal with an unchanged whole-state oracle, and compaction carrying the exact ref root while preserving outcome replay. Existing old-domain, native/catalog/recovery and final-policy tests also passed. The factory future retains a Send assertion. All-target workspace Clippy passed with warnings denied (29.36 seconds). Rebuilt stock-Git smart-HTTP passed (87.63 seconds). Signed-push/registered-key/audit-after-restore integration passed (3.44 seconds). The affected OpenSSH RSA/ECDSA authentication case passed (1.87 seconds). Formatting/diff checks passed; all 21 changed/new files passed whitespace checks and 51 local document links resolved. + +Membership/ancestry plus exact current policy/check proof, short atomic catalog/ref/outcome publication, all producers/readers, complete collection/isolated restore and capacity qualification remain open. These changes are not selected by production serving. Prior-head CI at 95650f0 passed Verify run 37056401256 and failed run 37056394903: the multi_server SSH algorithm case failed AddrInUse (95 passed, one failed, nine ignored; 470.72 seconds). Its HTTP fixture probed a port and dropped the listener before startup. The same fixture now binds port zero in the server and uses the actual bound address; no authentication assertion or deadline changes. Both CI runs passed at 07be856: [run 1](https://github.com/crabbuild/canopy/actions/runs/37060379452) and [run 2](https://github.com/crabbuild/canopy/actions/runs/37060375239). Those checks precede the fresh initialization increment, which requires its own CI. + +## Coalesced existing ref batches + +The shared tree now uses one streaming builder for initial imports and existing-base sorted upserts. Ref preparation validates expectations and namespaces in name order, retains the canonical digest of the original intent order, and consumes the sorted borrowed map directly. The walker opens affected paths and emits unchanged child references without scanning their leaves. Its frontier lifts pending lower groups before higher subtree reuse, preserving range order and child heights under the existing fanout/byte/height bounds. It keeps an active block plus one completed block at each level so small final tails can be balanced by bytes and fanout, retaining bounded memory without degrading repeated prefix insertion density. This remains a conditional data plane; the authoritative final root transaction and producer/reader hard cutover remain open. + +The original 512-update/20,000-ref regression failed the node-write bound in 11.21 seconds. It retained exact expected versions and namespace rules; the per-update path-copy loop produced too many immutable artifacts. The first all-target compile check passed (1m 34s). An explicit future-Send assertion then exposed a borrowed mapped-iterator lifetime error; consuming map values in a named streaming adapter makes those lifetimes explicit without removing the assertion or allocating another record inventory. The first closure-only typing adjustment remained insufficient. All 12 focused ref-state tests passed (25.69 seconds), including the unchanged original write-bound regression. Four new cases cover 512 reversed updates against 20,000 refs within 40 new artifact objects/loaded nodes, sparse prefix/gap/suffix/namespace changes against 100,000 refs within 100 new objects and 64 loads, late-input failure with immutable retry/idempotency and empty/disordered inputs, and long-name multi-level byte bounds in both formats. The existing large-plan test now rewrites all 20,000 refs against a nonempty base within 500 new objects and 600 loads, checking every name/version/tip. Independent standard BTreeMap models check complete sparse/long inventories. Both formats retain old roots, exact plan digest and expected versions. The explicit future-Send assertion passed. The final library suite passed all 443 unique tests (6 Git-format, 14 object-storage and 423 server; server 189.50 seconds). Two additional density checks preserve 544 refs within 11 cold-loaded nodes after 32 prefix insertions and 17,024 refs within 145 nodes at height two after 640 insertions crossing an internal split. The original leaf-density repro failed with 37 nodes in 0.44 seconds; retaining and balancing the final pair fixes it without weakening the read bound. Initial and existing-bulk write bounds still pass after the grouping change. All-target workspace Clippy passed with warnings denied (39.38 seconds). Rebuilt stock-Git smart-HTTP passed (61.99 seconds). Signed-push/registered-key/audit-after-restore integration passed (4.01 seconds). Formatting and diff checks passed; all 10 changed/new files passed whitespace checks and 51 local document links resolved. Coalescer CI produced one success and one AddrInUse integration failure, recorded above. + +The first broader coalescer run passed 440 unique tests and failed the existing queued-spool cancellation check (420 server passed/one failed; 171.22 seconds). An isolated run passed, then the unchanged case reproduced on repetition 15. Spool destruction released disk credit before transfer admission, so zero usage became observable during that cleanup gap. File-first, admission-next, disk-last field order removes the gap while retaining ownership through queued work. The unchanged cancellation case passed 1,000 repetitions in 34.33 seconds and the final full suite. The earlier full run remains failed; deadlines, resource bounds and cancellation assertions were preserved. + +## Immutable ref state implementation + +The [immutable ref-state data plane](design/immutable-ref-state.md) now reuses the shared authenticated tree and existing versioned ref structures. It adds variable-length ordered keys, encoded-byte splits, retained deletion versions, live-subtree summaries and streaming sorted construction for large initial inventories. A bounded snapshot stores the longest default branch and root fences outside the command envelope. These are conditional primitives; the serving path still uses SQL refs. + +Final publication must switch catalog, ref and response roots together while checking current authority and policy facts. It must remove both full-plan command transport and the per-update SQL loop. Existing-root coalescing is now implemented through the shared streaming rewrite. Sustained bulk/provider/hot-root qualification, every producer/reader conversion, full retained-root collection and mixed-load capacity gates remain required. The final workspace library run passed all 437 unique tests (6 Git-format, 14 object-storage and 417 server; server 173.27 seconds). Eight new tests cover exact versions and delete/recreate ABA rejection, atomic namespace swaps and Unicode names, a 20,000-ref initial plan beyond the inline envelope, one-path cold lookup/seek and bounded single-update writes, 40,000-record tombstone seeks and namespace conflicts, long-name leaf/internal byte splits, large snapshot fences, narrower typed-root bounds, purpose/format/repository/digest refusal and sorted-builder ordering/zero-live authentication. Existing catalog/native-input/recovery/publication tests also passed. + +The first focused run passed four tests and failed two: the new large-plan fixture requested an encoder limit beyond the runtime codec's permitted maximum, and a live seek lost a later sibling (13.58 seconds). Bounded range encoding corrected the fixture without increasing any production limit. The expanded pre-fix run passed six tests and failed both the cursor regression and its actual namespace-check seam (12.51 seconds). Advancing from an exhausted seek branch to later ancestor siblings fixes the missed conflict; both unchanged correctness regressions passed in the full run. No existing typed-root bound, deadline or integrity assertion was weakened to make these checks pass. The new snapshot has its own 256 KiB cap, while wire-request and native-result wrappers retain their original limits. All-target workspace Clippy passed with warnings denied (1m 25s). Formatting and diff checks passed. Rebuilt stock-Git smart-HTTP passed (142.36 seconds). Signed-push/registered-key/audit-after-restore integration passed (4.57 seconds). All 24 changed/new files passed whitespace checks and 53 local document links resolved. Both Rust/harness CI runs passed at 487ebd2. Fresh CI is required for the subsequent coalescer changes. These timings are correctness-suite observations, not large-team throughput measurements. + +## Durable native result validation + +The [native result checkpoint](design/durable-native-result.md) reuses the original wire root and existing completion/plan structures. It retains the exact response, options, versioned ref plan and scoped native signature witness before Bind; attaching it freezes the captured descriptor inventory. Plans stream in at most 64 KiB frames through an admitted spool instead of a second full encoded vector. Recovery requires exact MAC-registered custody, authenticated bodies and fresh logical scope; signed recovery also requires the exact tenant/application Directory with an enabled account and write-scoped key. Whole-root adoption preserves the original artifacts without copying, independently of source-pin expiry after destination registration. + +The final workspace library run passed all 429 unique tests (6 Git-format, 14 object-storage, 409 server; server 192.20 seconds). Five new checks cover a 20,000-update plan larger than the 4 MiB inline envelope and multipart native response in both formats, frozen input inventory and staging/bound reopening, refusal before exact checkpoint registration and after body corruption/repository revocation, scoped signer lookup with foreign Directory/account/scope/key revocation, malformed frame/count/actor/trailing data and disk rejection, and all roots plus adoption/predecessor/maximum actor within the unchanged 1 KiB envelope. Existing queued-cancellation coverage now includes the consuming plan reader. The restored-owner test recovers response/plan/options after old-pin expiry; the actual native receive/publication/cold-clone/fsck composition now publishes the recovered completion in both formats. Synthetic completions and signer witnesses qualify custody/retention, while native validity comes from the independent native composition and signature checks. + +All-target workspace Clippy with warnings denied passed (1m 54s). Rebuilt stock-Git smart-HTTP passed (64.98 seconds), and the signed-push/registered-key/audit-after-restore integration passed (3.54 seconds). Formatting, diff checks, changed/new-file whitespace and local documentation links passed. Both CI runs passed at the native-result commit 1319d0a, including Rust and harness jobs. Fresh CI is still required for subsequent ref-state changes. + +The initial full run passed 428 tests and failed the new maximum-envelope fixture (408 server passed, one failed; 155.22 seconds). Its synthetic repository ID lacked canonical UUID bits, so token validation refused it before envelope sizing. The corrected fixture preserves every size/context assertion and maximum actor, roots, adoption and predecessor fields. The earlier run remains failed; no production limit or deadline was increased. + +The final short ref-plan command interface, production producer/reader selection and takeover orchestration, complete collection/isolated restore, whole-operation containment, read acceleration, continuously fair maintenance and full-history/large-team mixed-load qualification remain mandatory. This increment does not complete the hard cutover or establish production capacity. + +## Owned production push preflight + +The [durable original-request API](design/durable-push-request.md) now retains the same encoded spool before native receive and registers a request-only checkpoint. Captured native descriptors append through the existing path-copy input index, preserving the exact request root, with predecessor CAS and a 256-revision bound while unbound. The staging service reuses its completed checkpoint slot without replacing pending/uncertain commands or old observer receipts. Staging/bound reopening authenticates root/body, rechecks custody and recomputes the original request identity. Actual restored-owner adoption preserves original roots after source-pin expiry. Request and artifact digests share one 64 KiB hashing scan. The [durable native-result API](design/durable-native-result.md) retains the versioned plan in bounded frames, exact response/options and scoped signature witness, freezes the captured input inventory, and reconstructs under current checkpoint custody and Directory key authorization. The final short ref-plan command interface, production selection/orchestration, full collection/isolated restore and capacity qualification remain open. + +The final library run passed all 424 unique workspace tests (6 Git-format, 14 object-storage, 404 server; server 145.67 seconds), including six new tests for large plain/gzip request recovery and native append in both formats, restored-owner request adoption after source-pin expiry, late-part corruption/partial-spool release and revoked custody, absent/lost/panicked append recovery with original observer receipts, canceled queued hashing/upload-open ownership and exact SQL predecessor/phase/revision guards. The real SHA-1/SHA-256 receive/publication/cold-clone/fsck composition now registers the request before native receive and appends captured pairs. Request fixtures contain synthetic signature/pack text; they establish byte/intent retention, not native validity. All-target workspace Clippy with warnings denied passed (29.37 seconds). Stock-Git smart-HTTP passed on the rebuilt binary (44.88 seconds), including production request identity/replay and push/clone behavior. + +Initial validation found the completed checkpoint slot rejecting the real native append with Duplicate; the original regression reproduced twice, and the exact-predecessor replacement fix passed the real composition. A new SQL regression showed that checking only OLD.generation allowed simultaneous Bind/append; requiring both old and new phases to remain unbound fixes that bypass. The first recovery fixture used time zero for a wall-clock mutation identity and now uses the existing clock helper. The first broad run passed 403 server tests and failed one corruption fixture (180.17 seconds): its body was below the actual 8 MiB part boundary, so its added second key was never read. The fixture now derives its size from PART_BYTES and asserts that the second part exists by size. The final run passed the unchanged corruption/refusal assertions. The earlier broad run remains failed; no deadline, authority or integrity assertion was weakened. These checks do not qualify production cutover, remote durability, isolated restore or full-history/large-team throughput. + +The production gateway now consumes EncodedPush and PushPreflight, reusing the existing disk-accounted GitInput, BeginRequest and command parser. The v3 request digest binds the canonical Cell context, repository format, actor, logical operation, HTTP metadata and exact encoded bytes before completed replay. Decoding consumes the encoded owner; handle_push receives normalized input, parsed intent and immutable identity together. The parser checks repository-format OIDs before converting zero IDs into create/delete intent, including signed and shallow commands. Signature verification remains native. RepositoryModule's source digest now includes preflight, push handling, command parsing and input spooling. See the [owned preflight contract](design/owned-push-preflight.md). + +Five new tests cover scoped identity changes, both-format normalization and native spool rewind, mismatched signed/unsigned/shallow and all-zero OIDs, signed options, invalid authentication/scope and malformed or oversized gzip with disk release. The stock-Git HTTP fixture now admits only the encoded spool on a completed gzip retry, proving replay does not allocate a decoded spool. Changing only gzip header metadata under the same completed operation must conflict without advancing refs. + +All 418 unique workspace library tests passed on the final source (6 Git-format, 14 object-storage, 398 server; server 113.20 seconds). Stock-Git smart HTTP passed (37.68 seconds), the sequential signed/push-option selection passed four tests and explicitly ignored one isolated provider check (9.20 seconds), and the sequential SHA-256 selection passed five tests and explicitly ignored three isolated provider checks (18.38 seconds). The latter includes actual signed SSH push with fresh-disk restore, HTTP push/clone/fetch/restore, reviews and generated merge candidates. Final all-target workspace Clippy with warnings denied passed (29.93 seconds), along with formatting and diff checks. The first focused five-test run passed; the first 418-test library run and stock-Git HTTP run also passed before source-digest coverage and the all-zero regression extension. No runtime failure or widened assertion/deadline occurred in this increment. Fresh CI is required for the new commit; the preceding lifecycle CI is recorded below. + +This handoff still enters the selected legacy publication path. Connecting it to packed native capture and the supervised lifecycle, an authenticated immutable wire-plan root for large ref plans, exact response/signature reconstruction, every producer/reader conversion and fresh-schema selection remain open. Existing parser bounds do not qualify whole-operation RSS, full history or large-team capacity. The v2 digest is replaced directly, without a replay compatibility adapter; the required release is a fresh-data hard cutover. + +## Final publication lifecycle ownership + +StagingTicket::publish now seals bound preparation and transfers a private final push/compaction command into held admission in the existing PublicationCoordinator. It requires the lifecycle's exact shared session fence/clock/ceiling and returns the original ready value on refusal. The staging job stores a small ticket; the fair coordinator charges and retains the existing proof/command layout. Existing workers/results, due renewal and accepted checkpoints drain before activation, with a final query at the latest known custody receipt. Accepted final intent continues through close. An observation-only ticket and pending_publication recover canceled observers. + +Held activation/discard are serialized and remain possible after close. Discard succeeds only before dispatch and drops resources before credits. Final transport checks local custody before initial execution and after authoritative absence, while resolving known committed outcomes first. Unknown outcomes retain both services' reservations; staging recovery queues the same exact publication ticket. Published preserves the original outcome without querying the retired preparation record, then fences/releases the lifecycle. See the [final lifecycle contract](design/final-publication-lifecycle.md). + +The final workspace library run passed all 413 unique tests (6 Git-format, 14 object-storage, 393 server; server 109.79 seconds). Thirteen new tests cover held quota/contention/foreign/duplicate/closed refusal with original ready-value reuse, canceled observations, resource-drop ordering, activation/discard races, final worker/result drain, exact renewal/checkpoint ordering and closed recovery, original final receipts after revocation or local expiry, initial queued-expiry refusal, shared-coordinator recovery and maintenance publication. Existing native receive now constructs its final proof in a bound-owned worker, publishes through this handoff, then performs a cold stock-Git clone and fsck for both formats. Existing maintenance uncertainty/class-budget tests now also exercise held admission. All-target workspace Clippy with warnings denied passed on the final source (26.15 seconds). Formatting, diff checks, whitespace for 20 changed/new files and 59 local documentation targets passed. + +An initial held-test compile attempt had fixture mistakes (unshared count helper, request Clone, typed tenant and Debug-bound unwrap_err); corrected using existing helpers/types. The first broad publication run passed 151 and failed three new tests (119.63 seconds): internal response access was incorrectly treated as external authorization, and virtual-clock jumps expired unrelated SDK deadlines. Tests now assert internal original-response preservation alongside externally authenticated replay denial, use a tighter one-second lifecycle profile for guarded recovery, and isolate each virtual-time scenario in a fresh runtime so offsets cannot accumulate across cases. One native-fixture compile attempt could not convert its non-Send test identity error inside the owned worker; preparing the identity before worker admission corrected it. The final focused five-case lifecycle run passed (2.40 seconds), followed by 412 workspace library tests (server 118.39 seconds). The final shared-coordinator recovery change and added regression then passed Clippy and all 413 tests above. No authority assertion, SDK deadline or production lifetime bound was widened. The preceding bound lifecycle commit 0c38d4a passed both CI runs, including workspace integrations, Python harness and RustFS compatibility. Both CI runs passed at b2a7949, including workspace integrations, Python harness and RustFS compatibility: [run 1](https://github.com/crabbuild/canopy/actions/runs/37013508378), [run 2](https://github.com/crabbuild/canopy/actions/runs/37013518780). Production producer/reader conversion, durable wire-plan/response/takeover reconstruction and fresh-schema selection remain open. Complete collection/drain/isolated restore, physical rewrite/read acceleration, resource containment, continuous hot-root progress and full-history/large-team mixed-load campaigns remain required before hard cutover. + +## Bound lifecycle ownership and renewal + +StagingCoordinator now retains operation/actor admission after Bind and accepts ReadyStaging::claim_bound for bound takeover. The same job, exact command slot, actor worker semaphore, WorkSlots and checkpoint slot own bound Claim/Renew/registration, callbacks and retained typed results through cancellation and shutdown. Bound outcomes preserve their original receipt before fresh token/floor/format queries. open_base shares the supervisor session rather than opening an unrelated deadline/fence. spawn_bound uses the existing worker/result path; phase handoff invalidates staged contexts. Known bound renewal and checkpoint receipts survive failed fresh custody without authorizing proof reuse. See the [bound lifecycle contract](design/bound-preparation-lifecycle.md). + +bound_lifetime_ms adds a separate immutable local residence ceiling from before Bind preparation or bound Claim admission. Default is 60 seconds, with nonzero profiles up to MAX_LEASE_MS; the existing overall session lifetime also applies. Session factories, bases and reads respect the ceiling, including direct renewal/refresh. Renewal schedules use the raw freshly observed SQL deadline, preventing a tight loop when local use is clipped to the ceiling. Graceful close keeps renewing until callbacks and retained results drain, then fences the shared session before releasing operation admission. SQL pins retain their independent expiry; this local limit does not prove floor-capacity qualification or remote deletion authority. + +The unchanged nine staging tests passed (1.36 seconds). All 24 focused service checks passed (10.61 seconds), including eight new bound lifecycle checks for automatic renewal without caller ownership, canceled workers/results and graceful close, exact absent/lost/panicked Claim and Renew with closed recovery, revoked/expired/superseded custody with original receipts, phase and residence fencing of shared native bases/inflight work, reused actor worker/result caps and resource-drop ordering, serialized bound checkpoint adoption/recovery with source-root reuse, and actual Cell owner restoration followed by bound workers/renewal. Native catalog assembly now runs in a bound-owned worker before its existing attestation checks. The first compile attempt exposed a local deadline borrow conflict, fixed by reading it before moving the shared session. The first new-test attempt had three test-plumbing errors (Receipt field access and a Result alias); corrected without widening deadlines or assertions. No focused runtime test failed. All 400 unique workspace library tests passed (6 Git-format, 14 object-storage, 380 server; server 128.60 seconds) after the ceiling guard and handoff-helper extraction. The final reader-before-fence regression and post-query ceiling recheck passed all-target workspace Clippy with warnings denied (31.84 seconds) and all 24 focused service checks (9.31 seconds). Formatting, diff checks, whitespace for 13 changed/new files and 51 local documentation targets passed. Both CI runs passed at 0c38d4a, including workspace integrations, Python harness and RustFS compatibility. + +Final-publication lifecycle serialization now exists; production producer/reader conversion and durable wire-plan/response/takeover reconstruction remain open. The fresh schema is still unselected. Complete collection/drain/isolated restore, physical rewrite/read acceleration, resource containment, continuous hot-root progress and full-history/large-team mixed-load campaigns remain required before hard cutover. + +## Bound Claim/Renew dispatch + +ReadyPreparation::claim and PreparationSession::ready_renew now retain exact commands 12/13 in the existing PublicationCoordinator with an 8 KiB foreground reservation. They reuse account/operation/fair queue admission, close/cancellation ownership and SDK recovery. PreparationCommandOutcome preserves the original committed reply and receipt separately from a freshly queried session. Renewals share the existing deadline/fence and preserve the original floor; Claims expose a new attempt/floor only after fresh token/base/format matching, retaining the previous independent pin. Rejected renewals fence the original session; ambiguous commands keep their existing deadline/evidence. No known commit is erased by failed fresh custody, and no preparation outcome can acknowledge a Git push. See the [bound preparation contract](design/bound-preparation-dispatch.md). + +The first focused run passed five new checks (1.85 seconds), covering both formats, exact absent/lost/panicked recovery, closed and canceled observation, refused command reuse, committed outcomes with revoked/expired/superseded custody, absence with current authority checks, expired-source Claim, bounded factory refusal and permanent local fencing. An additional fixture restores the actual Cell owner, recovers a lost Claim reply, checks the new owner/creating namespace and old pin, and renews the restored session. The real retained-input publication fixtures now acquire bound sessions through supervised Claim. All 392 unique workspace library tests passed (6 Git-format, 14 object-storage, 372 server; server 136.79 seconds), including all six new checks and the retained-pair publication fixtures. Initial Clippy rejected two nested-if lints and the enlarged ReadyPublication variant. The private preparation request is now boxed, following the existing staging pattern, and the nested guards are collapsed without suppressions or changed authority/deadline checks. Final all-target workspace Clippy with warnings denied passed (41.67 seconds). All 17 focused bound/admission/ownership/recovery/retained-input checks passed on the final representation (8.89 seconds), including all six new checks and both-format physical publication. Formatting, diff checks, whitespace for 13 changed/new files and 49 local documentation targets passed. Both CI runs passed at 7353e55, including workspace integrations, Python harness and RustFS compatibility. + +Automatic bound renewal scheduling, maximum lifetime/floor residence and preparation worker ownership, serialization with checkpoints/final publication, production integration and durable takeover reconstruction remain open. Process-local command ownership does not survive process loss; this is not full-history, restore/collection or large-team capacity qualification. + +## Bound checkpoint dispatch validation + +PreparationSession::ready_inputs now retains the exact adopted-input registration command in the existing PublicationCoordinator. It shares foreground account/operation/fair queue limits and uses an 8 KiB reservation instead of a full push's 8 MiB. Each admitted job reuses the existing structure with its factory-derived reservation. Cancellation, close and uncertain recovery retain the command and session; refused admission returns the same ready value. RegisteredNativeInputs keeps the original committed result alongside fresh input-pin/bound-session custody. Revocation, expiry or Claim can fence the shared session without losing that original receipt. Checkpoint outcomes cannot become a push response. See the [checkpoint contract](design/native-input-checkpoint.md). + +All 386 unique workspace library tests passed (6 Git-format, 14 object-storage, 366 server; server 119.49 seconds). Six new bound checkpoint checks cover both OID formats, absent/lost/panicked exact dispatch, canceled observers, closed recovery, refused ready-value reuse, committed recovery with failed fresh custody, denial after authoritative absence without attaching inventory, real retained-pair verification/publication after source-pin expiry and mixed checkpoint/push byte/account admission. The existing 300-record bound-adoption test now registers through the dispatcher and reuses its exact root. The earlier focused run passed six checks (five new plus that existing test, 9.32 seconds). Both CI runs passed at the preceding capture-fence commit f59617d and custody commit 63299e3, including workspace integrations, Python harness and RustFS compatibility. All-target workspace Clippy with warnings denied passed (39.96 seconds). Formatting, diff checks, whitespace for 13 changed/new files and 44 local documentation targets passed. Both CI runs passed at df7d79f, including workspace integrations, Python harness and RustFS compatibility. These fixtures do not qualify production selection, complete collection/isolated restore or full-history/large-team capacity. + +Service-owned long bound-preparation renewal, bound Claim orchestration, production producer/reader wiring and durable takeover reconstruction remain open. Registration does not renew custody, advance an original generation floor, or grant remote deletion authority. + +## Native capture fence release validation + +Completed capture now explicitly unlocks its exclusive native fence when the final owned InputFile pin drops. Unix CLOEXEC descriptors can survive in an unrelated child paused before exec, so closing only the parent file can retain a completed capture's lock and spuriously refuse the next native admission. The deterministic regression fails before the fix with WouldBlock on macOS and Linux. It also verifies that a remaining owned input still excludes admission, then verifies admission succeeds after final capture release while the unrelated child remains paused. Active native workers' shared locks and descendant drain, queued upload ownership and production deadlines are unchanged; see the [capture contract](design/native-input-capture.md). + +All 380 unique local workspace library tests passed (6 Git-format, 14 object-storage, 360 server; server 120.67 seconds). All-target workspace Clippy with warnings denied passed on the cleaned source (29.98 seconds). The first macOS regression attempt failed because its hard-coded /bin/true path was absent; using the executable lookup corrected the fixture before the observed WouldBlock failure. The pre-fix Linux server run passed 357 tests and failed three (225.22 seconds): the new regression and two permissions fixtures run incorrectly as root. The original capture rejection fixture passed in that run. This establishes the inherited capture-lock defect, while the exact failing phase in the earlier CI remains unproven. The final unprivileged Linux Docker workspace library run passed all 380 unique tests (6 Git-format, 14 object-storage, 360 server; server 147.38 seconds), including the original capture fixture and both permissions checks. That run used Rust 1.97.1 and Git 2.39.5; local macOS used Rust 1.98.0. Diagnostic instrumentation was removed before final Linux validation and Clippy. Formatting, diff, changed-file whitespace and local documentation targets passed. Both CI runs passed at f59617d, including workspace integrations, Python harness and RustFS compatibility. This increment is not a full-history, production cutover or large-team capacity qualification. + +## Retained physical input custody validation + +Authenticated adopted-input custody now composes with physical verification, closure preparation, reconciliation and final publication; see the [checkpoint contract](design/native-input-checkpoint.md). The private bridge checks exact descriptor membership and live successor authority. Catalog certificate v3 binds the checkpoint digest, and final publication rechecks the independent pin in its transaction. Original native/metadata namespaces remain intact; successor index nodes use the new namespace. Ordinary raw namespace rejection and completed original-receipt recovery remain enforced. + +The final workspace library run passed all 379 unique tests (6 Git-format, 14 object-storage, 359 server; server 152.75 seconds). All six new custody checks passed, including moving nonempty bases and cold Git clone for both OID formats. The actual owner-restoration test now publishes retained physical inputs after old-pin expiry. The initial run passed 378 tests and failed its final assertion because publication correctly retires the active operation. The corrected fixture checks the checkpoint before publication, then verifies the absent active query, retained independent pin and original exact registration receipt after completion; production guards and deadlines are unchanged. Final all-target workspace Clippy with warnings denied passed (1m 09s). Both CI runs passed at 63299e3, including workspace integrations, Python harness and RustFS compatibility. Production producer conversion, service-owned bound Claim/renewal and durable takeover reconstruction, complete collection/isolated restore and full-history capacity remain open. + +## Staging checkpoint supervision validation + +The final workspace library run passed all 373 unique tests (6 Git-format, 14 object-storage, 353 server; server 145.02 seconds). All five new service checks passed: canceled observers and exact absent/lost/panicked registration recovery in both OID formats, registration-before-Bind ordering, one-slot/foreign/duplicate/closed refusal, original committed recovery with revoked current access, and authoritative expiry after absence without attaching an inventory. The real SHA-1/SHA-256 receive/publication/cold-clone fixture now uses service-owned registration. The initial run passed 372 unique tests and failed the new expiry fixture (server 352 passed/one failed, 172.62 seconds): changing only its pin expiry violated the existing deferred operation/pin foreign key. The corrected fixture changes both expiries in one transaction; production deadlines, guards and assertions are unchanged. Final workspace/all-target Clippy with warnings denied passed (1m 22s). Formatting, diff, changed-file whitespace and local documentation links passed. Both CI runs at c200d63 failed the existing native capture rejection fixture with Input(Io(Kind(WouldBlock))) after 352 server tests passed and one failed (101.72 and 99.44 seconds). Those c200d63 CI runs remain failed. The later f59617d regression/fix and green CI establish the inherited exclusive-lock defect; the precise phase of the earlier failures remains unproven. + +Staging and bound checkpoint registration supervision are now implemented, but production producer selection, service-owned bound Claim/renewal and durable reconstruction remain open. ClosureVerifier::begin_pack still rejects raw foreign creating namespaces; the new authenticated retained-input bridge supports exact borrowed witnesses. Production producer integration remains required before the fresh schema can be selected. This result is not a production cutover, provider durability, restore/collection or large-team capacity qualification. + +## Input checkpoint increment validation + +The workspace library run passed all 368 unique tests (6 Git-format, 14 object-storage, 348 server; server 126.73 seconds). All seven new checkpoint checks passed, including actual owner restoration followed by independently downloading/decoding a real retained native pair after the old pin expires. The SHA-1/SHA-256 receive/publication/cold-clone fixture now persists and reads an input checkpoint. Synthetic 300-record indexes qualify bounded inventory traversal/envelope size, not native body validity or capacity. Final workspace/all-target Clippy with warnings denied passed (37.16 seconds). Both CI runs passed at b30feb0, including workspace integrations, Python harness and RustFS compatibility. Production checkpoint scheduling, durable wire-plan/response recovery, fresh schema selection, collection/restore and full-history capacity remain open. + +## Native input capture validation + +The initial workspace run passed all 361 unique library tests (6 Git-format, 14 object-storage, 341 server), two CLI checks, ten Directory Cell checks and two Git HTTP checks. All three new capture tests passed. The multi-server portion passed 84, failed 12 and ignored nine (420.76 seconds). It exposed loose-file/external-blob assumptions in existing producer fixtures when receive.unpackLimit=0 was applied globally; the same run also logged Cell deadlines and system-wide file exhaustion. Packed receive is now scoped to the explicit run_native_receive API until the production hard cutover. The two composition/rejection tests passed again on the final API (1.75 seconds), and the queued-upload ownership test passed separately. All 12 unchanged failing integration cases passed sequentially after the correction (360.52 seconds), including selective HTTP/SSH fetch, bulk mirror, large-object recovery, cross-gateway publication and disconnected/refused SSH pushes. The original concurrent workspace run remains failed; this is not a clean full-workspace or capacity qualification. Final queued-upload ownership passed in 0.10 seconds. Owner-restart, Repository Cell and stock-Git smart-HTTP passed on rebuilt binaries (4.11, 27.88 and 51.35 seconds). Final workspace/all-target Clippy with warnings denied passed (33.40 seconds). Formatting, diff checks, all 12 changed/new files' whitespace and 17 local document targets passed. + +## Outcome-only increment validation + +The outcome-only workspace library run passed 357 tests and failed one existing native ancestry test with Base(Inactive); server results were 337 passed/one failed in 240.26 seconds. All six new outcome tests passed. The unchanged ancestry case passed separately in 31.86 seconds with its original assertions and deadline. The concurrent run remains failed and is not a clean full-suite or capacity pass. The contract and tests preserve expiry rejection rather than widening the lease to hide the failure. Owner-restart and Repository Cell integration checks passed (5.60 and 43.36 seconds). Stock-Git smart-HTTP passed on a rebuilt test binary (80.15 seconds); its first attempt could not start because the binary was absent from the shared target directory. Final workspace/all-target Clippy with warnings denied passed (1m 12s). Formatting, diff checks, new-file whitespace and 31 local documentation links passed. + +## Implemented primitives and fixes + +- StagingTicket::register_inputs now synchronously admits one bounded input checkpoint and exact mutation identity into service ownership. The existing supervisor dispatches command 29, retains exact evidence on absent/lost/panicked replies and restores the original receipt through recovery. Due renewal precedes registration; accepted registration precedes Bind or graceful stop. A separate 4 KiB reservation accounts for the queued request/retained checkpoint result while the job remains admitted. pending_inputs recovers a dropped observer. Known registration preserves its original receipt before a fresh authority query, so revocation can fence the stage without losing committed recovery. The real SHA-1/SHA-256 receive/publication/cold-clone fixture uses this path. Bound registration now uses the publication dispatcher; production wiring and service-owned bound Claim/renewal remain open. + +- Creating input checkpoints now reuse NativePackDescriptor, the shared RangeIndex/NodeRef and existing independent catalog lease rows. A purpose-separated NativeInputCertificate binds the exact actor/request/owner/namespace and an authenticated input index. Fresh command 29 records an immutable envelope/digest after current write/fence/phase/expiry checks; query 30 verifies retained source custody without treating revocation as collection authority. ReadyStaging::claim shares exact service dispatch; staging and bound preparation adoption reuse the exact immutable input root without copying nodes, preserve native incarnations and check source expiry again in final registration. A committed destination checkpoint independently retains its referenced input namespaces. See the [checkpoint and adoption contract](design/native-input-checkpoint.md). The physical-input custody bridge is now implemented. Production staging checkpoint wiring, service-owned bound Claim/renewal and complete retained-root collection remain required before cutover/collection. + +- The new GitHttpBackend::run_native_receive API retains pack/index pairs even for small staged pushes. GitHttpBackend::stage_native_packs captures bounded request-private inputs under an admitted StagingContext namespace, sharing NativePackDescriptor, PhysicalLimits and the existing pinned-file uploader. An exclusive native fence keeps the cache immutable through hash/upload workers and cancellation; no decoded bodies, full OID inventory or legacy Git rows are created by capture. The [capture contract](design/native-input-capture.md) defines its limited authority and the remaining production/input-retention requirements. Production registry selection, producer/reader orchestration and large-team qualification remain open. + +- PreparationSession now reuses the authoritative lease/deadline/renewal capability independently of the catalog reader. Its outcome-only factory takes no loader, scratch, disk budget or native scope. A purpose-separated, bounded OutcomeCertificate binds the admitted token, actor, format, immutable floor and exact native response/options/signed witness. CompleteCatalogPush reuses command 19 and the existing response tables, checks current write authority before new outcome writes, publishes no catalog/ref changes and preserves original completed replay before stale-owner rejection. ready_outcome enters the existing foreground dispatch/recovery path and retains only the session and exact SDK command. See the [outcome-only completion contract](design/outcome-only-completion.md). Production HTTP/SSH wiring and capacity qualification remain open. + +- Native shutdown now uses the same node resource pool as every admitted worker. Closure serializes with admission and is permanent across all cloned scopes; drain requires healthy, zero foreground and maintenance vectors. All registered observers wake on final release, cancellation does not reopen admission, and poison/underflow/quarantine cannot prove drain. Normal shutdown and enrolled startup failure retain Cell authority, workspace exclusion and heartbeat renewal until tracked and native work settle, before Cell shutdown and advertisement withdrawal. The [native admission contract](design/native-resource-admission.md) defines this ordering and the pending-through-restart quarantine behavior. OS containment, account fairness, full-history profiles, production packed-catalog conversion and large-team capacity remain open. + +- Native process launch now requires one private vector permit for process slots, CPU admission units, memory and descriptors. The server validates required native_limits before workspace/provider work and shares one pool across gateways, caches, decoded/history actors, candidates, HTTP/SSH and ref listing. Preparation APIs require an explicit node scope; physical-index binding retains read admission in its blocking worker. Foreground and maintenance shares are disjoint. Completed repack workers release claims before validation, and returned caches preserve their original scope. Typed exhaustion becomes HTTP 503; canceled or uncertain drain retains claims. See the [native admission contract](design/native-resource-admission.md). Claims are estimates; hard OS bounds, native account shares/fair waiting, full-history profile qualification and production packed-catalog cutover remain open. +- The existing GitProcess is now one shared native process guard for transport, verification, decoded-object/history actors, cache maintenance, blob extraction, candidates and ref listing. Unix wait requires inherited completion-descriptor EOF before leader reaping; standard streams alone cannot establish descendant drain. Cancellation transfers the leader, descriptor and generic cache/input/account owner to a bounded observer-independent reaper, releasing owners only after reaping and EOF. Failed/shutdown/saturated reaping quarantines owners. Private child wait prevents early PID reuse; completion descriptors stay outside standard slots. See the [native ownership contract](design/native-process-ownership.md). The fence covers participating descendants, not helpers that explicitly close it; OS containment, native account/fair preparation scheduling, full-history profile qualification, platform qualification and production cutover remain open. +- Physical verification now reuses one admitted append-only dependency file per at-most-512-object metadata page instead of opening one file per structural object. Existing VerifiedObject witnesses retain exact private offset/length/digest ranges and rehash them on every SQL replay. Object writers remain exclusive through queued cancellation; failed storage growth/write poisons all ranges. File credit stays held until the last producer, witness or worker drains. Native SHA-1/SHA-256 512-commit tests exercise one-file retention, interleaved append/replay and reverse repeated replay; late second-range corruption rolls back an entire wide-tree batch. See the [edge-spool contract](design/verified-edge-spool.md). This bounds dependency files per physical page, not global file/process usage or large-team capacity. +- Ref ancestry now reuses the existing DiskBudget growth helper and visits/answers tables instead of reserving its entire database ceiling at construction. Initial charge is 192 KiB; parent batches, expansion, cached answers and at-most-512-key resets grow only after admitted, rolled-back capacity errors. A mutable walker borrow makes traversal exclusive within one proof operation. Pair answers bind to the exact StoredCatalog; different-catalog reuse rejects. Failure/cancellation permanently fences reuse and interrupts SQL while queued workers retain scratch ownership through drain. Live preparation checks also cover asynchronous identical-tip and cached-answer paths. Native SHA-1/SHA-256 tests cover memo reuse, foreign catalog rejection, denied growth as an error and canceled queued work; the 100,001-entry synthetic queue checks indexed paged cleanup and reuse under a 64 MiB budget with a 32 MiB ceiling. See the [growth and ancestry execution contract](design/admitted-sqlite-growth.md). Native commit-graph acceleration, full process admission, production integration and large-history/mixed-load qualification remain open. +- Metadata, directory and operation-local closure construction now reuse the existing DiskBudget/SQLite structures with admitted geometric growth. Each spool starts at 64 KiB (192 KiB including journal/overhead credit), doubles only after a completely rolled-back SQLite capacity failure, and admits the new capacity before raising the page cap. Commit precedes cursor/digest advancement; verified edge inputs remain owned and rehash on each replay. Closure lookup creation and pending-degree initialization use indexed pages of at most 512. Sealed files retain only their exact byte charge; existing cancellation/cleanup ordering remains. See the [growth contract](design/admitted-sqlite-growth.md). Three focused rollback/admission/ceiling checks and native SHA-1/SHA-256 fixtures exercise this path; full-history profiles, native/file/process admission, authenticated takeover adoption, production integration and capacity qualification remain open. +- StagingCoordinator now owns exact Begin/Renew/Bind commands and input producers independently of observers. Defaults admit 32 operations/eight per actor and 64 producer/result slots/eight per actor, reserving 8 KiB per exact command pair. Fresh authoritative queries establish renewal deadlines; sealing stops new workers and renews custody while tasks and retained typed results drain before Bind. Results remain charged until single handoff and are recoverable by internal typed lookup. Command uncertainty retains original identities and evidence, sharing exact dispatch/resolution with publication. Producer/authority failure cancels and joins tasks and drops completed resources before credits. The [staging service contract](design/staging-service-lifecycle.md) defines ownership, APIs, limits, exact recovery and close/drain. Nine native/lifecycle tests establish small-fixture composition, not production integration, process-loss reconstruction or large-team capacity. +- Long-running input custody now reuses the existing preparation token, creating namespace allocator, operation and lease rows with an unbound NULL generation. BeginStaging/RenewStaging/CheckStaging/ClaimStaging retain input namespaces without a catalog floor; BindStaging adds the current floor once, preserving namespace and expiry. A generated non-null binding value closes SQLite's nullable composite-FK escape. Regular preparation queries cannot open a staged base; final attestation/publication reject absent floors. Nine tests cover exact replay, owner restoration, quotas, revocation/expiry, immutable phase binding, 10,240 intervening facts with bounded reaping and pre-bind SHA-1/SHA-256 native physical witnesses feeding the existing private catalog proof. See the [staged input contract](design/staged-input-retention.md). The service renewal/drain primitive is implemented; production integration, authenticated takeover adoption, full-history growth qualification, remaining-floor configuration and full-history capacity are still open. +- The existing `PublicationCoordinator` now admits both privately prepared push completions and catalog compactions. `ready_compaction` retains the verified input and exact command 22; typed outcomes/errors preserve the command's original result/evidence. Both classes reuse one bounded queue, supervision, pending lookup, recovery and close/drain path. Defaults reserve four of 32 operation slots for maintenance, with 8 KiB command credits per compaction, at most two maintenance waits among eight total, and at most three foreground starts before eligible maintenance. Actor counts are separate by class; the same admin's foreground queue cannot consume its maintenance quota. Terminal tickets release proof ownership before credits; uncertain maintenance stays charged, while reserved foreground capacity remains available. Stats expose class occupancy and traces name the class. The [shared dispatch contract](design/shared-publication-dispatch.md) defines the new typed API and execution/recovery rules. Continuous producer/maintenance preparation, CPU/I/O shares, durable service reconstruction and production wiring remain open. +- `CompactionPlanner::prepare_next` now chooses verified bounded jobs from geometric per-level object targets and ingress pressure. Defaults use 262,144 objects at the first level, ratio four, an eight-root urgent watermark and a maximum three-job urgent burst before an eligible higher-level turn. Urgent jobs preserve level rotation; ingress slots rotate and per-level indexed cursors wrap after exhaustion. Fixed-size `CompactionPressure` reuses logical NodeRef summaries without claiming a global unique inventory. Invalid profiles, repository/format reuse and terminal-level over-capacity reject. Failed preparation leaves rotation unchanged; successful private preparation advances an advisory cursor without claiming durable publication. Current admin access is checked even for no work. See the [geometric maintenance execution contract](design/geometric-directory-maintenance.md). Production service dispatch, resource shares, uncertain-outcome orchestration, reconstruction and capacity gates remain open. +- CI for coverage commit `9cf90bd` exposed a queued partition cancellation cleanup race: disk admission could reach zero before the private workspace was removed. The shared admitted-file and reader fields now release workspace roots before disk/file-slot admission; failed cleanup still conservatively retains disk admission. The original lifetime assertions and five-second bound remain unchanged. +- `StoredRun` now always separates authenticated physical file facts from a contiguous logical `RunCoverage`, using the existing canonical header fold for both full files and projections. Directory-root v3 and range-index v2 reject prior layouts; fanout is 128, nodes remain at most 64 KiB and point selection remains at most 48 candidates. Multiple projections reuse one authenticated physical cache entry and disk charge. See the [persisted coverage contract](design/directory-run-coverage.md). +- `PreparedCompaction::prepare_range` promotes one queried ingress or level run into its adjacent level. When all overlaps exceed a job budget, it selects a bounded consecutive target prefix and moves a verified source prefix while retaining the exact verified suffix in the original physical file. One scan of the complete parent projection folds and validates parent, prefix and suffix before any fragment escapes. Physical byte admission deduplicates shared files; target record admission remains bounded. Replacement path-copies the source/target indexes and preserves unrelated files, old roots, sources and refs. Reconciliation permits unrelated level updates but rejects changed source or overlapping target inputs. Disjoint promotion verifies and reuses the original artifact without merge scratch. Command 22/query 23, certificate purpose, checkpoints and outcomes are reused. Source and first overlapping target files must fit the physical job budget. Full-parent scans repeat per window; geometric advisory selection is now implemented; amplification qualification, continuous fair service scheduling and long-lived inputs remain open. +- `PreparedCompaction` now merges 2–32 selected ingress roots from an authoritative pinned catalog into one replacement root. It streams authenticated run descriptors, verifies each input's count/range/canonical inventory, resolves overlaps through the existing directory builder and partitions the unique output through the existing range index. Defaults cap input at 128 runs/256 MiB, use a 256 MiB verification-spool ceiling (192 KiB initial charge, up to 768 MiB scratch reservation) and cap output files at 64 MiB. Sources, higher levels and refs are preserved. Reconciliation retains concurrent ingress and rejects if another compaction replaced a selected root. Certificate domain `canopy.catalog-attestation.v3` binds a maintenance purpose that cannot authorize ref publication. Admin-only command 22 atomically publishes the catalog and immutable logical compaction outcome; query 23 provides read-only recovery. Existing operation/floor/fence checks, optional checkpoint rebinding and SDK durable replay are reused. The native 32-root fixture frees 31 slots while preserving old-reader access. This is a bounded selected-ingress merge, not geometric higher-level compaction, physical pack rewriting, continuous maintenance orchestration or a capacity result. +- Directory snapshots now retain at most 32 overlapping **run-set roots**, with nonoverlapping runs indexed under each root by the existing persistent range index. Higher levels reuse that same representation, and point lookup still selects at most 48 runs. The current directory-root v3 codec rejects the old v1/v2 layouts; no compatibility adapter is retained. Catalog opening authenticates all bounded level-zero and higher-level roots. `CatalogPreparation::new_with_run_limits` separates output file size from the admitted verification-spool class. Its finish path verifies the complete private input run against the closure witness, streams and uploads bounded output files, inserts their descriptors into one private incoming root, and waits for exact output inventory/count/range verification before that root can escape. Reconciliation reuses the incoming root. Small pushes reuse their original sealed file; large output retains one 512-entry input page and one admitted output builder, preserving canonical headers, preferred source keys and placement versions. This implements bounded output partitioning; full-history incoming spool qualification, production input ownership, compaction/publication scheduling, production selection and capacity qualification remain open. +- `PreparedCatalog::ready_push` and `PublicationCoordinator` now reuse the exact typed Cellule prepared command rather than rebuilding a response/proof after ambiguous acceptance. The per-repository dispatcher admits bounded total/account/encoded-byte work, rotates ready accounts with FIFO account queues, and allows a bounded number of commands to await durability concurrently. Catalog reconciliation, native verification, ref-proof construction and uploads stay outside dispatch. Caller cancellation does not cancel admitted execution; pending/invalid published results and command-task panics retain the original command and verified inputs under admission until explicit evidence resolution. Recovery reexecutes only after authoritative absence; committed results keep the original receipt, while unknown/expired/changed-incarnation resolution remains uncertain. Terminal tickets release command/inventory resources before their quota becomes reusable and read exact responses without retaining scratch. Close/drain hands unresolved tickets to the service for recovery. Bounded stats and queue/residence trace events are exposed. Shared foreground/maintenance final-command fairness is implemented above; frontier pipelining, a durable local outbox, production wiring and capacity qualification remain open. +- `PreparedCatalog::reconcile` now privately retains the verified incoming directory/source roots and admitted closure scratch. It selects a current certified base through query 21 under the original attempt, shares renewal/deadline fencing across selections, and checks incoming canonical headers plus previously certified external anchors in indexed pages of at most 512. Missing anchors and changed body/kind/size/graph identities reject the proposed root. Reconciliation reuses physical verification and the incoming DAG, merges only the incoming source tree into the selected current roots, and uploads only new catalog/index nodes. An unchanged frontier reuses its reader and prepared root; initial source assembly does not copy its own tree. No full pack decode, historical graph scan, new namespace or durable Claim occurs. Local scratch remains charged until every prepared selection and queued worker drains. +- Conditional certificates now bind the original retention floor and its immutable certification digest separately from the actual selected base. The issuer freshly validates the original lease and selected generation fact; registration/final publication check the exact operation/pin floor, while publication CAS checks the selected current fact. A fresh certificate may publish a reconciled root while preserving an immutable optional checkpoint only when all verified input/inventory/count, actor, scope and attempt facts match; the checkpoint alone grants no new authority and its stored bytes remain unchanged. The maximum actor/ref/response binding still fits the 1 KiB envelope. Existing owner/ref/policy checks, atomic response completion and exact/logical replay remain intact. Reconciliation and bounded account-fair command dispatch are implemented primitives; frontier orchestration, changed-run-only acceleration, bulk input lifecycle and production hard cutover remain open. +- `CheckPreparationFrontier` (query 21) reads the original authorized attempt/floor and current immutable root in one committed snapshot without Claim, namespace allocation, lease extension or another durable command. Its bounded codec checks repository/format, generation order and same-generation identity. Existing independent generation pins now retain the full range at or above their immutable floor, including detached Claim/Abort pins and expired pins until reaped. An indexed minimum over those floors bounds at-most-512-fact cleanup; a schema DELETE guard protects the range and reserved empty root from other SQL writers. These are reconciliation prerequisites. Private incoming-descriptor retention, selected-base certificate rebinding and bounded canonical/dependency reconciliation are implemented below; frontier orchestration and production invocation remain open. Bulk inputs must have a separate long-lived retention lifecycle before production cutover: indefinitely renewing a busy repository's original floor would exhaust the 8,192-fact quota. No remote deletion or capacity claim follows from these changes. +- `PreparedCatalog::push_completion` now binds the exact native response ID/status/headers/body, options, ref-plan wire digest and optional signed bytes/signer/key into the conditional catalog certificate. The opaque `VerifiedPushCertificate` carries its actual Cell target and request digest; both this factory and existing completion reject a witness from another context. Native success reports must match the certified plan exactly; partial native refusals and sideband progress remain intact. Ref checks and completion use one final certificate issuance, avoiding an intermediate issuer query/signature. +- `CompleteCatalogPush` (typed command 19) invokes the authenticated catalog/ref publisher and saves the exact response, options, signed-certificate ownership/chunks and completed logical outcome in that same transaction. The ordinary inline path adds no response/plan/certificate staging commands; Begin plus final completion remains the two-command minimum. Current ACL/policy/ref refusals save a rewritten native report, including an explicit HTTP failure when report-status was declined. Empty-command/native-error outcomes publish no catalog generation. A changed catalog remains a retryable rejection with the operation intact rather than a permanent client refusal. Completed payloads, response bytes and signed records reject mutation/replacement; the fresh options field permits the full legal JSON-escaped note set. The service wrapper preserves uncertain command errors without placing a local preparation timeout around durable acknowledgement. `completed_push_response` reads the exact stored bytes at the committed receipt, validates chunk count/length/digest and binds the original logical identity/actor; it does not run the legacy rejection rewriter. Read-only `CheckCompletedPush` (query 20) reuses BeginRequest for preflight. `replay_push_response` rechecks current read access and loads the exact bytes at the observed receipt after restart, without constructing another preparation. Begin refuses completed IDs or conflicting pending identities before allocating a namespace. Original typed-command replay still returns its recorded outcome without rerunning the handler. +- Nine completion tests exercise SHA-1/SHA-256, exact native report framing, 600 KiB signed and response payloads crossing existing chunk boundaries, maximum escaped options, tampering/stripping, current permissions/policy, late-chunk rollback, signed-byte replay, native errors, checked no-op refs, moving-base retry, completed-request preflight without reallocation, and original outcome/body restoration after owner loss. Signed transport tests use a trusted native-witness fixture; actual signature verification remains in the existing gateway and its integration tests. These APIs are registered only in the fresh-schema fixture. Production HTTP/SSH invocation, reviewed merges, immutable roots for payloads exceeding the 4 MiB inline envelope, whole-operation resource admission and capacity qualification remain open. Outcome-only issuance now uses PreparationSession directly; production producers still need conversion to that path. +- `PreparedCatalog::ref_proof` checks new ref targets against its own prepared certified catalog, requires commits for branch heads, and binds the exact existing `PushPlan` wire bytes plus ancestry evidence into the conditional MAC certificate. Current fast-forward requirements are queried during issuance and checked again at publication. Nontrivial ancestry uses admitted SQLite scratch with indexed, at-most-512-entry traversal pages and disk-backed pair-result caching; incomplete or canceled work fails rather than returning a negative result. Chunked plan hashing avoids a second full-plan allocation. Tests cover both object formats, actual 1,200-commit native histories, a separate 100,001-entry scratch-queue/query-plan fixture, long ref-name wire equivalence and confirmed cancellation retaining file admission until the worker drains. The queue fixture is not a 100k-commit native ancestry qualification. Native commit-graph acceleration, large-history latency and whole-operation resource qualification remain open. +- `PublishCatalogRefs` (typed command 18) authenticates that exact ref-plan binding and rechecks the admitted owner fence, attempt/pin/expiry, current ACL, exact base/current catalog generation, expected refs and current branch/check/reporter policy. It atomically writes the immutable generation fact, current catalog, refs and original logical publication outcome in one Cell transaction. Every rejection precedes writes; later errors roll back the whole transaction. Exact command replay and logical same-plan replay return the original receipt, including after owner restoration. A prior catalog-only checkpoint can be retained while the fresh ref-bound certificate authorizes publication. Schema guards prevent replacing immutable generation facts and completed publication identities/outcomes. Seven tests cover both formats, tampering, missing/wrong-kind targets, policy changes, stale catalog/ref decisions, injected late failure, expiry, owner succession and replay. This command is registered only in the fresh-schema test module, not production. Its inline input limit is 4 MiB; large ref plans need an immutable plan-root interface. This refs-only command has no network payload; CompleteCatalogPush above composes its core with exact reports/options/signed bytes. Production invocation and reviewed merges remain open; separate ref publication and response commands would not satisfy the contract. +- Shared ref validation is now independent of the legacy object/ancestry SQL tables. It reuses current Write ACL checks, object-format validation, exact OID/version CAS, retained tombstones and indexed namespace paging. Only successful read-only validation constructs the private result; that result borrows the immutable plan and binds the actual target, owner fence and command execution sequence before applying writes. The production wrapper still requires certified graph roots and current branch/merge policy before consuming it. Three fresh-schema tests cover rejection without writes, tombstone/namespace replacement, stale CAS and a conflicting descendant beyond the first 256-row page. This shared validator remains separate from catalog membership and ancestry certification; the new ref-proof factory and typed publisher above compose those boundaries. +- The typed push API now allocates its independent graph-preparation, policy-proof and final-command futures separately. The repository integration test reproducibly overflowed its standard debug-build thread stack; debugger traces showed large nested polling frames without recursion. Separating those transport-heavy phases allows the original unmodified test harness to pass without increasing its stack. The independently encoded graph-command fixture also now uses the server's existing codec version 3, preserving its rejection/rollback assertions. These are serving-path correctness fixes, not measurements of the new publisher's throughput. +- Cellule stamps each activation-specific admission capability with `OwnerFence { incarnation, epoch }` and exposes it through `CellHandle` and `CommandContext`. Local typed commands and authenticated inbox effects capture the destination admission stamp; client bytes cannot select it. Existing admission identity checks prevent stale handles from executing against a successor. Migration and failed-transfer admission replacement preserve the stamp; activation rejects a stamp/control disagreement. Recorded outcomes bypass the handler and retain their original result across takeover. This adds no authority read per command and changes no persisted SQL or peer wire bytes. Preparation and the typed catalog/ref publisher store and check this fence; production completion must consume the same binding. +- `packs::publication` implements Begin, Claim, Renew, Abort, fresh lease queries and bounded reaping through Cellule's public typed command/query path. Its fresh schema candidate reuses product tables and removes legacy Git bodies and per-object graph/placement tables. Runtime-stamped incarnation/epoch/execution sequence distinguish attempts after takeover and record pruning. Independent generation pins survive replacement and abort until expiry; renewal checks the actual owner fence and cannot shorten or resurrect a lease. Deferred composite foreign keys bind each operation to its exact pin generation/expiry. Generation facts are immutable, bounded at 8,192 including the empty root; operations and pins are separately bounded at 1,024/4,096. Indexed reaping removes at most 512 rows per class, preserving current and pinned roots. It never deletes remote artifacts. These commands and the fresh schema are **not registered on the production serving path**; no compatibility Cell or dual writer was introduced. +- `PreparationBaseResolver` opens only through a trusted application CellClient and a fresh authorized lease query, validates the canonical repository target and exact token/root context, and reuses `CatalogReader::headers` plus `CatalogFiles` for at-most-512-OID batches. A monotonic deadline starts before the query; loading/resolution are timed and fenced on expiry. Renewal checks fresh query state after its durable acknowledgement, even when the command returns an old recorded success; failures fence the session. Eight tests cover native SHA-1/SHA-256 base reads, true owner succession and exact outcome replay, wrong incarnation/context, stale/recreated attempts, revocation, rebase retention, quota rollback, deferred pin-binding constraints and indexed bounded cleanup. Earlier trusted fixtures seed generation facts; new publisher tests now resolve actual atomically published bases. **Production orchestration and network completion are still missing**. Preparation pins do not close serving/backup retention, arbitrary clock jumps, suspended reader drain or remote collection safety. +- `CatalogPreparation` now composes complete physical partition witnesses, the query-derived base resolver, operation-local closure, disk-backed incoming directory assembly and authenticated source/catalog uploads. It accepts neither caller-provided output roots nor a generic BaseResolver/ClosureWitness. Every source leaf derives from an exact witnessed shard and its completed metadata upload. The incoming run's complete unique canonical inventory must match the internally produced closure witness. Shared source indexes copy changed paths; the pinned certified base's directory/source subtrees remain intact. Empty incoming operations reuse those facts without adding a run. The resulting `PreparedCatalog` has private construction, binds the operation token/base/catalog and canonical/input inventories, and remains **conditional preparation**. The certificate factory and typed publisher authenticate and publish those facts; production orchestration and network completion remain required. Physical witnesses now retain an ArtifactStore capability binding; matching repository/digests from a different backend cannot register missing destination pack/index references. This is an in-process capability check, not a persisted provider identity or cross-process attestation format. +- `PreparedCatalog::certificate` now issues a bounded conditional MAC certificate after fresh lease/ACL checks, reusing the trusted application SQL capability and repository secret with a separate BLAKE3 derivation domain. It binds tenant/application, repository, actor, admitted owner/attempt, base/catalog and canonical/input inventories. Raw decoded bytes remain untrusted until signature verification; raw descriptors cannot use the factory. The normal final publisher can carry it inline. Optional `RegisterCatalogAttestation` checkpoints at most 1 KiB plus its digest on the existing bounded operation row and its independent retention pin, rechecks live authority/fence/pin/base and rejects changed registrations. Claim clears the active operation checkpoint while preserving the old pin’s certificate; renewal preserves it; expiry/reaping and exact recorded replay do not confer fresh authority. Registration changes neither refs nor the current catalog. This module remains outside production registration. The typed publisher above now authenticates inline certificates and preserves existing checkpoints; the completion API above now publishes the exact network response, while production invocation remains missing. +- Creating artifact namespaces now use a durable per-repository allocation watermark, encoded in the existing 16-byte descriptor operation field as `CANOPY01` plus a positive big-endian integer. Begin/Claim allocate in their existing admitted transaction; duplicate Begin and recorded replay preserve the original allocation. The schema prevents counter rollback/deletion/replacement and enforces exact operation/pin logical-ID, owner-epoch, namespace, generation and expiry bindings. Explicit insertion guards cover [SQLite REPLACE deletion behavior](https://www.sqlite.org/lang_conflict.html), which can bypass delete triggers when recursive triggers are disabled. Each independent pin retains its namespace and optional certificate after Claim/Abort; its identity and registered certificate cannot be changed. The assembler and conditional certificate bind the creating namespace, while the existing logical request ID still keys durable outcomes. No hash truncation or permanent namespace-tombstone table is introduced. Normal owner recovery preserves the watermark; isolated rollback restores require a fresh provider namespace. This does not enable remote garbage collection or close recovery-root inventory/clock/drain qualification. +- Assembler scratch owns a private directory retained by admitted directory/closure files, including queued blocking jobs and uploaded file readers. Every async assembly phase is bounded by the live local preparation deadline; failure/cancellation poisons the builder. Six new tests cover both formats, exact three-shard source inventories, repeated inputs, empty-base and certified-base assembly, matching base overlap/external dependencies, missing closure dependencies, incomplete/out-of-order inputs, wrong operation/backend, admission failure and confirmed queued cancellation with retained workspace/admission. The incoming verification union uses one admitted directory spool; bounded output splitting, range compaction and geometric selection are implemented above. Full-history input profiles, continuous maintenance and whole-operation/native resource qualification remain required. No serving path or fresh deployment marker selects this assembler yet. +- `canopy-git-format::pack_index` checks native v2 indexes for both object formats using bounded buffers and position-based reads. Lookup does not retain a heap OID inventory. Index checksums and membership are not decoded-object, closure or authorization proofs. +- The serving cache uses checked file-backed index handles. Physical index entry counts explicitly include overlaps and are never a unique-object coverage proof. The count-based fetch coverage shortcut was removed. Overlapping indexes, repeated registration and source-cache teardown have regression coverage. +- `canopy-object-storage::artifact` supplies operation-scoped pack/index/metadata and directory/catalog-node namespaces, exact framed input, bounded 8 MiB parts, whole BLAKE3 verification, authenticated `CANOPY02` manifests, create-only collision checks and poisoned failed readers. The descriptor caps artifacts at 512 GiB and manifests at 65,536 part digests. Metadata and directory files share pinned, cancellation-safe upload/download code. The API is **not yet wired into the server's durable pack path**. +- `packs::metadata` implements immutable SQLite metadata shards, reusing canonical object fields and typed edges. Batches/pages are bounded at 512 records; sealing verifies an exact native index ordinal slice, local child kinds and structural edge shape, then binds inventory and file bytes. It supports SHA-1/SHA-256, retry-safe duplicate insertion and atomic conflict rejection. Native ordinal iterators can start directly at a shard boundary without rescanning preceding OIDs. +- Shard download binds repository/operation/pack/artifact identity, admits disk before transfer, verifies authenticated parts and the whole file, and opens SQLite read-only with bounded cache and no mmap. Queued blocking reads/writes/hash/open work retain spool pins through cancellation. Failed file/journal cleanup conservatively retains admission for workspace recovery. Eleven tests cover native fixtures, ordinal partitioning, wide-tree paging, corruption, conflicts, quota rollback, concurrent upload, canceled queued reads and cleanup failure. This is a storage primitive, **not trusted body verification, closure certification or publication integration**. +- `packs::directory` implements immutable canonical header/source runs. Source copying and merge pages are bounded at 512 records, canonical conflicts roll back a batch, and failed whole-source copies poison their builder. Local relocation compares exact expected placement, increments the location version and preserves the canonical inventory. Merge and lookup select the highest location version while still rejecting any overlapping canonical-header disagreement. +- The persisted directory range index uses authenticated nodes with fanout ≤128, encoded size ≤64 KiB and height ≤7 (at most eight nodes on a point path). Insert/remove copy affected paths; retained old roots remain readable. Cursors retain one path, seek without scanning earlier descriptors and poison on failed/canceled advancement. The node cache has at most 64 entries. Tests exercise split/removal, old-root reads, enclosing-range overlap, malformed codecs, forged summaries/context and cold path-read counts. +- Authenticated directory snapshots cap ingestion at 32 overlapping run-set roots and higher levels at 16 nonoverlapping range-index roots. Each ingress root indexes disjoint bounded output runs. Selection opens at most 48 run candidates and compares every matching canonical header before choosing placement. Root codecs reject oversized/truncated/trailing input. Sixteen shared-index/directory/snapshot tests pass locally, including a complete 48-candidate native fixture and stale relocation rollback. These components now compose with the prepared catalog and typed Cell publisher, but do not independently provide authorization/closure proofs or establish capacity; production serving remains unconverted. +- `packs::sources` implements the immutable source-descriptor tree by reusing the directory range tree with sealed typed 48-byte operation/digest keys and a distinct codec domain. Source leaves bind exact metadata, native pack/index descriptors and physical object count; fanout is 128, nodes are at most 64 KiB, height is at most seven and the cache holds at most 64 nodes. Conflicting bindings under one source key are rejected, and removal compares the exact descriptor incarnation. Constant-space coverage checks require a complete ordered ordinal partition and poison on failure. A native artifact binding check streams each physical pack once; remaining shards reuse its checked native index and endpoint checks. Directory-to-source resolution verifies the exact metadata descriptor and canonical header with admission retained by the metadata file pin. Five new source/shared-codec tests cover SHA-1/SHA-256 native fixtures, split/removal, retained roots, bounded cold lookup, separate domains, malformed fields/counts, corruption, incomplete/duplicate partitions and canonical-pointer conflicts. This source tree does not independently establish decoded-body verification, closure or authority. The assembler and typed publisher now compose it into published catalogs; the shared admitted loader below supplies local file admission. Production generation orchestration remains open. +- `packs::catalog` persists one root descriptor binding the directory snapshot and source tree, capped at 1 KiB with no predecessor-root chain. It reuses authenticated artifacts and the bounded Cellule codec with domain `canopy.catalog-root.v1` followed by NUL. Independently encoded SHA-1/SHA-256 golden vectors are saved under `docs/design/catalog-root-v1-*.hex`. `CatalogReader` validates repository/format, authenticated root summaries and directory/source/header consistency. Worker-owned index clients retain two bounded 64-node caches across snapshots, so an unrelated publication does not discard unchanged index nodes. Five tests cover native round trips, retained roots, cache reuse, malformed/foreign/dangling roots, corruption and exact wire bytes. Uploading/opening this root is **not publication, authorization, descendant verification or closure certification**; generation leases, certified facts and serving integration remain required. +- `CatalogFiles` is a repository/format-specific shared loader for authenticated directory runs and metadata shards. It caps open/reserved file slots (32 by default, hard maximum 128), including borrowed readers, evicted files and canceled blocking jobs; caches at most 16 files by default; and retains a private directory until the final file pin drops. The existing 8 MiB SQLite cache, mmap-disabled opens, file-size limits, whole-artifact verification and shared disk budget are reused. Fixed 16-way miss locks coalesce equal loads without a growing key registry. Cache hits compare the complete descriptor; an unborrowed file is evicted on the blocking pool when a slot is needed, and all-borrowed saturation fails admission. `CatalogReader::headers` groups at most 512 requested OIDs by selected directory/source file, uses one prepared indexed SQL statement per batch, and releases each file group before opening the next. Canonical disagreements across any selected run and preferred-source header mismatch reject the batch. Point and batch SQL reads run on blocking workers with pins retained through cancellation. Seven tests cover SHA-1/SHA-256 concurrent cache reuse across roots, single-slot batches, exact descriptor conflicts, borrowed saturation/eviction, failed/corrupt downloads, canceled queued downloads/SQL reads and a fully authenticated directory with a late source-header disagreement. The existing full 48-candidate and overlap-conflict tests also cover batches. This is a service-owned local file/cache primitive. The preparation adapter above reuses it; the loader itself is not an authorization boundary, certificate issuer or remote retention protocol. It now supports the typed preparation/publication pipeline; production serving remains unconverted. +- `packs::verification::CanonicalVerifier` streams native decoded objects through canonical Git hashing, body BLAKE3 and the new constant-state `graph::stream` parser. It retains a 64 KiB input chunk and a chunk-bounded edge buffer, emits batches of at most 512 typed dependencies, and uses persistent native reads with caller-owned file-backed index iteration. Tree names, commit messages/signatures and tag names do not become object-sized Rust allocations. Parser semantics preserve the existing typed graph rules, including gitlinks and duplicate occurrences; native bodies retain ordered commit parents. Failed or canceled inspections poison the actor, preventing another read or successful finish. Four verifier tests, three parser tests and one framing/hash test cover native SHA-1/SHA-256 inventories, 8 MiB commit messages, long fields, corruption, bounded batches and confirmed pending-sink cancellation; the existing 40 MiB blob test now exercises this path too. The disk-backed witness and isolated physical stages below build on this inspector; global closure and typed publication now compose through CatalogPreparation; production integration remains missing. These Rust bounds do **not** prove native child-process memory bounds or large-team capacity. +- `CanonicalVerifier::inspect_to_disk` now produces privately constructed `VerifiedObject` witnesses. Structural dependencies spool as fixed-width OID/type occurrences in batches of at most 512; blobs create no scratch file. Each write reserves disk before growth, and queued blocking jobs retain file ownership and admission through cancellation. The configured per-object edge-byte cap is checked before growth. Metadata assembly replays at most 512 dependencies per page and verifies the complete scratch digest before committing a batch of at most 512 witnesses in one SQLite transaction. This amortizes SQLite durability without loading wide-tree dependencies into memory. Late corruption, canonical/typed overlap conflicts and failed imports roll back the entire batch and permanently prevent sealing. Five added tests cover native SHA-1/SHA-256 1,600-entry trees, exact inventories, repeated witnesses, quota failure, canceled queued writes, late scratch mutation and rollback/poisoning. Callers still must retain admitted native resources, isolate the exact physical pack, finish the native actor successfully, certify closure and execute fenced Cell publication. The witness and metadata APIs are not a publication certificate or the completed serving path. +- `PhysicalVerifier` now orchestrates a fresh isolated native workspace containing exactly one authenticated pack/index pair and no alternates or loose objects. `NativePackDescriptor` extracts the existing shared artifact/context fields from `SourceRecord`, making native verification possible before metadata exists without changing source/catalog wire bytes. Pair download reserves all file bytes before provider reads, writes directly into the admitted cache without a second pack copy, and pins queued I/O through cancellation. Native `index-pack --verify` checks actual index/pack relationships after complete BLAKE3 and native checksums; the bounded canonical inspector then decodes every native ordinal and assembles exact contiguous metadata shards on blocking workers. Only complete coverage plus successful actor finish constructs `PhysicalPackWitness`, which binds the exact ordered sealed descriptors in constant space. Failed/canceled/incomplete assembly cannot finish or resume. Eight tests cover both object formats, 1,600-entry trees, three-shard coverage/source upload, descriptor tampering, native CRC lies under valid artifact/index hashes, provider corruption/admission, queued assembly cancellation, an ambient duplicate hiding an unresolved thin delta, native-completed thin packs including appended bases, and graph dependencies in other packs. This is the physical/canonical preparation stage, not global graph closure, authorization or fenced catalog publication. Producer integration, hard native RSS/whole-operation limits and capacity qualification remain open. +- `packs::closure` implements admitted operation-local typed closure, consuming only a complete `PhysicalPackWitness` and its exact sealed metadata partition. It reuses canonical headers/edge folds and the directory's inventory encoding. Input copying checks every sealed byte and canonical/typed duplicate; all incoming OIDs and dependencies are resolved in ordered batches of at most 512 against the exact base catalog/generation. Only incoming graph rows and requested base headers enter scratch. Indexed topological processing rejects missing/wrong-kind children and cycles without loading history or a heap graph; at most 512 vertex/edge updates share a transaction, and wide reverse fanout carries a constant-size cursor between batches. Queued workers retain file ownership/admission; cancellation interrupts SQL VM work and bounded file hashing. Scratch is explicitly disposable and uses `synchronous=OFF`, never as an acknowledged/recovered root. `ClosureWitness` checks the reused incoming directory inventory; its certification is conditional on a trusted authoritative base resolver and later fenced publication. Eleven tests cover both formats, native three-shard/repeated-pack and overlapping multi-pack inventory, external commit/tree dependencies, exact base binding and canonical conflicts, 10,000-deep chains, wide fanout, cycles, indexed plans, quota/full-disk cleanup, canceled queued/active work and changed sealed bytes. The preparation adapter above supplies query-derived conditional base resolution; the retained reconciliation inventory below now checks changed-generation canonical conflicts and anchors. Production orchestration, serving/backup pins and network publication wiring remain open. +- The shared external helper now checks copied destination parts before hashed-manifest publication, compares existing manifests exactly and exposes typed integrity-error classification. LFS reuses it and retains its corruption error semantics. +- Gateways own their pack readers and cache budgets. Repository body reads borrow a live reader through a weak registry. A reader from a different gateway workspace cannot be used as a native alternate for the current workspace. +- Native-fenced cache cleanup can be deferred through a bounded queue while keeping disk admission and borrowed alternates. It releases them only after acquiring the fence and removing the generation. Runtime shutdown, queue saturation and filesystem failures conservatively retain admission for startup recovery. + +## Work still required + +1. Route the actual HTTP/SSH native service through the implemented trusted catalog/ref proof, exact-response completion API and receipt-bound response reader. Preserve actual signature verification and invoke the completed-request preflight/replay API before repeating native preparation; integrate reviewed merge authorization. Deliver an immutable root for ref plans exceeding the 4 MiB inline command envelope. Integrate the implemented bounded collision/dependency reconciliation and account-fair command dispatcher with frontier/maintenance orchestration; qualify changed-run skipping and root-upload pipelining/grouping so catalog preparation cannot become another exclusive throughput bottleneck. Integrate serving/backup leases and the admitted CatalogFiles cache into production generation orchestration. Integrate the implemented bounded output partitioner with larger admitted input spools and long-lived input ownership; integrate the implemented geometric planner into resource-admitted continuous flush/range-compaction service scheduling. Validate canonical inventory preservation and effective preferred-source retention across levels. Install the fresh schema candidate with all converted product handlers and the deployment marker as the release schema. Reuse canonical metadata and graph meanings rather than introduce a compatibility adapter. +2. Complete the all-object native-pack cutover across network pushes, mirror imports, generated files, merge/squash/rebase candidates, browser/graph readers and backups. Remove durable inline/chunked/external Git-body alternatives and legacy pack-as-blob artifact wrapping. +3. Replace ingest's transient whole-index vectors and verified-OID sets with bounded, disk-backed processing; bound structural parsing and native process resources. Integrate authenticated artifacts with publication and cache installation. +4. Integrate the implemented owner-fenced preparation, trusted ref proofs and catalog/ref publisher with the production completion command and concurrent preparation. Remove serialization only after overlapping-object identity and coverage races are correct. Preserve ref CAS, ACL/policy checks and exact durable outcomes. +5. Qualify the complete Cellule publication durability gate at the target mixed write rate. Implement qualified publication grouping or select a qualified existing follower-log mode if needed; do not acknowledge local-only state. +6. Implement independent read workers, native MIDX/commit-graph acceleration, persistent batch readers, authorized generated-pack caching and bounded partial/full cache coverage. +7. Integrate the implemented bounded ingress and adjacent-level range compaction and geometric selection with continuous maintenance preparation and production service dispatch. Shared fair final-command dispatch now exists; CPU/I/O resource shares, outcome reconstruction and production orchestration remain required. Verified prefix windows now handle overlap sets exceeding a job budget; qualify their repeated parent scans and ensure physical files fit the selected resource profile. Implement physical pack rewriting. Implement complete retained-root inventory, renewable generation reader pins, online repository-scoped reclamation and isolated backup restore. The creating-attempt namespace and independent certificate-pin bindings now exist; integrate those facts with the complete retained-root inventory and deletion state machine before scheduling remote deletes. Cache repacking alone does not reclaim durable bytes. +8. Integrate the admitted ancestry fallback with large-history product operations and native commit-graph acceleration, and implement protected-branch merge queue bindings without weakening correctness under stale candidates or policy changes. Qualify positive and negative native histories exceeding 100k commits; the current 1,200-commit native and 100,001-entry scratch tests do not close that gate. +9. Install the new deployment-format marker and fresh schema together; reject incompatible old roots. Preserve the hard cutover; no migration or legacy reader. +10. Run full-history Kubernetes, Linux and Chromium campaigns, adversarial SHA-1/SHA-256 fixtures, fault matrices and the single-hot-repository workload: 4 pushes/s +70 fetches/s working-day soak and provisional 35 pushes/s +700 fetches/s peak. Verify every acknowledged write after owner loss and isolated restore, and report driver saturation and admission failures. + +The historical 10,000-repository campaign is not evidence for a 10,000-engineer monorepo. No large-team throughput, latency, provider durability or multi-year metadata-growth gate has passed yet. + +## Verification of this increment + +- Native close/drain passed all 352 unique library tests (6 Git-format, 14 object-storage, 332 server) in the full-workspace run, plus two CLI checks, ten Directory Cell checks and two Git HTTP checks. Five new resource tests cover terminal closure, admission races, multiple/canceled observers, actual two-worker final-release races and poisoned/underflowed proof refusal. The new real-server check holds both native classes for longer than one lease and verifies live renewal, delayed Cell shutdown, workspace exclusion, canceled observation and eventual withdrawal/reuse. Existing closed-stream and escaped-session native fixtures now await the pool drain directly. Server library tests finished in 172.49 seconds. The parallel multi-server run passed 94, failed two and ignored nine (403.96 seconds): large-object push returned HTTP 503 and bulk mirror returned HTTP 500, with Cell SQL deadlines and pending ingestion mutations in the log. This is a failed mixed run, not a clean workspace or capacity pass. Both unchanged cases passed together on a sequential rerun (204.54 seconds), preserving the original assertions and deadlines. Owner-restart, Repository Cell and stock-Git smart-HTTP integrations also passed separately (4.06, 30.17 and 72.68 seconds). Final all-target workspace Clippy with warnings denied passed (48.74 seconds), along with formatting, diff checks, new-file whitespace and 32 local documentation links. Workspace doctest targets passed with zero declared tests. Fresh CI is required for this shutdown increment. These checks do not qualify OS containment, full-history profiles, account fairness or the large-team mixed-load gate. + +- Native admission passed all 346 unique library tests (6 Git-format, 14 object-storage, 326 server) in the final full-workspace run, plus CLI, ten Directory Cell tests and two Git HTTP tests. Six new resource checks cover each limiting dimension, class isolation, concurrent atomic admission, invalid/overflowing profiles and poison quarantine. Existing native descendant/failure fixtures assert retained and returned claims; typed capacity tests distinguish HTTP 503 from unrelated WouldBlock I/O. A new two-backend SHA-256 integration verifies shared exhaustion and recovery. The initial resource run exposed completed repack guards holding claims into verification; both guards now drop after drain before validation, with the original cache tests passing in the final run. Multi-server passed 93, failed three and ignored nine: cold residency admission elapsed, large-object push returned HTTP 503 and bulk mirror returned HTTP 500. The log records Cell SQL command deadlines and pending ingestion mutations. All three unchanged cases passed together on a sequential rerun (134.98 seconds); the mixed run remains failed. Owner-restart, Repository Cell and stock-Git smart-HTTP integrations passed separately (3.66, 22.65 and 58.10 seconds). All 70 Python tests passed, and the two fleet checks passed again after recording native limits in readiness metadata. The final six resource checks also passed with actual CPU/memory profile-sum overflow assertions; a new CLI configuration test validates the example and rejects missing limits, unknown fields and negative capacities. Final all-target workspace Clippy with warnings denied, formatting, diff checks, four new-file whitespace checks and 40 local documentation links passed. Workspace doctest targets passed with zero declared tests. Both native-admission CI runs passed at `9942a8f`, including workspace integration tests and RustFS compatibility. This shutdown increment requires fresh CI. No hard native RSS, OS process containment, full-history estimate or large-team mixed-load gate is qualified by these checks. +- Shared native process ownership passed all 339 unique library tests (6 Git-format, 14 object-storage, 319 server) in the final full-workspace run, including three Unix fault checks for closed standard streams, escaped-session ownership and closed daemon descriptors. CLI, Directory Cell and Git HTTP checks also passed. Multi-server passed 93 tests, ignored nine provider checks and failed three cases: default-branch restore with address-in-use, large-object push with HTTP 503 and bulk mirror push with HTTP 500. The log records a pending Cell ingestion mutation; no native drain-fault test failed. All three unchanged cases passed together on a sequential focused rerun (206.63 seconds). The owner-restart, Repository Cell and stock-Git smart-HTTP suites also passed separately (2.30, 16.35 and 48.76 seconds). Final all-target workspace Clippy with warnings denied, formatting, diff checks, four new-file whitespace checks and 32 local documentation links passed. This remains a failed mixed run, not a clean full-workspace result, and does not establish mixed-load capacity. Both CI runs passed at preceding shared-spool commit `966d710`, including workspace tests and RustFS compatibility; native ownership changes need their own CI. +- Shared dependency spools passed `cargo +1.98.0 test --workspace --lib --locked` (336 unique tests: 6 Git-format, 14 object-storage, 316 server). The final five focused spool checks also passed with the added real write-failure and poisoned-reuse assertions. All-target workspace Clippy with warnings denied, formatting, diff checks, new-file whitespace and 32 local documentation links passed. Both CI runs passed at preceding ancestry commit `20f7130`, including workspace integration tests and RustFS compatibility; shared spool changes need their own CI. A native 512-distinct-commit fixture retains one file for SHA-1/SHA-256, checks exact tree/parent ranges with interleaved and reverse repeated replay, and holds full credit until the last witness drops. The wide-tree fixture uses shared ranges through admitted SQLite growth; corruption in a second range rolls back all headers/edges. Cancellation retains object-writer exclusivity through queued drain; denied growth, failed writes and actual file-length changes fail closed. These are ownership/integrity and per-page file-count results; global file/process admission, producer cutover and full-history mixed-load qualification remain open. +- Admitted ancestry growth and traversal fencing passed `cargo +1.98.0 test --workspace --lib --locked` (333 unique tests: 6 Git-format, 14 object-storage, 313 server), all-target workspace Clippy with warnings denied, formatting and diff checks. Three new native SHA-1/SHA-256 checks cover positive/negative memo reuse in one exact catalog, foreign catalog rejection, denied growth that cannot become a negative answer and cancellation with queued scratch ownership retained through drain. The 1,200-commit native fixtures run under a 4 MiB shared budget with a 16 MiB file ceiling; the separate 100,001-entry synthetic queue verifies indexed cleanup in batches of at most 512 and cached-answer preservation. These tests establish admitted scratch and reuse correctness, not native 100k-history latency, production integration or large-team capacity. +- Admitted SQLite growth passed all 330 library unit tests (6 Git-format, 14 object-storage, 310 server) in the full-workspace run and the final focused deep-chain/wide-fanout check with the new OID/distinct-child query-plan assertions. All-target workspace Clippy with warnings denied, formatting, diff checks and 25 local documentation links passed. Native SHA-1/SHA-256 1,600-blob/wide-tree verification passes under a 2 MiB shared budget with a 16 MiB metadata ceiling. Three growth checks cover written-prefix rollback, admitted cap ordering, preserved committed rows on denial, exact clipped ceilings and non-capacity errors. The full multi-server run passed 95 tests and ignored nine declared provider checks, but large-object restore received HTTP 503 after a Cell query deadline. Its unchanged focused rerun passed (71.86 seconds); owner-restart, Repository Cell and smart-HTTP suites also passed. The mixed-run failure remains recorded and full-history/mixed-load performance is unqualified. Both CI runs passed for growth commit `5a3b300`, including workspace integration tests and RustFS compatibility; ancestry changes need their own CI. +- Service-owned staging passed `cargo +1.98.0 test --workspace --lib --locked` (327 unique tests: 6 Git-format, 14 object-storage, 307 server), all-target workspace Clippy with warnings denied, formatting and diff checks. All nine new lifecycle tests passed, including automatic renewal, per-actor worker isolation, retained typed result handoff, exact Begin/Renew/Bind recovery, original receipts and native SHA-1/SHA-256 composition. The final full run includes the tightened account-worker assertions and boxed ready/terminal representations found by Clippy. Both GitHub CI runs passed at preceding staged-custody commit `057918b`, including integration tests and RustFS compatibility. These are bounded lifecycle/composition results; producer conversion, durable authenticated reconstruction, complete process admission and the full capacity campaigns remain open. +- Staged input retention passed `cargo +1.98.0 test --workspace --lib --locked` (318 unique tests: 6 Git-format, 14 object-storage, 298 server), all-target workspace Clippy with warnings denied, formatting and diff checks. All nine new staging tests passed, including actual restored-owner receipts, normalized deferred phase bindings, shared caps, revocation/expiry and native pre-bind physical witnesses. The 10,240-generation fixture injects trusted tiny facts to establish retention behavior, not import or publication throughput. Both GitHub CI runs passed at the preceding shared-dispatch commit `41df757`, including integration tests and RustFS compatibility. The input/producer/recovery/remaining-floor capacity gates in the staged input contract remain open. +- Shared foreground/maintenance dispatch passed `cargo +1.98.0 test --workspace --lib --locked` (309 tests: 6 Git-format, 14 object-storage, 289 server), all-target workspace Clippy with warnings denied, formatting and diff checks. Five new tests cover class bursts/account rotation with maintenance concurrency blocked, exact class/byte/actor reservations, native foreground progress while maintenance is paused, canceled maintenance observers and current admin revocation, and SHA-1/SHA-256 absent/lost-acknowledgement/panic recovery with exact original receipts and one immutable compaction outcome. All 14 dispatcher checks pass. The geometric native debt-drain fixture now uses the shared dispatcher and checks every replacement's canonical/source/version inventory, unchanged refs and old-reader access. A final focused mixed-admission rerun also rejects a cross-class logical-ID duplicate while preserving the original ready values. Both GitHub CI runs passed at shared-dispatch commit `41df757`, including workspace integration tests and RustFS compatibility. These results qualify bounded dispatch/composition, not continuous production maintenance, service reconstruction or mixed-load capacity. +- Geometric selection and the admitted-file cleanup-order fix passed `cargo +1.98.0 test --workspace --lib --locked` (304 tests: 6 Git-format, 14 object-storage, 284 server). Five new tests cover checked geometric targets, exact-threshold idleness, saturation and terminal-level refusal, urgent ingress with all 15 promotable levels pressured, bounded root/level rotation, structurally valid foreign repository/format rejection, SHA-1/SHA-256 native publication until ingress and level debt drain, complete canonical/source/version equivalence, unchanged refs, pinned old-reader access, admission failure retaining the same ingress choice and current admin rechecks when no work is selected. Final all-target Clippy with warnings denied, formatting and diff checks pass. The original queued cancellation fixture passed 20 consecutive focused reruns with its assertions and five-second bound unchanged. The final policy rerun also verifies a structurally valid foreign object-format snapshot. These tests establish advisory job selection and cleanup ordering, not a continuously dispatched service, publication throughput or full-history capacity. +- Coverage and bounded-window compaction passed `cargo +1.98.0 test --workspace --lib --locked` (299 tests: 6 Git-format, 14 object-storage, 279 server). Four new tests cover exact SHA-1/SHA-256 parent/prefix/suffix folds and poisoned fragments, projection-aware shared file caching and clipped lookup, independent fixed-byte codec vectors and old-domain rejection, and repeated native compaction windows with at most two input records. Each native window checks complete canonical/source/version preservation, unchanged refs, exact physical suffix reuse, stale competing work and pinned old-root reads. Reduced fanout required the retained-root fixture to select its actual last record. The admission fixture now checks valid disjoint-prefix progress before an unaffordable target and retains refusal assertions when no prefix can progress. All-target Clippy with warnings denied, formatting and diff checks passed. These are bounded correctness fixtures, not full-history or continuous mixed-load capacity qualification. +- Adjacent-level range maintenance passed `cargo +1.98.0 test --workspace --lib --locked` (295 tests: 6 Git-format, 14 object-storage, 275 server) with the default test concurrency. Six new tests cover bounded inclusive overlap seeks for SHA-1/SHA-256, enclosing/touching ranges and gaps, path-only cold reads, refusal to truncate an over-budget selection, native ingress/level promotion, verified artifact reuse without merge scratch, partial-root progress with unchanged files, disjoint level reconciliation/checkpoint rebinding, new/changed/missing target rejection, run/byte limits, invalid positions/formats, exhaustion and partitioning a native input into multiple target files followed by a native overlapping merge. Complete canonical entries, preferred source/version, refs/ref generation and pinned old-catalog access are checked. An existing 1,023-operation quota fixture twice exceeded the runtime deadline in full runs while passing unchanged in isolation; its setup now reuses prepared statements in batches of at most 512 operation/pin pairs. Quota, expiry, reaper and query-plan assertions and production deadlines remain unchanged. The final full run, all-target workspace Clippy with warnings denied, formatting and diff checks pass. These fixtures qualify bounded range replacement, not geometric scheduling, large-level subdivision, continuous maintenance, production integration or mixed-load capacity. +- Selected-ingress compaction passed `cargo +1.98.0 test --workspace --lib --locked` (289 tests: 6 Git-format, 14 object-storage, 269 server) and workspace all-target Clippy with warnings denied. Seven new tests cover individual input inventory poisoning, native SHA-1/SHA-256 publication filling all 32 ingress slots and reclaiming 31, exact canonical/source/version preservation, unchanged refs and ref generation, old-reader access, exact/logical replay and preflight, concurrent ingress and checkpoint rebinding, stale competing compaction, purpose/MAC/ACL rejection, atomic late-write rollback, immutable outcomes, actual owner restoration, admission limits, floor/base substitution and expiry. The native fixture uses small overlapping inventories; it establishes correctness of this primitive rather than full-history throughput or steady-state maintenance capacity. Production registry selection, fair maintenance dispatch, large-input lifecycle, geometric compaction and release qualification remain open. +- Indexed ingress roots and bounded output partitioning passed `cargo +1.98.0 test --workspace --lib --locked` (282 tests: 6 Git-format, 14 object-storage, 262 server). Five new checks cover native SHA-1/SHA-256 inventories spanning more than 32 output files, exact source/version preservation, one candidate per ingress root, byte accounting, failed admission, small-file reuse, poisoned count/digest/range exhaustion, queued cancellation and output workspace survival after the input/stream drops. The existing native assembler fixture now forces more than 32 output files through one root and verifies every object, repeated inputs, empty follow-up and certified-base dependency reuse. The full 48-candidate test also verifies rejection of the old directory-root v1 domain. An existing queued-SQL cancellation fixture now warms its range node before blocking the SQL worker, isolating the intended boundary introduced by the indexed ingress layout. Final all-target workspace Clippy on Rust 1.98 with warnings denied, formatting and diff checks pass. The preceding commit `3995698` passed both GitHub CI runs, including workspace tests and RustFS compatibility; those checks preceded this layout change. These tests establish this primitive's correctness, not full-history import growth, continuous compaction, production cutover or large-team capacity. +- Bounded publication dispatch passed `cargo test --workspace --lib --locked` (277 tests: 6 Git-format, 14 object-storage, 257 server), including nine new dispatcher checks. They cover FIFO account rotation, count/account/byte/concurrency limits, retained admission failures, foreign/duplicate submissions, queued and running resource accounting, concurrent command progress, oversized inline rejection, native SHA-1/SHA-256 publication after caller cancellation, absent/lost-acknowledgment/task-panic recovery, current authorization after queue admission, stale catalog rejection followed by reusable-input reconciliation, and close/drain with explicit recovery. Native recovery checks release scratch while resolved tickets remain alive and preserve the stored wire response without another generation increment. A final focused run passes all nine checks after strengthening original-receipt verification across unrelated intervening SQL changes and checking the public uncertain-work stats. Final all-target workspace clippy with warnings denied, formatting, diff checks and explicit new-file whitespace checks passed. These tests qualify the dispatcher primitive, not production scheduling, provider grouping, owner-loss reconstruction or mixed-load capacity. +- Reconciliation passed `cargo test --workspace --lib --locked` (268 tests: 6 Git-format, 14 object-storage, 248 server), including 55 publication tests. Five new publication tests cover simultaneous SHA-1/SHA-256 physical inputs, canonical overlap, preserved namespaces/input digests, immutable optional checkpoints, selected-base ref and exact-response publication, unchanged external anchors, missing dependencies, fabricated body/graph conflicts, revocation/expiry/Claim, the maximum 1 KiB certificate envelope, and actual owner restore with original replay plus stale pending-proof rejection. Two retained-scratch tests cover indexed 512-row pages over 10,000 synthetic incoming/anchor entries and confirmed queued cancellation retaining admission until drain. That query-plan fixture is not a 10k-object native capacity test. A final focused run of all five reconciliation tests also passes after adding correctly MACed original-floor substitution rejection. The production implementation is unchanged by that final assertion. Final all-target workspace clippy with warnings denied, formatting, diff and new-file whitespace checks passed. Fair scheduling, large-import lifecycle, production invocation and capacity qualification remain open. +- Frontier querying and floor retention passed `cargo test --workspace --lib --locked` (261 tests: 6 Git-format, 14 object-storage, 241 server), including all 50 publication tests. Five new tests cover both formats, unchanged attempt/namespace/expiry and receipt across repeated frontier reads, actor/request/context/expiry rejection, Claim/Abort retention of intervening roots, expiration before reaping, reserved-root protection, bounded indexed cleanup and invalid frontier codecs. The quota fixture now verifies that reaching capacity cannot bypass a retained floor. All-target workspace clippy with warnings denied, formatting and diff checks passed. This fresh-schema prerequisite is not wired into production or selected-base certificate issuance; no mixed-load or large-import capacity gate was exercised. +- Exact-response completion and replay passed `cargo test --workspace --lib --locked` (256 tests: 6 Git-format, 14 object-storage, 236 server), including all nine new completion tests and 45 total publication tests. The final run covers typed command/query transport, read-only preflight, exact receipt-bound responses, maximum escaped options, signed witness scope, atomic late-error rollback and original response replay after actual owner restoration. Production `repository_cell` and stock-Git `smart_http` integrations passed; selected `multi_server` push-option tests passed four checks, including real native signed-push verification/audit restoration and SSH options. One isolated RustFS signed-push test remained explicitly ignored. Final all-target workspace clippy with warnings denied, formatting and diff checks passed. These checks qualify this API increment and existing serving regressions; fresh-format HTTP/SSH wiring, source-independent recovery and mixed-load capacity remain unqualified. +- Trusted ref-plan certification and atomic typed catalog/ref publication passed `cargo test --workspace --lib --locked` (247 tests: 6 Git-format, 14 object-storage, 227 server), including 36 publication tests. New checks cover native SHA-1/SHA-256 membership/kind and ancestry, signed plan/evidence tampering, current policy/check/reporter changes, exact catalog CAS, late-error transaction rollback, checkpoint compatibility and original-outcome replay after actual owner restoration. The production `repository_cell` and stock-Git `smart_http` integration suites and workspace all-target clippy with warnings denied passed. These verify the typed primitive and existing serving regressions; they do not establish fresh-format network completion, changed-generation progress, native 100k-history ancestry or large-team capacity. +- Shared ref-boundary extraction passed `cargo test --workspace --lib --locked` (237 tests: 6 Git-format, 14 object-storage, 217 server), including the three new fresh-schema ref tests. The production `repository_cell` and stock-Git `smart_http` integration suites passed after the future-allocation and fixture-codec fixes above; the final repository rerun uses the original test harness and standard stack. Workspace all-target clippy with warnings denied also passed. The implementation plan now removes superseded per-object SQL cursor/location/ancestry/backup dependencies and aligns routine collection with the mandatory repository-scoped retention design; exact native MIDX command/version qualification remains required. This increment does not implement the final catalog publisher or establish the large-team capacity gate. +- Durable creating-attempt namespaces and independent certificate pins passed `cargo test --workspace --lib --locked` (234 tests: 6 Git-format, 14 object-storage, 214 server), including all 23 publication tests. Three new namespace tests exercise delayed deletes for all five artifact kinds after Claim/Abort/recreation of the same logical request, allocator exhaustion with live-attempt preservation/exact replay, and counter/pin immutability plus SQLite REPLACE guards. Existing tests now verify certificate retention across Claim and genuine owner restoration, fresh namespace allocation after reaping and bounded invalid token codecs. The initial full run exposed a five-second deadline in the synthetic 4,095-pin fixture setup; setup now uses prepared statements in 512-row transactions, and the final full run passes the unchanged quota boundary. All-target workspace clippy with warnings denied, formatting and diff checks also passed. No remote collector or capacity gate is qualified by these tests. +- Conditional certificate issuance and optional operation-row registration passed `cargo test --workspace --lib --locked` (231 tests: 6 Git-format, 14 object-storage, 211 server). Six attestation tests cover native SHA-1/SHA-256 facts, inline issuance without a write, bounded registration, tamper/scope/conflict rejection, claim invalidation, expiry/reaping, actor revocation and genuine owner succession. Recorded outcome replay returns the original receipt without restoring cleared state or granting fresh publication authority. All-target workspace clippy with warnings denied, formatting and diff checks also passed. These checks do not test the missing final publisher, production hard cutover, safe remote reclamation or mixed-load capacity. +- The catalog assembler and artifact-store witness binding passed `cargo test --workspace --lib --locked` (225 tests: 6 Git-format, 14 object-storage, 205 server), including all 14 publication/preparation tests. Six new assembler tests exercise complete native three-shard inventories in SHA-1/SHA-256, duplicate source/input deduplication, preserved base subtrees, a commit-only physical input whose typed dependencies come from the queried base, and rejection of the same input against an empty base. Negative tests cover wrong backend/operation, incomplete/out-of-order partitions, disk admission and confirmed queued cancellation with retained private-directory lifetime. Workspace all-target check, all-target clippy with warnings denied, formatting and diff checks passed. Trusted fixture injection supplies the test's published base; no production certificate issuer, final publication transaction, HTTP cutover or capacity gate is implied by these results. +- The fresh schema, owner-fenced preparation commands and query-derived base adapter passed `cargo test --workspace --lib --locked` (219 tests: 6 Git-format, 14 object-storage, 199 server). Eight preparation tests use the public Cellule command/query path, genuine drain/reacquisition and authenticated native SHA-1/SHA-256 catalog fixtures. The schema tests reject incomplete/mutable facts and mismatched deferred operation/pin bindings, enforce the 8,192-fact cap, and verify admission resumes when a fact is removed. Rebase tests preserve a live old pin, then reap only its expired pin/obsolete fact while preserving the current generation. Foreign repository/format facts fail without creating operations or pins. Exact renewal replay followed by a fresh query cannot revive an expired base session. All-target clippy with warnings denied, workspace all-target check, formatting and diff checks passed; the final test rerun also passed after the reply representation was boxed to avoid an oversized enum. The current production module still uses its previous schema and publication path. No mixed-load, full-history, clock/drain or remote collection gate was exercised. +- Cellule revision `cea9b9a7913f88cca114a2010e6b5c0a0aacbcf0`: workspace all-feature tests passed (1,217 tests; 35 documented environment/qualification tests remained ignored), plus 46 local LTX tests/doctests without default features. All-target/all-feature check, clippy with warnings denied, Rustdoc with warnings denied, formatting, boundary/layout checks, 81 Rust documentation fences, 1,080 documentation links, and 28 SQL/peer assertions plus 559 runtime-guide links passed. Tests cover successor rejection and exact original-outcome replay, wrong incarnation, authenticated effect delivery/replay, schema/code migration and failed-transfer admission replacement. This is runtime correctness evidence; no large-team durability or throughput gate was exercised. +- The admitted catalog loader and batched header increment passed `cargo test --workspace --lib --locked` (211 tests), all-target clippy with warnings denied and formatting/diff checks. Seven added reader tests and the extended full-48-candidate/canonical-conflict tests exercise both formats, saturation, source-header conflict and confirmed queued cancellation. These are local API/primitive checks; the existing HTTP path still does not use the new catalog loader. +- After updating all Cellule pins to the published fence revision, `cargo test --workspace --lib --locked` (204 tests), `cargo test -p canopy-server --test smart_http --locked`, clippy with warnings denied, formatting and diff checks passed again. This earlier run qualified the dependency pin; preparation consumers were added afterward, while final publication remains open. +- `cargo test --workspace --lib`: 6 Git-format, 14 object-storage and 191 server tests passed (211 total), including 71 metadata/directory/index/snapshot/source/catalog/verifier/closure tests. Earlier runs found an LFS integrity-error regression and cache-lifetime/accounting failures; these were corrected before the recorded passing run. The first directory run tests also caught an incorrect minimum file-size assumption; valid three-page SQLite runs are now accepted. +- `cargo clippy --workspace --all-targets -- -D warnings`: passed. +- `cargo fmt --all -- --check`, `cargo check --workspace --all-targets` and `git diff --check`: passed. +- The native contract fixture passed for SHA-1 and SHA-256 using Git `2.50.1 (Apple Git-155)`. +- The earlier per-object SQL design fixture still composes with the workspace product schema and passes seven invariant checks. It is not the selected large-team release DDL. +- `cargo test -p canopy-server --test smart_http`: passed with stock-Git push/clone, gateway workspace/budget isolation, deferred cleanup, publication/policy races, same-ID replay and injected storage/preparation failures. This verifies the existing serving/publication path after the native-reader changes; the new catalog and physical/canonical/conditional closure stages are not connected to that path yet. + +These are local checks of this working-tree increment. They do not establish large-team capacity or close the remaining implementation scope. diff --git a/docs/large-repository-research-and-scope.md b/docs/large-repository-research-and-scope.md new file mode 100644 index 0000000..4cdb7f8 --- /dev/null +++ b/docs/large-repository-research-and-scope.md @@ -0,0 +1,281 @@ +# Large repository research and implementation scope + +Canopy should retain compressed Git packs in immutable object storage, keep repository authority in the Repository Cell, and serve Git from verified local SSD caches. The selected implementation now packs **all Git object kinds** and uses a hard cutover to fresh storage. It reuses canonical object, graph, ref, push and collaboration structures while removing legacy Git-body representations. + +**Status:** Research inspected on 2026-09-30 against Canopy revision `9c5f1d1bf837fdc2e38ee229aa409712b9579f69` and Cellule `a3fbfb0115a1ae2519ee8f8e0cf6b8e72fdaa303`. Code observations below describe the initial inspection; concurrent working-tree cache and timeout changes are distinguished in the design. No new large-repository benchmark is claimed. + +The [technical design](large-repository-storage-design.md) and [executable implementation plan](large-repository-implementation-plan.md) supersede this report's original blobs-first rollout and migration proposal. They specify the all-object verification boundary, schema, commands, cache synchronization, compaction, drained collection and fresh-format cutover. This report supplies host comparisons and architectural rationale. Historical engineering reports establish techniques, not a complete account of a vendor's current production architecture. + +## What large repository support means + +Repository size has several independent dimensions. A compressed pack size alone cannot predict the cost of hosting it. + +| Dimension | Representative pressure | Canopy requirement | +| --- | --- | --- | +| Historical body volume | Many versions of similar source files | Preserve delta compression across versions | +| Object and edge count | Millions of blobs and trees; deep commit history | Bounded ingestion, graph certification, indexes and metadata recovery | +| Working tree width | Many tracked paths at one revision | Fast tree browsing; support client sparse checkout without claiming it solves server storage | +| Reference count | Branches, tags and review refs | Snapshot pagination and bounded advertisements | +| Read traffic | CI jobs cloning the same revision | Pack output reuse, coalescing, cache affinity and eventually multiple read hosts | +| Write traffic | Concurrent developers or agents | A short authoritative publication transaction; verification outside it | +| Fork population | Many repositories sharing history | Eventually share immutable artifacts with explicit retention across repositories | +| Recovery volume | A large Cell and many packs after disk loss | Metadata restoration plus bounded parallel pack downloads and hot-repository prewarming | + +Qualify Kubernetes full history and selected release refs first, Linux mainline and stable histories separately next, then `chromium/src`. A Chromium development checkout is an additional multi-repository workload: its official instructions use `depot_tools` to obtain code and dependencies, offer `--no-history`, and describe a shared snapshot cache. Their build disk requirement is not the compressed size of `chromium/src`. [Chromium checkout instructions](https://chromium.googlesource.com/chromium/src/+/HEAD/docs/linux/build_instructions.md). + +Use Chromium as the reproducible public proxy for the Chrome-sized case. No claim about private Chrome repositories or Google's private storage implementation follows from that corpus. + +## Findings at the initial inspection + +The following are implementation observations, rather than predictions of the proposed design's speed. + +| Initially observed behavior | Evidence | Consequence | +| --- | --- | --- | +| `GitObjects` walks revisions and reads expanded bodies through a persistent `cat-file --batch` process | [git_objects.rs](../src/git_objects.rs), [gateway ingestion](../src/git_gateway.rs) | Ingestion already avoids a Git process per object, but still expands the body of every new object | +| Blobs up to 768 KiB are inline; larger blobs are external; oversized structural objects use SQLite chunks | [ObjectStorage](../src/lib.rs), [schema](../src/schema.sql) | Similar historical source versions lose pack delta compression when stored separately | +| Object writes admit at most 128 records, a 4 MiB command input and smaller body budgets | [object_batch.rs](../src/object_batch.rs) | Millions of objects mean many durable commands even after body bytes are removed | +| Hydration pages through new object headers and writes missing individual bodies; presence checks use loose paths | [hydration.rs](../src/git_gateway/hydration.rs), [git_cache.rs](../src/git_cache.rs) | An insertion cursor improves warm refresh, but cold recovery still reconstructs loose objects | +| SQL graph certificates and object edges protect ref publication | [graph preparation](../src/graph/preparation.rs), [graph.rs](../src/graph.rs) | Moving structural bodies immediately would change the existing verification contract | +| Native HTTP workers receive a 120-second operation deadline | [git_http.rs](../src/git_http.rs) | A legitimate large indexing or pack-generation job can fail before completion | +| Ancestry traversal stops beyond 100,000 discovered commits or 250,000 edges | [ancestry.rs](../src/ancestry.rs) | Hosting storage and large-history pull-request operations need separate qualification | +| Backup inventories understand external Git blobs and LFS bodies | [backup body inventory](../src/deployment/backup/bodies.rs) | Adding packs without extending that inventory would produce incomplete backups | + +The existing [Kubernetes evaluation](kubernetes-qualification.md) reports a clean full-history corpus with 1,663,509 reachable objects, 141,666 commits and about 1.22 GiB of source packed storage. Its push failed with a native HTTP timeout before durable ingestion. A separate tree fixture pushed successfully but failed its clone; the precise clone failure cause remains unconfirmed. There is no passing large-repository recovery result. + +A different local all-ref fixture inspected during the original storage investigation had about 1.8 million objects, 1.3 GiB packed, and 32 GiB of expanded bodies, of which about 31 GiB were blobs. It includes local test changes. The roughly 25-fold pack-to-logical difference is evidence that representation matters, not a measured 25-fold reduction in Canopy's database. SQLite, graph indexes, LTX history, backups and retired packs must be measured independently. + +**Sizing inference:** importing 1,663,509 new objects with a 128-record maximum needs at least 12,997 object publication batches if all are represented individually. Byte limits, graph certification and other commands add work. At an illustrative sequential 10–30 ms per durable batch, object commands alone would consume roughly 130–390 seconds. Those latencies are assumptions, not measurements of Cellule or the provider. Measure actual publication cost before choosing batching changes. + +## What other hosts teach us + +### GitHub + +GitHub's published Spokes design replicates at the Git application level, acknowledges updates through a quorum, and routes reads to synchronized replicas. That demonstrates the value of keeping Git computation near repository data and explicitly tracking replica freshness. [Stretching Spokes](https://github.blog/engineering/infrastructure/stretching-spokes/). + +Its maintenance work combines multi-pack indexes, reverse indexes, multi-pack reachability bitmaps and geometric repacking to avoid continually rewriting the entire repository. These are upstream Git techniques Canopy can reuse. [Scaling monorepo maintenance](https://github.blog/open-source/git/scaling-monorepo-maintenance/). + +**Recommendation:** borrow pack organization and indexes. Keep Cellule as Canopy's authority rather than adding a second quorum protocol around local Git refs. Durable SQL publication and local ref updates must not become competing commit points. + +### GitLab + +Gitaly provides a dedicated Git execution service; Praefect adds repository replication. GitLab's storage guidance makes local SSD requirements explicit. [Gitaly architecture and disk requirements](https://docs.gitlab.com/administration/gitaly/). + +Gitaly's output cache deduplicates identical concurrent fetch computations across HTTP/SSH and clone/fetch variants. Unique requests gain little, and cached output can increase disk writes. This is a different cache from the repository's installed object packs. [Pack-objects cache](https://docs.gitlab.com/administration/gitaly/configure_gitaly/#pack-objects-cache). + +GitLab also supports bundles on object storage/CDNs for bootstrap. [Bundle URIs](https://docs.gitlab.com/administration/gitaly/bundle_uris/). Its fork pools use Git alternates and require care when pruning shared objects. [Hashed object pools](https://docs.gitlab.com/administration/repository_storage_paths/#hashed-object-pools). + +**Recommendation:** separate worker admission, installed-pack caching and generated-output caching. Defer fork pools until shared retention is proven. A filesystem alternates file alone cannot express Cellule recovery or access rights. + +### Cursor Continuity and Origin + +Cursor's August 2026 report describes packs recorded in an object-store WAL, publication through an atomic index update, and local NVMe Git repositories as caches. Read hosts check freshness before serving. Compaction is performed once and its packs are distributed to readers. These are vendor-reported mechanisms and results, without an independent benchmark here. [Git at any scale](https://cursor.com/blog/git-at-any-scale). + +**Recommendation:** this is the closest architectural precedent for Canopy. Use immutable artifacts plus one authoritative publication path, and avoid recomputing the same repack on every read host. Cellule already supplies ownership, fencing, request outcomes and root publication; Canopy should use those instead of introducing an independent Git WAL authority. This comparison does not imply that Canopy currently has Continuity's performance or read scaling. + +### Gerrit and JGit + +Gerrit exposes bounded pack-window caching and separate repository-cache expiration. JGit's DFS configuration distinguishes block caching from delta-base caching. These controls illustrate that large packs need not all reside in application heap. [Gerrit core cache settings](https://gerrit-review.googlesource.com/Documentation/config-gerrit.html#core), [JGit configuration](https://github.com/eclipse-jgit/jgit/blob/master/Documentation/config-options.md). + +**Recommendation:** distinguish disk cache, OS page cache, pack indexes and decoded-object memory. Retain native Git initially. These public interfaces do not establish the private production backend of Google's hosted Git service, and they do not justify writing a Rust remote delta engine now. + +### Microsoft and Azure Repos + +Microsoft's published scale work combines demand-driven object transfer, sparse working trees, nearby caches, commit graphs and incremental pack maintenance. The historical GVFS protocol is distinct from standard Git partial clone. [Scalar scale lessons](https://devblogs.microsoft.com/devops/introducing-scalar/), [GVFS architecture](https://learn.microsoft.com/en-us/previous-versions/azure/devops/all/git/gvfs-architecture?view=azure-devops-2020). + +**Recommendation:** preserve stock Git partial clone, sparse checkout and protocol v2. Sparse indexes and filesystem monitoring primarily improve client work; they cannot compensate for expanding durable history on Canopy's server. A new required client or virtual filesystem would broaden this project unnecessarily. + +### Bitbucket + +Atlassian's 2021 Cloud engineering account describes caching generated packfiles to reduce repeated filesystem work for clones and fetches. It establishes a useful workload-specific optimization, not the complete current Bitbucket storage architecture. [Bitbucket performance account](https://www.atlassian.com/blog/bitbucket/extinguishing-our-performance-fires-and-rebuilding-for-the-future). + +**Recommendation:** prioritize reusable output for CI bursts after installed packs and authority snapshots work correctly. Cache protocol pack output, not an entire response containing another client's negotiation, progress or ref advertisement. + +### Kernel hosting and Chromium checkout + +Linux's documentation bootstraps full history from a downloadable `clone.bundle`, then fetches current updates from the Git remote. This is a concrete example of moving bulk bootstrap away from repeated live pack generation. [Linux full-clone workflow](https://docs.kernel.org/admin-guide/quickly-build-trimmed-linux.html#downloading-the-sources-using-a-full-git-clone). + +**Recommendation:** offer generated bundles for popular public repositories later. Keep the baseline stock smart HTTP/SSH path working, including clients that do not use bundles. Private bundle access requires its own authenticated delivery and revocation policy. + +## Architecture options and the recommended choice + +| Option | Storage and performance | Architectural cost | Decision | +| --- | --- | --- | --- | +| Current individual SQLite bodies | Simple transactional verification; loses cross-version delta compression | Large body traffic through SQLite/LTX and costly loose-object reconstruction | Baseline only | +| Per-object compression in SQLite | Reduces individual bodies; misses history deltas | Smaller change, but still replicates body pages | Diagnostic comparator | +| Pack chunks in SQLite | Preserves deltas | Pack maintenance still rewrites SQLite/LTX; native Git needs local materialization | Benchmark alternative if unified recovery proves substantially cheaper | +| Bare repositories authoritative on durable local volumes | Direct native Git access | Adds repository replica authority and recovery coordination beside Cellule | Poor fit for current ownership model | +| External immutable packs with SQL identity and placement | Preserves deltas and enables native file reuse | Requires verified manifests, placement versions and complete backup/retention | Recommended first architecture | +| All object kinds in packs, with smaller SQL metadata | Further reduces structural-body duplication and materialization | Requires trusted streaming structural verification; graph schema remains substantial | Selected for the fresh-format hard cutover | +| Individual Git objects in remote KV | Natural content addressing | Remote graph and delta dependencies can create many serial network reads | Reject as the default execution path | + +The recommendation preserves one Repository Cell per UUID and one fenced writer. Scaling a busy repository's reads does not require sharding its authoritative refs or collaboration across Cells. Large uploads and verification can run outside the writer; their final publication remains serialized. Read computation can eventually scale across hosts with verified authority snapshots and reusable immutable packs. + +```mermaid +flowchart TB + client[Stock Git clients and browser] --> gateway[Authentication and resource admission] + gateway --> directory[Directory Cell
names and accounts] + gateway --> cell[Repository Cell
refs, ACL, replay, canonical identities
graph proofs, pack catalog, placement] + gateway --> verify[Private Git quarantine
bounded verification] + verify --> artifacts[(Immutable packs, indexes and manifests)] + artifacts --> local[Verified local SSD pack cache] + cell --> snapshot[Authorized ref and placement snapshot] + snapshot --> worker[Native Git reader or writer preparation] + local --> worker + worker --> client + cell --> maintenance[Owner-fenced bounded maintenance] + maintenance --> artifacts + cell --> backup[Backup inventory and retained roots] + artifacts --> backup +``` + +Local refs are generated views of Cell refs. Incoming pack presence, object-store listings and Git's ability to find an OID establish neither publication nor permission. Packs can physically contain unpublished entries or hidden candidate objects; Canopy must continue to constrain exposure to its authorized reachability rules. + +## Canopy implementation responsibilities + +### Separate canonical identity from byte placement + +Keep canonical identity immutable: repository hash format, OID, kind, logical length and independent body digest. Add a versioned packed location with artifact identity, preferred location version and a repository placement epoch. Maintain insertion sequence during compaction; the new format starts from fresh data. Ref generation changes only when refs change. + +The existing batch treats representation as part of duplicate-object consistency. Introduce a new command/codec that distinguishes a conflicting canonical object from another valid copy of the same object. Replace old codecs in the new format; no compatibility decoder is retained. + +Use the proposed pack catalog and operation journal in the [storage design](large-repository-storage-design.md#data-model). Inventory and location reads must be paginated and receipt-bound; no command or RPC should return millions of descriptors in one result. Compact binary metadata can avoid carrying blob bytes or large JSON maps through publication. + +### Publish artifacts before refs + +The proposed durable sequence is: + +1. Admit and spool a request; authenticate; run native Git in an isolated quarantine. +2. Identify accepted ref changes and the quarantine's new objects. Capture packs or pack loose objects generated by small receives, merge and rebase. +3. Normalize thin packs against verified bases. A persisted pack must contain all of its delta bases, although its commits and trees may refer to objects in other published packs. Git supports thin-pack repair through `index-pack --stdin --fix-thin`. [Git index-pack](https://git-scm.com/docs/git-index-pack). +4. Verify artifacts, canonical objects, object format and structural closure under bounded CPU, disk and memory admission. Stream independent blob digests; do not materialize the whole pack's expanded contents in RAM. +5. Upload immutable parts and a complete manifest binding digests, lengths, format and index-to-pack identity. Bind trusted verification to the repository, operation and owner generation. +6. Stage identity/location records in bounded Cell commands and prepare graph certificates. Mixed packs are allowed; every object uses a packed location, and the trusted streaming verifier supplies structural edges for Cell closure checks. +7. Recheck ACL, branch policy and expected ref versions in the final Cell command. Publish the accepted refs and exact retry outcome through Cellule's durable root path. +8. Return success after authoritative publication. Warm-cache installation can follow and cannot turn a committed success into a rejection. + +Preserve ordinary versus atomic push behavior. A failed or interrupted push may leave staged artifacts, but must not expose uncommitted refs. Retry reconciliation uses retained operation outcomes, rather than guessing from native Git's local success. SHA-1 and SHA-256 repositories both require the same lifecycle proof. + +Stream every object kind through the same verifier. Generated objects produce small native packs too; there is no durable individual-body fallback. Git LFS retains its separate protocol identity and reuses the authenticated artifact transport. + +### Reuse packs through one storage resolver + +Introduce a Canopy-owned resolver for authorized object streams and native cache preparation. Browsing, patch preview, merge/rebase, HTTP and SSH must use it consistently. Cell queries return bounded descriptors; external reads happen outside the SQL command transaction. + +Install complete verified pack/index pairs by atomic local publication and retain request pins through the lifetime of native descendants. Replace loose-path presence tests with a validated index inventory or persistent native batch checks. Coalesce concurrent downloads of the same artifact. Keep generated ref snapshots private while sharing immutable object files safely. + +A warm fetch should refresh new inventory and reuse existing packs. It must not rehydrate every old blob, inspect every object row on every request, or recompress historical bodies. Keep local automatic GC disabled; managed maintenance constructs a new inventory rather than modifying files under live readers. + +Cold individual file reads initially download the selected pack. That is acceptable as a documented prototype limitation, but can be very expensive for a small preview in a large pack. Measure it. Mitigations in order are prewarming hot repositories, a bounded decoded-object cache, smaller incremental packs, and pack-layout changes during maintenance. A custom remote range/delta reader is a later project requiring independently authenticated chunks and limits on delta-chain request amplification. + +For exact `blob:none`, prepare structural pack coverage and retrieve explicitly requested blobs on demand. Incoming mixed packs can still amplify server-side cold transfer; partitioning structural objects and blobs during compaction mitigates that. Other supported filters need separate tests. Partial clone reduces client transfer; it does not automatically reduce server verification or cold pack-download work. [Git partial clone design](https://git-scm.com/docs/partial-clone). + +### Maintain packs and indexes in the background + +Use append-only packs on the foreground path and bounded geometric compaction afterward. Keep the largest old packs out of routine small-push maintenance. Pack-count and duplicate-byte triggers should come from measurements; the selected design starts with a 1 GiB output target and explicit physical-artifact/admission limits. [Git repack](https://git-scm.com/docs/git-repack). + +Build native multi-pack indexes, reverse indexes, reachability bitmaps and split commit graphs against a complete declared local inventory. These are rebuildable accelerators and never ACL or closure authority. Their benefit and construction cost need benchmarking. A split commit graph accelerates native Git; it does not remove Canopy's SQL ancestry limits. [Git multi-pack index](https://git-scm.com/docs/git-multi-pack-index), [Git commit graph](https://git-scm.com/docs/git-commit-graph). + +The qualified Git baseline is 2.50.1. Current online manuals include newer features, such as MIDX compaction formats requiring Git 2.54. Pin and test the actual command/version combinations; do not copy the latest manual's options into that baseline. [MIDX format compatibility](https://git-scm.com/docs/git-multi-pack-index). + +Compaction selects a fixed inventory, creates and verifies replacement packs, then switches preferred locations with version checks in bounded batches. Concurrent pushes remain outside that inventory. One owner produces durable replacements; future read hosts install those outputs rather than repeating compression. Ref generation and canonical identities remain stable through representation changes. + +Local process pins alone cannot protect remotely read artifacts or backups. Conservative retention is sufficient for a prototype. A sustained storage-efficiency release also needs a deletion proof covering active locations, ongoing operations, candidates, readers, retained recovery roots, backups and retained recovery roots. The selected first-release deletion mechanism is a proven deployment maintenance drain, not an online grace-period collector. Git cruft packs can inspire local unreachable-object handling but do not supply that cross-system proof. [GitHub garbage collection work](https://github.blog/engineering/architecture-optimization/scaling-gits-garbage-collection/). + +### Treat CI output reuse as a separate optimization + +After correct installed-pack reuse, cache generated pack output for repeated equivalent fetches. A proposed cache identity includes repository UUID, authorized advertised-ref snapshot, normalized wants/haves, shallow boundaries, filter, relevant pack options and Git version. Authorization is checked on every request; use no cross-repository sharing initially. Protocol-specific framing remains per request. + +Coalesce equivalent active computations as well as completed cache hits. Limit output bytes, lifetime and producer concurrency. A cache that writes every unique multi-gigabyte response can increase I/O and disk pressure; measure hit ratio and generated bytes before making it default. Public bundle/CDN bootstrap is a subsequent option, with explicit opt-in or negotiated capability and a normal fetch fallback. + +### Scope fork sharing after repository isolation works + +For heavily forked projects, repeated baseline history can eventually outweigh per-repository pack improvements. A later design can share immutable baseline packs within an authorized fork family while keeping each repository's refs, ACLs and additional packs in its own Cell. Treat the shared inventory as a separately retained durable resource, with explicit dependencies from member Cells and backups. Git alternates can be a local cache mechanism, but cannot be the durable sharing contract. + +This extension needs recovery after upstream deletion, fork detachment, visibility changes, independent backups and concurrent collection. Do not use one mutable reference counter updated independently across several Cells as the sole deletion proof. Compare retained bytes across an entire fork family before adopting sharing; it adds coordination that is unnecessary for the first standalone Kubernetes or Linux storage proof. + +## Cellule implementation responsibilities + +Inspection used the pinned source in Cargo's checkout, rather than assuming the earlier [composed primitives proposal](repository-cell-primitives.md) reflects the current dependency. That proposal cites `a28de7b`; Canopy currently pins `a3fbfb0`. The relevant exclusive-role restriction still exists in the inspected revision. + +| Area in pinned Cellule | Observed contract | Scope for this work | +| --- | --- | --- | +| `cellule-ltx/docs/publication.md` and runtime `publication/mod.rs` | Immutable root dependencies are prepared; runtime control CAS establishes authority; pending outcomes are confirmed afterward | Reuse this commit point; do not add a Canopy WAL head as independent authority | +| Runtime `cell/catalog/mod.rs` | A catalog entry declares one `CatalogRole` | Repository SQL plus same-Cell Blob/Queue/Workflow is not a configuration toggle | +| Runtime `primitives/blob/api/mod.rs` | Blob client requires Blob role and routes by key shard | Existing namespace clients do not automatically bind artifacts to the repository Cell | +| Runtime `primitives/blob/mod.rs` | Blob upload parts are at most 256 KiB; reads are bounded | Stream and batch pack-sized artifacts through an appropriate API; avoid enormous command payloads | +| Runtime `primitives/maintenance.rs` | Primitive work inspection is conditional on exclusive role | Composed capabilities need complete inspection during transfer, recovery and code retirement | + +**Selected sequencing:** the implementation keeps SQL Repository Cells and store immutable artifacts through Canopy's existing external-body pattern. It does not need to wait for full same-Cell primitive composition. Canopy's SQL inventory and backup extension must then explicitly account for those external artifacts, as they already do for Git/LFS bodies. + +Cellule's required changes remain Git-agnostic: expose the runtime-admitted owner fence to command handlers, and provide complete paginated retained-root inventory for the drained collector. Reuse existing control, pin, receipt and root types. The detailed API and acceptance tests are in implementation packages B and J. + +Measure publication/restore cost and admission fairness using larger bounded metadata commands first. General same-Cell capability composition, queues and workflows remain independent work; they are not prerequisites for this storage release. Canopy's supervised worker and durable pack-operation records handle restart/takeover discovery. + +**Part-count inference:** a 1 GiB pack divided into 256 KiB parts needs 4,096 parts; at Canopy's current external-body 8 MiB granularity it needs 128. This is 32 times as many parts, not necessarily 32 times the total request cost. Benchmark upload batching, metadata, concurrency and provider requests before adopting the current generic Blob primitive unchanged. + +Cellule should own fencing, transactions, receipts, root publication, scheduling foundations and generic resource admission. Canopy should own pack verification, Git graph policy, native caches, ref exposure, maintenance selection and protocol behavior. Keep Git commands and OIDs out of the generic runtime API. + +## Recovery and consistency requirements + +A read host must obtain a valid Cellule serving capability and a selected ref snapshot, satisfy any minimum receipt, then install the required placement inventory. A pack cache hit cannot substitute for authority freshness. If the snapshot requires an unavailable or corrupt artifact, fail explicitly or recover it; never silently serve an older successful branch state. + +Restore metadata first, then obtain packs from its committed catalog. Metadata readiness and full Git readiness should be observable separately. Bound concurrent restore/download work and give hot repositories preference so a cold storm cannot consume all foreground slots. + +Extend pinned backup enumeration to include complete pack/index manifests and every required part, alongside canonical/graph metadata, LFS and collaboration. Verify restoration in an isolated prefix with access to the original artifacts revoked. A backup that restores refs but cannot clone their objects fails this gate. + +The selected release uses a fresh format/prefix, application identity and local data directory. It has no populated legacy migration or dual-read phase. Optional ordinary Git/LFS import preserves repository content, not old hosting metadata. Format checks reject incompatible service, backup and restore roots before writes. + +## Delivery scope and acceptance gates + +The [implementation plan](large-repository-implementation-plan.md) is the authoritative breakdown: format/metrics, Cellule fence, artifact transport, metadata/closure commands, streaming verifier, all producers, all readers, product/ancestry, compaction, backup/retained roots, drained collection, and corpus/provider qualification. Each package names files, interfaces, tests and review deliverables. + +A small fresh-format vertical slice proves storage plumbing, not Kubernetes/Linux support. Those claims require their full-history workloads, owner-loss and source-independent backup recovery, plus measured retention. Chromium additionally requires `chromium/src` qualification; packing every object kind does not eliminate graph-index, SQL/LTX or native-worker scaling limits. + +Measure canonical metadata and graph bytes before considering compact graph projections or repository sharding. One Repository Cell remains the selected authority. Sharding adds distributed coordination for refs, policy and reachability and is not part of the chosen solution. + +## Measurements and proposed performance gates + +Run the existing [large repository benchmark](../scripts/benchmark_large_repository.py) and extend its reports. Do not automatically download the entire Chromium development checkout to assess `src` hosting. Record source refs/OIDs, clean versus synthetic fixture status, object-format and count by kind, logical bodies, pack inventory, Git version, provider and host limits. + +Measure independent components: + +```text +live durable bytes = current packs/indexes/manifests + + SQL canonical/graph metadata and indexes + + necessary Cell recovery artifacts + legacy bodies + +total retained bytes = live durable bytes + staging + retired generations + + backup generations + orphaned artifacts + +peak local disk = SQLite/WAL + pinned pack caches + input spool/quarantine + + verification/download scratch + compaction outputs + + generated output cache + concurrent requests + +operation time = admission/queue + network spool + native Git + + verification + metadata/graph commands + + durable publication + cache/pack output +``` + +Report provider PUT/GET/range/CAS calls, bytes and retry rates alongside CPU, process-tree/cgroup memory, page cache, descriptors and peak filesystem use. Separate foreground and maintenance work. Log repository/operation IDs for diagnosis; avoid unbounded repository/OID metric labels. + +| Proposed gate | Measurement and interpretation | +| --- | --- | +| Correct large hosting | Initial import, v0/v2 full clone, filtered/shallow clone, incremental push, strict full `fsck`, exact refs and independent body checks | +| Packed body efficiency | After compaction, pack/index body artifacts at most 2 times a native packed baseline for identical object inventory and resource policy; report SQLite and retained bytes separately | +| No durable raw Git-body growth | Count representation bytes and detect producer paths falling back to inline bodies | +| Incremental work | A one-file commit neither reconstructs historical blobs nor rewrites the full repository; account for thin-pack bases and graph work explicitly | +| Warm service parity | Selected initial threshold: warm incremental fetch p95 within 25% of matched bare Git at equal concurrency; clone times and Canopy overhead reported separately | +| Small operation isolation | Proposed initial threshold: metadata/tree p95 within 2 times unloaded baseline during an admitted large import or compaction, with explicit queue/rejection rates | +| Recovery | Published objects recover after process death, owner takeover, fresh disk and isolated backup restore; measure metadata and clone readiness separately | +| Retention | Repeated incremental pushes and compactions reach a stated bounded retained-byte envelope after proven drained collection and the declared backup/root retention policy | +| Density alongside large repositories | Repeat small-repository density tests while one hot large repository receives bounded traffic; include cold activation and fairness | + +All numeric gates are proposed comparisons. Select absolute import, preview, incremental-push and recovery objectives after the baseline package on the intended bounded Linux/provider profile. The current large baseline fails, so use stage comparisons and native Git as references instead of calculating speedups from nonexistent successful Canopy runs. No expected kernel/Chromium capacity number is asserted here. + +Test failures at artifact upload, manifest publication, metadata staging, final ref publication, cache installation and compaction switches. Include client disconnects, disk exhaustion, corrupt parts/indexes, missing bases, provider outage, stale owners, ACL revocation, hidden candidate objects, backup overlap and old readers. An acknowledged push must survive every later cache failure. + +## Recommended next implementation + +Execute packages A–F of the [implementation plan](large-repository-implementation-plan.md) to establish the fresh-format authority boundary and first packed publication path. Then complete all readers/product semantics, compaction, source-independent recovery and drained collection before the large-corpus release gate. + +The [executable schema and native Git checks](large-repository-storage-design.md#validation-delivered-with-this-design) validate the chosen building blocks. Bulk metadata publication, graph processing, cold transfer and SQLite/LTX overhead remain measured engineering risks; compression alone is not a claim of Linux or Chromium performance. diff --git a/docs/large-repository-storage-design.md b/docs/large-repository-storage-design.md new file mode 100644 index 0000000..7a0feb4 --- /dev/null +++ b/docs/large-repository-storage-design.md @@ -0,0 +1,298 @@ +# Canopy packed repository technical design + +**Status: implementation specification, not implemented or capacity-qualified.** This is the selected design for a hard cutover to a fresh deployment format. It supersedes the earlier hybrid blob-pack proposal. The companion [implementation plan](large-repository-implementation-plan.md) defines deliverables and release gates; the [research report](large-repository-research-and-scope.md) supplies host comparisons and primary sources. + +**Large-team amendment:** [Large-team scalability requirements](large-team-scalability.md) takes precedence for the >10,000-engineer workload. It replaces per-object mutable Cell metadata and placement with certified immutable metadata segments/catalogs, requires concurrent preparation and independent read workers, and replaces globally drained normal deletion with fenced generation retention. The DDL below remains an earlier design fixture and is not the release schema for that workload. + +Store **every Git object in immutable native packs** and bulk canonical metadata/typed graph facts in certified immutable segments. The Repository Cell stores the selected catalog root, refs, authorization, collaboration and push outcomes. Serve native Git and browser reads through verified disposable local caches. Reuse Git's pack/index formats, Canopy's canonical fields, graph meanings and product tables, its hashed-part artifact transport, and Cellule's SQL publication and fencing. There is one Git object representation in the new format. + +There is no legacy reader, online migration, dual write, binary downgrade or transitional `ObjectStorage` enum. A deployment starts in a fresh prefix with a new application identity and empty local data directory. Existing deployments are not modified by this design document. Importing a Git mirror is supported as a new repository import; preserving old collaboration state is outside this cutover. + +## Decisions and boundaries + +| Concern | Selected implementation | +| --- | --- | +| Durable Git bytes | Self-contained `.pack` and v2 `.idx`, immutable object-store artifacts; both SHA-1 and SHA-256 repositories | +| Canonical metadata | Reuse canonical `objects` fields and typed edges in immutable SQLite shards; mutable Cell stores catalog roots/certificates | +| Physical placement | Leveled immutable OID directory selects a certified metadata/pack descriptor; native indexes retain offsets | +| Graph authority | Reuse typed edge, ordered parent, closure and ancestry meanings; trusted verifier certifies canonical facts against pinned catalog generations | +| Publication | Concurrent private preparation, short fenced catalog/ref CAS, existing policy checks, `pushes` and durable responses | +| Transport | Generalize the existing `CANOPY02` hashed-part manifest and upload helper for packs, indexes and LFS | +| Native execution | Stock Git in private request workspaces with pinned immutable object-cache generations | +| Maintenance | Geometric compaction and repository-generation retention; fenced online deletion after all recovery/backup/reader roots release it | +| Cellule work | Expose admitted owner fence to commands; expose complete retained recovery-root inventory for collection | +| Excluded | Cross-repository deduplication, custom delta decoder, remote pack range execution, required custom client and logical unreachable-object pruning | + +Mixed packs are valid. During compaction, partition selected objects into structural objects and blobs before packing; this uses the same pack format and tables, improves structural-history cache locality, and avoids a separate structural-body representation. Incoming mixed packs need not be rewritten on every push. + +Kubernetes and Linux exercise history and graph scale. Qualify `chromium/src` as the public Chrome-sized repository corpus; a complete Chromium checkout adds dependency repositories and is a separate workload. Existing local measurements do not establish current upstream sizes or Canopy capacity. + +## Architecture and trust + +```mermaid +flowchart LR + client[Git client or browser] --> gateway[Authenticated gateway] + gateway --> verify[Native quarantine and trusted verifier] + verify --> artifacts[(Immutable packs, indexes and metadata)] + verify --> publisher[Short fenced publication] + publisher --> cell[Repository Cell] + cell --> sql[Catalog root, refs and policies
Product state and push outcomes] + artifacts --> cache[Verified pinned SSD cache] + cell --> cache + cache --> native[Native Git or object reader] + native --> client +``` + +The Cell remains the authority for publication, but **structural byte verification moves out of the synchronous SQL command**. Native Git verifies pack decoding and object hashes; a trusted Canopy verifier independently streams canonical hashes, BLAKE3 body digests and typed edges. Cell commands accept the resulting metadata only through the enrolled internal node boundary. The existing signed `/internal/cell` transport and release identity are reused. These are server attestations, not cryptographic proofs that a SQL command can validate against remote bytes. + +Public APIs must not accept a caller-supplied “verified” inventory, artifact path or graph certificate. The pack worker receives repository-scoped authorization from Canopy, never an object-store prefix from client JSON. Generic internal SQL remains privileged; HTTP, SSH and product API handlers may not expose it. A compromised enrolled node is already inside this trust boundary. The new boundary must be recorded in the persisted contract and covered by forgery tests. + +Native pack presence never authorizes an object read. Preserve the existing ref-snapshot, reachability and hidden-ref checks, including explicit `want` validation and partial-clone requests. Extra physical entries from thin-pack completion or compaction remain inaccessible unless the request is authorized to retrieve them. + +### Code baseline + +This design was prepared against Canopy HEAD `9c5f1d1bf837fdc2e38ee229aa409712b9579f69`, pinned Cellule `a3fbfb0115a1ae2519ee8f8e0cf6b8e72fdaa303`, and the current working tree. The working tree already contains native cache pack retention, background cache repacking and an HTTP worker deadline of 3,600 seconds. An `object_pending` queue also appeared during the concurrent evaluation work; the supplied DDL defines the required queue whether or not that optimization remains in the checkout. Preserve and adapt that work. It does not yet make packs durable. The historical Kubernetes result used a 120-second deadline; do not confuse that result with current code. + +## Earlier per-object data model fixture + +The model and per-object command protocols below predate the large-team amendment. They preserve detailed canonical, graph, ref and artifact invariants for review, but their mutable per-object Cell tables, placement-switch commands and drained normal collection are superseded. Implement the amendment's catalog/segment architecture directly in the fresh deployment format. + +The executable [proposed DDL](design/packed-repository-schema.sql) replaces only Git storage, graph queue, refs and generation definitions. The [schema checker](design/check_packed_repository_schema.py) composes it with the unchanged product tables in `crates/canopy-server/src/schema.sql`. It is a design artifact, not a runtime migration. SQL constraints enforce local shape and selected invariants; typed commands enforce the cross-record invariants below. + +| Structure | Reuse or change | +| --- | --- | +| `objects` | Preserve stable `sequence`, unique `oid`, `kind`, logical `size`, BLAKE3 `digest`. Add preferred `pack_id`, monotonic `location_version`, `edge_count`, `edge_digest`. Remove `storage`, `body`, `external_sha256`, `chunk_id` | +| `object_uploads`, `object_chunks` | Remove. Push-response and certificate chunks are unrelated and remain | +| `packs` | New artifact descriptor, owning operation, native checksum, pack/index lengths and digests, staged inventory cursor, state, immutable `sealed_generation` | +| `pack_operations` | New resumable ingest/compaction identity, owner fence, attempt counter, state and timestamps | +| `pack_operation_inputs` | New fixed pack-ID set for a compaction; prevents selecting a moving input inventory | +| `object_edges` | Preserve `(parent, child)` identity; add expected child kind and retry-safe `waiting` flag | +| `object_pending` | Reuse the evaluated queue shape, adding edge-stream cursor/digest, remaining-child count and completion flag | +| `object_closure`, `commit_parents`, `commit_ancestry` | Reuse meanings and identifiers. Closure remains an immutable canonical-object fact | +| `ref_generation` | Reuse singleton; add `pack_generation` for sealed artifacts/location changes. Ref generation still changes only for ref/product semantics | +| Refs, pushes, ACLs, issues, checks, pulls, LFS | Reuse existing structures; no parallel product schema | + +A pack is self-contained for **delta bases**, not necessarily for commit/tree dependencies. An object may refer to a parent commit or subtree in another pack. Normalize thin packs before publication. No preferred location can depend on loose cache files or a disposable alternate for delta reconstruction. + +Canonical identity is `(object format, OID, kind, size, body BLAKE3, edge count, edge digest)`. Matching duplicate OIDs keep their existing preferred location during ingestion. Any mismatch rejects the entire metadata batch. Compaction changes only `pack_id` and `location_version`; sequence, canonical identity and closure survive unchanged. Native indexes contain offsets, CRCs and delta lookup information; do not duplicate them in SQL. + +### Artifact layout and integrity + +Use repository-scoped paths derived exclusively from validated IDs: + +```text +repos//git-packs///pack +repos//git-packs///index/ +repos//git-packs///metadata/ +.parts/<16-digit-hex-part-number> +``` + +The creating operation UUID is immutable even when its owner/attempt changes. Including this existing ID in the path prevents a delayed delete from an old collector from deleting a later identical upload: after retirement, a path can never become a new preferred location again. A new upload after deletion uses a new operation UUID and pack ID. Both `pack` and `index/` are manifests, not raw file bodies. Reuse the existing `CANOPY02` encoding: 8-byte magic, little-endian u64 total length, then one 32-byte BLAKE3 digest per 8 MiB part. Count is `max(1, ceil(size / 8 MiB))`; reject any extra/truncated manifest bytes. Bound count to 65,536: one physical artifact is at most 512 GiB and its manifest at most 2,097,168 bytes. This is an explicit artifact limit, not an advertised repository capacity. Larger inputs must be normalized into admitted packs or rejected with a resource-limit error before publication. Normal pack target is 1 GiB; Git can exceed its target for a single large object. + +Generalize `publish_lfs` into `publish_hashed` and share `ArtifactDescriptor { size, digest, manifest_digest }`. LFS keeps its SHA-256 protocol identity and hashed manifests. Remove `CANOPY01` Git-body code in the cutover; retaining the LFS encoding is reuse, not a legacy Git reader. + +The SQL descriptor pins complete-file BLAKE3, manifest BLAKE3, length and native pack checksum. The index descriptor independently pins index bytes and its associated pack checksum. Verify each part, the concatenated file and Git's native checksums. A native SHA-1 trailer alone is insufficient as the artifact integrity boundary. Conditional-create collisions require reading and verifying the existing content; the current helper's `AlreadyExists` success shortcut must be tightened. A complete manifest is published last, after all parts are confirmed. Incomplete uploads confer no publication rights. + +The pack header count is bounded by Git's u32 format. All size arithmetic uses checked conversions; SQLite integers are signed. No SQL part table is needed because the authenticated manifest already inventories parts. + +### Inventory encoding + +Use fixed canonical encodings for the verifier and command implementation. The following is normative; implement golden vectors for SHA-1 and SHA-256. + +- Kind codes: blob=1, tree=2, commit=3, tag=4. Integers below are unsigned little-endian u64. OID width comes from the repository format; do not pad a SHA-1 OID. +- Edge record: `child_oid || expected_kind:u8`. Deduplicate by child OID; a conflicting required kind rejects the object. Sort by raw OID bytes. Gitlinks are excluded, matching existing graph semantics. +- Edge seed: `BLAKE3("canopy.edges.v1\0" || oid_width:u8 || parent_oid)`. +- Fold each edge as `BLAKE3(previous_digest || ordinal:u64 || edge_record)`, with ordinal starting at zero. Final count plus digest bind the complete edge stream. +- Header record: `oid || kind:u8 || size:u64 || body_digest:32 || edge_count:u64 || edge_digest:32`. +- Pack seed: `BLAKE3("canopy.inventory.v1\0" || oid_width:u8 || pack_digest:32)`. +- Fold headers with the same ordinal rule, ordered strictly by raw OID bytes. The final digest and count must match the registered descriptor. Include every physical pack entry, including canonical duplicates. + +`body_digest` is BLAKE3 of the decoded body; OID is Git's repository hash of `kind + " " + decimal_size + NUL + body`. Hash chains bind completeness and order within a trusted attestation. They do not independently prove that the verifier interpreted bytes correctly. + +## Interfaces and command boundaries + +Replace `StoredObject` body transport with the shared metadata types below. Reuse `ObjectId`, `ObjectFormat`, `ObjectKind`, request IDs and the existing bounded codec. + +```rust +struct ArtifactDescriptor { size: u64, digest: [u8; 32], manifest_digest: [u8; 32] } +struct ObjectHeader { + oid: ObjectId, kind: ObjectKind, size: u64, digest: [u8; 32], + edge_count: u64, edge_digest: [u8; 32], +} +struct PackLocation { pack_id: i64, creating_operation: RequestId, location_version: u64, + pack: ArtifactDescriptor, + index: ArtifactDescriptor, git_checksum: ObjectId } +struct OperationToken { id: RequestId, fence: OwnerFence, attempt: u64 } +// Service-layer interface, outside SQL handlers; exact Rust ownership/lifetimes follow project conventions. +trait GitObjectReader { + async fn resolve(&self, oid: ObjectId) -> Result<(ObjectHeader, PackLocation)>; + async fn open_verified(&self, oid: ObjectId) -> Result; + async fn prepare_native(&self, snapshot: RefSnapshot) -> Result; +} +``` + +The source tree has no `OwnerFence` accessor on `CommandContext` today. Add a small Cellule API returning the runtime-admitted `(incarnation, epoch)` from its existing control authority. Thread that value through command execution and any replay context; never derive it from caller input or logical time. `BeginOperation` returns it. Every worker mutation compares its token with both the current admitted fence and stored operation token. Claiming an interrupted operation increments `attempt` and binds the current fence. Old attempts cannot continue staging or switch locations even when their commands are routed to the replacement owner. + +Keep existing command IDs where responsibilities remain. New-format codec registration supports only the selected codec; no adapters are retained. + +| ID | Command and codec | Contract | +| --- | --- | --- | +| 5 | `PutObjects`, codec 5 | Metadata headers only. Token, pack ID, expected ordinal/digest, up to 2,048 sorted headers. Verify identities, insert new objects/pending rows, advance pack cursor atomically | +| 6 | `CertifyObjects`, codec 4 | Tagged actions `StageEdges`, `CloseEdges`, `AdvanceClosure`; bounded as described below | +| 7 | Existing `CertifyAncestry`, codec 2 | Reuse validated parent-path certificates; change service-side traversal, not proof meaning | +| 20 | `PackOperation`, codec 1 | Tagged `Begin`, `Claim`, `Register`, `Seal`, `Ready`, `Complete`, `Retire`; each has the same token rules | +| 21 | `SwitchPackLocations`, codec 1 | Up to 512 `(oid, expected_pack, expected_version, output_pack)` CAS entries; canonical output inventory must have been verified | +| 22 | `CollectPack`, codec 1 | Maintenance-only mark/delete-receipt actions bound to a verified drained collection session | +| 3, 4, 8–10 | Existing ref/push/policy/candidate commands | Retain current responsibilities; update only dependencies on object body/storage checks, bump codec only if wire input changes | + +`Register` accepts two artifact descriptors, checksum, format, object count and inventory digest after upload/readback verification. It creates a staging pack with seed digest. Duplicate live-artifact registration succeeds only for an identical descriptor, and returns the existing operation identity; a caller must finish or claim that operation rather than silently replacing ownership. `Seal` requires all headers, matching inventory chain and count; it atomically increments `pack_generation`, records `sealed_generation` and transitions the pack to sealed. `Ready` requires every output pack of the operation sealed. After that, no output packs may be added. `Complete` requires no unresolved new graph work for ingest, or finished location switches for compaction. A compact `Begin` includes its entire at-most-32-pack input set in the same transaction; require each input sealed and reject another unfinished compaction for the repository. The set is immutable afterward, so there is no separate input-selection cursor. A deleted pack is an audit tombstone excluded from live-digest uniqueness; a later identical upload receives a new pack ID and seal generation, never revives an old ID. + +Use a 1 MiB input limit for new metadata commands and small receipt outputs. Every command is one Cell transaction, but internally uses multi-row SQL in blocks of at most 256 inserted rows and reads at most 512 rows per `context.sql` call. The pinned Cellule SQL API limits a batch to 128 statements, results to 1,000 rows and a statement to 32,766 parameters. It does not require one durable invocation per object. Do not request 2,048 result rows in one generic SQL call. Reject over-limit input before mutation. + +Each state-changing command uses the existing durable request-ID outcome mechanism. A transport retry reuses the identical request ID and bytes. New request IDs with stale expected cursor/digest return `CursorConflict` and the caller queries progress. Commands either commit the whole batch or reject it. After takeover, the new attempt re-verifies artifacts and resumes at committed cursors, issuing new request IDs. + +Errors are typed: `FenceMismatch`, `CursorConflict`, `IdentityConflict`, `InvalidArtifact`, `MissingDependency`, `WrongObjectKind`, `UncertifiedGraph`, `ResourceExhausted`, `UnsupportedFormat`. Uncertain transport outcomes are replayed, not translated into definitive push rejection. Map new internal errors into existing Git/HTTP error handling without exposing storage credentials or paths. + +## Ingestion and verification + +1. Authenticate, capture existing policy/ref state and admit CPU, disk and transfer work. Begin the ingest operation to capture its fence/attempt and immutable artifact namespace, then create a private native request repository using the existing isolated environment. Keep published cache generations pinned for base lookup. +2. Run receive-pack and preserve its accepted ref plan, hooks and atomic-push semantics. Inspect only files created in this request's private object directory. Do not scan a shared mutable pack directory to infer the incoming pack. +3. Retain valid incoming packs when self-contained. Complete thin packs with native `index-pack --stdin --fix-thin`; materialize loose outputs using `pack-objects`. Normalize over-size artifacts within admission limits. Validate the resulting pack/index pair in an isolated object directory with no alternates to prove delta independence. +4. Stream every physical object using persistent native batch processes. Independently hash its canonical OID and body BLAKE3. Stream-parse tree entries and commit/tag headers using existing parsing rules; do not allocate the full tree, blob or commit message. Spool headers and typed edges into a temporary on-disk SQLite database for sorting/deduplication. Index that scratch database by `(pack, oid)` and `(parent, child)`. Bound its cache and account for its disk bytes. +5. For a known OID, still verify physical bytes and compare canonical identity. A cache hit or duplicate pack OID does not bypass corruption/collision detection. Reject duplicate physical OIDs within one pack if inventory enumeration is ambiguous. +6. Upload and verify pack/index artifacts under that operation namespace. Register the descriptors and stage all headers for all its output packs before staging any edges. A graph dependency may resolve to another new object in this operation or an already certified existing object. A dependency on an unrelated unfinished operation returns a retryable dependency error; resume that operation instead of stealing its objects. +7. Seal all output packs, mark operation ready, stream edges, then run bounded closure propagation. Publish no requested ref until its tip is certified. Finish ingest metadata even if final ref CAS or policy rejects the push, so a failed push cannot permanently strand an OID in an incomplete representation. +8. Call existing completion logic. In its transaction, recheck current authorization, branch policy, expected OIDs/versions, closure and required candidate/check certificates. Commit refs, ref generation and durable wire response together. Artifact upload or local receive-pack success is not the commit point. +9. Return the committed response. Cache installation is best effort after commitment; a cache failure must not turn an acknowledged push into rejection. + +Operations interrupted after header publication are recovered by re-reading their authenticated artifacts and completing metadata/closure. The source pack contains the information required to rebuild a lost verifier spool. A terminal invalid pack discovered on recovery is an integrity incident and blocks affected publication; do not skip it. Before any headers were committed, an unused upload can be removed during drained orphan collection. Orphans do not become roots just because they appear in object-store listing. + +All producers use this pipeline: network pushes, mirror imports, web-created files, merge commits, squash and rebase candidates. For small generated objects, accumulate a bounded private batch and produce a small native pack; do not reintroduce a durable inline fallback. + +### Bounded graph closure + +Graph preparation must handle a single very wide tree and a blob referenced by millions of trees without a large transaction or in-memory reverse-edge list. + +1. Inserting a new header creates `object_pending` with zero counters and the edge seed. For a zero-edge object, `CloseEdges` still validates the seed/count before marking complete. +2. `StageEdges` accepts one parent's next at-most-2,048 sorted edges, expected received count and digest. Validate each child exists, has the required kind, and belongs to this operation or is already certified. Reject repeated/conflicting children. Insert each edge with `waiting=1` exactly when the child is not yet in `object_closure`. Increment `received_edges`, fold the chain, update `last_child`, and increase `remaining_children` only for newly inserted waiting edges. +3. `CloseEdges` compares received count and chain with the immutable header. Mark complete only on equality. For a commit-to-commit edge, populate the existing `commit_parents` projection. This projection is unordered; never use it as proof of parent order. +4. `AdvanceClosure` certifies at most 512 ready objects whose edge stream is complete, remaining-child count is zero and preferred pack is sealed. Insert closure idempotently. Keep the pending row while reverse propagation is unfinished. +5. For certified children still pending, process at most 2,048 waiting reverse edges per command using `object_edges_by_child`. Conditional `waiting=1 -> 0` updates and matching parent-counter decrements occur in the same transaction. Retry cannot decrement twice. Delete a child's pending row only after no waiting reverse edges remain. +6. An edge inserted after its child was certified starts with `waiting=0`; it never relies on an already-finished reverse-edge cursor. This closes the late-parent race. Missing or cyclic graphs cannot reach closure and cannot publish refs. Report unresolved graph work when no progress remains after all streams close. + +Command handlers verify pending `(sequence, oid)` matches the same object, edge count never exceeds the header, counters never underflow, and only the creating/claimed operation stages that object's edges. Pack and object metadata cannot be changed by graph processing. These semantic checks supplement the executable DDL. + +## Reads and caches + +`GitObjectReader` is the common service-layer path for Git, browsing, raw files, diffs, archive generation and merge preparation. Cell commands do not perform object-store or native-process I/O. `crates/canopy-server/src/git_read/mod.rs` receives the resolver instead of reading SQL bodies; keep API response-size and tree-pagination limits as explicit product limits, independent of streaming storage limits. + +Resolve an object only after authorization and canonical publication checks. Fetch its descriptor from the Cell, install both pack and index under disk admission, verify them, then use a persistent `cat-file` reader. Streaming callers avoid a body-sized allocation. Callers asking for a bounded preview retain that explicit limit. Missing/corrupt artifacts fail closed; a different local object of the same OID is not an unverified repair source. + +The current pack-aware cache uses a heap `HashSet` of every indexed OID. Replace that inventory with native `.idx`/MIDX lookup or a persistent batched native presence process. A million-object repository must not acquire an additional Rust heap copy of its full OID set on every node/request. OS page cache, mmap/index residency and native process memory still count in resource metrics. + +### Cache synchronization without a new authority + +A ref snapshot captures existing ref versions, a maximum object sequence `S`, and pack generation `G` in one Cell read. Every referenced tip must already have closure. Object insertion is allowed only into staging packs; all headers precede seal. + +For a cold full cache, keyset-page `objects` through sequence `S`, resolving preferred descriptors of sealed packs (including a location captured before retirement). Deduplicate pack IDs in bounded disk-backed scratch, not a whole-repository heap set. Ignore staging placements: they cannot be needed by the captured refs. Concurrent location moves are safe because each returned pack remains physically available until drained collection. Mixed old/new locations are acceptable; this is an availability inventory, not an authority snapshot. + +For warm refresh from generation `G`, install sealed packs with `sealed_generation > G` up to the new snapshot generation, in pages of 512. This covers newly publishable objects even when their headers were inserted long before certification. Do not use object sequence alone as the warm cursor: an older pending object can become certified later. Compaction location changes do not invalidate previously installed canonical bytes. A cold cache may already include a newer location; repeated installation is idempotent. + +A generation records its coverage and ref snapshot separately. Only mark coverage complete after all required transfers succeed. Native refs in each private request workspace reflect the captured authorized snapshot, not a concurrently mutable shared ref directory. MIDX and commit-graph files belong to that immutable generation. Installation builds a private directory and atomically exposes the complete pack/index inventory. Cache generation `Arc` pins live until all child processes exit. Eviction cannot unlink an in-use generation. + +For a filtered cold cache, enumerate all required structural objects through the same sequence boundary. Its coverage is explicitly structural-only; never mark it as full. Physical mixed packs may bring extra blob bytes into server cache even for `blob:none`. Before later unrestricted use, install missing blob-bearing packs or perform a full cold inventory. Native commit-graph, MIDX, bitmaps, sparse checkout and filter support are accelerators or protocol facilities; none replaces authorization. + +Initial cold reads download whole selected packs. This deliberately avoids a new remote delta engine. Measure cold one-file read amplification. Structural/blob partitioning during compaction and smaller incremental packs mitigate it, but a tiny requested blob in a large blob pack can still incur a large download. This limitation is part of the first release contract. + +### Product and ancestry changes + +`crates/canopy-server/src/pulls/candidates/mod.rs` currently compares an expected generated commit with its SQL body. Compute the same canonical expected bytes, OID and BLAKE3 and compare against certified metadata instead. This preserves exact tree, parent order, author and message semantics. Rebase validation must inspect ordered parent headers through the resolver or compare each expected canonical body digest; the unordered `commit_parents` table is insufficient. + +Remove the service traversal's fixed 100,000-commit/250,000-edge ceiling. Reuse native history enumeration with a disk-backed temporary predecessor/path table and existing `CertifyAncestry` proof batches. A concrete implementation streams `rev-list --parents ` from a fully prepared authorized cache, validates OID syntax, records actual parent links and a predecessor chain in scratch SQLite, finds the requested ancestor, then submits the path from ancestor toward descendant in 128-step batches. Each proof step is still checked against authoritative `commit_parents` in the Cell. False/malformed native output cannot manufacture a successful ancestry certificate. Negative results require complete traversal; budget exhaustion is an explicit resource error, never “not ancestor.” Temp state may be rebuilt after worker loss; committed certificates are reusable. + +Generate native commit graphs with changed-path Bloom filters on prepared generations for merge-base/history work. Keep request resource budgets. Removing arbitrary graph-count rejection does not grant unlimited execution time, and large-history PR operations have their own release tests. + +## Online compaction and offline collection + +Select at most 32 input packs or 8 GiB of compressed preferred data per compaction job initially. Freeze their IDs in `pack_operation_inputs` and pin verified input artifacts. New pushes append packs outside that set. Stream objects whose current preferred pack belongs to the selection into two OID spools: structural and blob. Include all such objects, even unreachable ones. Emit self-contained output packs with a 1 GiB target, native delta depth 50, two threads and 64 MiB window memory; these are starting admission settings, not proven throughput optima. + +Verify output canonical identity and inventory with the same verifier. Seal outputs before switching. `SwitchPackLocations` re-reads each OID's identity and performs CAS against its expected input pack/version; it verifies the output belongs to this operation's attested inventory. Since inventory rows are not duplicated durably, the trusted worker supplies membership attestations bound to the registered output digest and current operation token. The Cell validates canonical identity and operation provenance, not native index membership. A conflicting location is skipped and counted; it is never overwritten. Increase `pack_generation` for committed placement batches. Repeat until no preferred locations reference selected inputs. + +This avoids an unbounded final transaction. Crash recovery re-verifies outputs, scans remaining input locations and resumes. A pack retires only when no object prefers it and no other unfinished operation uses it as an input. Retirement is irreversible; handlers enforce forward-only pack states. If a later receive contains only already-certified canonical objects, verify them and use their current placements without registering another durable copy. Closed operations retain audit identity; input rows can be removed when complete. Do not mutate bytes under a digest path or edit a sealed inventory. + +Retired packs stay readable while the service is running. A local cache pin alone cannot protect a reader on another node. The first release therefore implements durable collection only through the existing deployment maintenance workflow: + +1. Begin maintenance using the existing authority CAS; drain all enrolled nodes and native processes. A heartbeat timeout alone is not proof. Use recovery/drain of actual Cells for unavailable owners, as existing maintenance requires. +2. Establish a stable retention inventory: current Cell roots, every supported retained recovery root, backup pins, unfinished copy/restore roots and pack operations. Prevent new user/backup roots, pins and source reads during the collection session. Only the collector may publish metadata roots; its allowed changes are retirement/deletion receipts and cannot add artifact references. Those descendants therefore cannot expand the fixed live-artifact set. +3. Enumerate pack/index manifests reachable from those snapshots, not just current `objects` rows. A retained historical SQL root can still name a retired pack. If the runtime cannot prove this inventory complete, refuse deletion. +4. Mark eligible retired packs `deleting` in a maintenance-bound command, persist a collector manifest under the existing maintenance operation ID, then idempotently delete unrooted artifact manifests/parts. Delete manifests before parts to fail closed on unexpected access; retries use the collector manifest. Mark `deleted` after provider confirmation. Recheck the maintenance fence before each delete batch. Fence checks alone cannot cancel an already-issued provider DELETE: irreversible retirement and never-reused operation paths ensure any delayed delete still targets an obsolete artifact, even after serving resumes. +5. Reconcile unregistered upload staging prefixes against current operations and the fixed root set. Preserve the entire upload namespace of every unfinished operation, including parts not yet registered; a completed operation cannot reopen that namespace. Delete only proven unreferenced entries; list results are candidate discovery, not proof. Verify remaining artifacts, finish maintenance, and resume service. + +The Cellule retention-inventory API must cover every recovery path the runtime can select, not merely the newest control pointer. Do not implement deletion by guessing LTX filename ages. Complete independent backups can release their source pins according to the existing backup contract; their copied artifacts then live in another root. Backup roots cannot be collected as though they were disposable service caches. + +Logical pruning of unreachable canonical objects is outside this release. Storage can grow with retained Git history, failed-push objects and backup policy. The design bounds representation/compaction amplification only after measuring those roots; it does not promise bounded lifetime bytes for an ever-growing canonical inventory. Monitor staged/retired bytes and require an operational collection schedule before claiming storage efficiency. + +## Recovery, backup and failure behavior + +Extend `crates/canopy-server/src/deployment/backup/bodies.rs` to inventory distinct packs/indexes referenced by **all `objects.pack_id` values and the inputs/outputs of unfinished pack operations** in the pinned SQL snapshot, plus existing LFS artifacts. Apply that same root rule to each retained historical snapshot. Retired/deleted catalog rows and completed operation records alone do not retain artifact bytes; otherwise collection would never reclaim a pack. An old root that still has an object location in an old pack does retain it. Stream part verification and copy with bounded concurrency; set backup completion only after every required artifact and metadata root is verified. Preserve original SQL pack IDs inside the copied snapshot; paths derive from repository IDs, creating-operation IDs and content digests, not deployment-local absolute paths. + +Restore into a fresh reserved destination. Rehydrate solely from destination roots/artifacts. The release drill removes access to the source prefix and local caches, then performs full clone/fsck, selected file reads, LFS checks, collaboration checks and durable push replay. A missing pack is a failed backup/restore, never a skipped optional cache. + +| Failure point | Required result | +| --- | --- | +| Interrupted multipart upload or lost create response | Retry/verify existing immutable destination; no pack seal until complete artifacts exist | +| Crash after upload, before SQL registration | Unreferenced artifact; recover by verified re-registration or drained orphan collection | +| Crash during headers/edges | Resume persisted ordinal/hash; rebuild verifier scratch from pack | +| Ownership change | Stale fence/attempt rejected; replacement claims and re-verifies before continuing | +| Crash after seal, before ref completion | Objects may remain unreferenced; existing refs/outcome unchanged | +| Lost response after ref commit | Existing durable request/outcome replay returns exact committed result | +| Cache loss after acknowledgment | Rebuild from SQL descriptors and durable artifacts, preserving ref state | +| Corrupt part/index or descriptor mismatch | Quarantine local file, stop affected read/publication; surface integrity error | +| Crash halfway through compaction switches | Both input and output remain valid; resume remaining CAS rows | +| Concurrent backup and compaction | Pinned old root retains old packs; collection cannot remove them | +| Crash during drained deletion | Remain in maintenance; replay collector manifest under recovered fence | + +## Resource policy and observability + +Proposed initial defaults, configurable per deployment: two native threads per foreground worker, one compaction job per node, one write preparation per repository, 64 MiB pack window, 32 MiB scratch SQLite cache, 64 KiB parsing/hash buffers, four concurrent artifact transfers per worker, 8 MiB parts. Two simultaneous one-object streams must not imply two body-sized allocations. Native indexing has additional object-count-dependent memory and must be measured and admitted; these settings are not a hard RSS guarantee. Enforce process/container memory and disk quotas in deployment qualification. + +Separate 120-second network/provider idle deadlines from total work budgets: one hour for interactive native work and four hours for explicitly admitted bulk import. Progressing CPU-only indexing is not a network timeout. Every native process must have cancellation, bounded stderr capture and reaping. In-process reservations account for input, normalized output, index, verifier spool, compaction output and pinned old cache generations; OS quotas handle estimate errors. Reject admission before exhausting disk. Backpressure the sender while spooling, rather than buffering a full request body. + +Record compressed input, logical decoded bytes, durable pack/index/manifest bytes, SQL snapshot bytes, retained LTX/recovery bytes, backup bytes, retired/staged bytes and cache/scratch peak separately. Measure durable command count, metadata rows per second, graph edges per second, provider requests, cache hit rate, transfer coalescing, verification CPU, time to first fetch byte, clone/fetch p50/p95, ownership takeover time and cold recovery time. Counters must distinguish physical pack objects from unique canonical objects. + +Use these accounting equations: + +```text +live durable bytes = preferred-pack union + its indexes/manifests + current SQL +retained durable bytes = live + retired/staged artifacts + recovery roots/LTX + backup copies +peak local bytes = installed pinned generations + input + normalization/output + index + verifier spool +``` + +Gate target: after compaction and drained collection, pack/index bytes are at most 2x a native Git baseline packing the same complete canonical inventory with the same kind partition and settings. Report SQL/retention overhead independently and also report total storage versus an unconstrained native baseline; do not hide it in the ratio. Warm incremental fetch p95 target is at most 1.25x the equivalent locally cached native Git baseline on the same machine and request corpus. These are proposed acceptance targets, not measured results. Publish absolute times, provider costs and memory peaks even if relative targets pass. + +## Hard cutover + +Keep the existing root marker key `canopy-root-v1.json`, but require a new envelope: + +```json +{"format":"canopy-pack-v1","purpose":{"kind":"service"}} +``` + +Reuse `RootPurpose` inside the envelope for service, backup and restore. New code rejects a missing/unknown format before loading Cells; old code's required top-level `kind` and unknown-field rejection reject the new envelope. Apply the check to service startup, maintenance, backup and restore, and version local cache directories by the same format. Do not add a legacy parsing branch. Test both directions explicitly. + +Use a fresh provider prefix, application identity and local data directory. Keep the bootstrap schema version at 1 within this new incompatible deployment format; update module source/release digests and new command codecs. Existing repository/table/type names are reused, not migrated. Do not connect new binaries to an old prefix through a schema edit. + +Cutover procedure: qualify the new build/provider; reserve the new format root; enroll compatible nodes; create/import repositories; verify clone, product state, backup/restore and owner-loss behavior; direct traffic to the new deployment. If old repository content is wanted, export/import through ordinary Git and LFS while the old deployment is separately available. Old refs, ACL/product history and replay identities are not automatically transferred. Retain the old deployment under its own identity until its owner decides its archival policy. + +Rollback before new writes can return traffic to the old deployment if it still exists. After new writes, rollback means restoring a verified new-format backup or rebuilding the new-format release; there is no downgrade that preserves new writes in the old format. + +## Validation delivered with this design + +Run from the repository root: + +```sh +python3 docs/design/check_packed_repository_schema.py +python3 docs/design/check_native_pack_contract.py +``` + +The first composes the proposed DDL with existing collaboration tables and checks seven storage/graph invariant cases. The second checks SHA-1/SHA-256 native packing, thin completion, isolated pack decoding, MIDX and commit-graph commands in disposable repositories. Both passed locally with Python 3 and Git 2.50.1 (Apple Git-155). They validate schema/command building blocks, not runtime integration, owner-fencing behavior, provider durability or large-repository performance. Those remain explicit deliverables in the implementation plan. diff --git a/docs/large-team-scalability.md b/docs/large-team-scalability.md new file mode 100644 index 0000000..94b1918 --- /dev/null +++ b/docs/large-team-scalability.md @@ -0,0 +1,281 @@ +# Large-team scalability requirements + +**Status: required design amendment; implementation and capacity qualification remain open.** Native packs solve durable byte storage, but the initial packed-repository specification is insufficient for a single repository used by more than 10,000 engineers. This amendment takes precedence over that specification and its proposed SQL schema wherever they conflict. A 10,000-repository density test is a different workload. + +The selected direction remains immutable native Git packs, verified local caches, stock Git clients, and one fenced Repository Cell for authoritative ref and product publication. Keep the hard cutover. Reuse Git indexes, commit graphs, existing OID types, canonical metadata fields, typed graph semantics, ref expectations, push identities, policy checks and durable outcomes. Reusing these structures does not require keeping every historical object and edge in the Cell's mutable SQLite database. + +## Capacity assessment + +**The amended architecture is a plausible target for this workload; the current implementation is not qualified for it.** Large-history storage and hot-repository write concurrency require separate evidence. Keep one Repository Cell as authority only if its complete mixed-load durable service fits the measured budget; bulk verification, artifact transfer and fetch generation must scale outside that serial lane. + +| Boundary | Implemented evidence | Remaining release condition | +| --- | --- | --- | +| Historical bytes and canonical metadata | Immutable native artifacts, bounded directory/source indexes and admitted metadata files | All producers/readers converted; multi-year lookup and compaction growth measured | +| Atomic publication correctness | Trusted ref-plan proof and catalog/ref/exact-response transaction; signed-input binding, rollback and owner-replay tests | Production HTTP/SSH invocation, reviewed merges and whole-operation qualification | +| Hot-repository progress | Exact catalog CAS, retained floors, bounded incoming/anchor reconciliation and bounded account-fair command dispatch | Frontier orchestration, production integration and measured progress under continuous publication | +| Durable write capacity | Runtime owner fencing and receipt replay tests | Complete provider durability gate fits mixed-load service budget, with qualified grouping if needed | +| Fetch capacity | File-backed indexes and admitted shared catalog-file loader | Independent read workers, native accelerators, authorized pack reuse and measured egress | +| Sustained storage efficiency | Bounded catalog/operation/pin primitives | Online compaction, complete retained-root inventory, fenced readers and safe reclamation | + +The typed publisher is a correctness milestone, not a throughput result. Its current exact-base requirement is deliberately fail-closed. At the target publication rate, routing all stale work through Claim and full preparation would multiply durable commands and can starve independent branch pushes. The reconciliation/coordinator deliverable is therefore a prerequisite for the capacity claim. + +## Workload and assumptions + +Use the single hot repository as the restrictive case, alongside a fleet of smaller repositories. Engineer count alone does not determine capacity. Distinguish locally created commits, pushed commits, receive-pack requests, main-branch landings and CI fetches. For initial sizing, conservatively assume one pushed commit per request; later substitute measured batching and payload distributions. + +| Input or derived quantity | Initial value | Interpretation | +| --- | --- | --- | +| Engineers | 10,000 | Minimum sizing population; expose as a variable | +| Commits per engineer per workday | 10 | User workload | +| Workday | 8 hours | Sizing assumption, not a fixed product constraint | +| Pushed commits per workday | 100,000 | Engineers × commits | +| Average pushed commits/s during workday | 3.472 | 100,000 / 28,800 | +| Provisional peak multiplier | 10 | Replace with arrival measurements; it is not evidence | +| Qualification peak | 35 pushes/s | Rounded up from 34.722 | +| Provisional CI fetches per pushed commit | 20 | Illustrative fanout, including retries and repeated jobs | +| Average CI fetches/s during workday | 69.444 | Excludes interactive reads and clones | +| Qualification fetch peak | 700 fetches/s | Rounded up from 694.444 | +| Workdays per year | 250 | Sizing assumption | +| Annual commits | 25 million | Does not include historical imports | + +At 100 genuinely new Git objects per commit, the illustrative annual addition is 2.5 billion objects. At 256 bytes of mutable metadata per object, that alone is 640 GB decimal, before edges, indexes, WAL/LTX history, replication and backups. These are sizing inputs, not measured Git object counts or SQLite row sizes. Measure the imported corpus and a representative sequence of real changes, including the new trees created along changed paths. + +For a 700 fetch/s peak and a 10 MiB average response, output bandwidth is 7,000 MiB/s, approximately 58.7 Gb/s of payload, before protocol overhead and redundancy. Generated-pack caching saves computation; it does not remove network bytes. Filtered and incremental fetches materially change this figure. Do not provision from a full-clone average or assume every CI request has an identical negotiation transcript. + +### Sizing populations above 10000 engineers + +The following projections retain the provisional burst and CI assumptions. They are arithmetic budgets, not measured capacity. Rates scale with population; a single repository's authoritative lane does not gain capacity by adding gateways. + +| Engineers | Commits per workday | Average commits/s | Provisional pushes/s peak | CI fetches/s peak | Mean durable service budget per command with two commands/push | +| --- | --- | --- | --- | --- | --- | +| 10,000 | 100,000 | 3.47 | 35 | 700 | 8.57 ms | +| 20,000 | 200,000 | 6.94 | 70 | 1,400 | 4.29 ms | +| 50,000 | 500,000 | 17.36 | 175 | 3,500 | 1.71 ms | + +These service budgets allocate 60% of the serial lane to pushes and exclude other product/maintenance work. Measure Begin and completion separately and sum their costs; the equal-cost per-command figure is only a shorthand. If the complete durable mixed-load budget cannot fit, qualified runtime publication grouping or a qualified follower-log mode is required before keeping the one-Cell architecture for that population. Additional read workers address read throughput independently. + +## What the current evidence establishes + +The [complete-corpus report](performance/2026-09-30-full-corpus.md) verifies 10,000 repository identities. Its populated sample consists of small two-commit fixtures. Its unique-branch push window offered 1 request/s across a 100-repository population and completed 29 of 30 scheduled arrivals; p99 was about 4.72 seconds. Those measurements neither establish a single repository's write ceiling nor satisfy the new workload. They are from a stated earlier revision and must not be presented as current capacity measurements. + +The current gateway holds a per-repository mutex through decode, native receive-pack, ingestion and completion in `crates/canopy-server/src/git_gateway/mod.rs`. Concurrent upload spooling avoids slow-client interference but does not parallelize preparation. Ingestion still creates durable per-object SQL metadata, structural SQL bodies and graph projections. Pack-backed blobs are partial implementation, not the specified all-object cutover. + +Current changes replace the serving cache's full OID HashSet with checked, file-backed native indexes. They also introduce operation-scoped authenticated artifact transport. Unit and cache tests validate those boundaries. They do not prove memory bounds for the whole native Git pipeline, concurrent publication throughput, online collection safety or large-team capacity. Index lookup still scans the installed index handles, and ingest still has transient object-count-dependent collections; these must be replaced or bounded before qualification. + +## Required architecture + +```mermaid +flowchart LR + clients[Engineers and CI] --> ingress[Authentication and bounded admission] + ingress --> preparation[Concurrent private native Git preparation] + preparation --> bulk[(Immutable packs and metadata segments)] + preparation --> publish[Repository publication coordinator] + publish --> cell[One fenced Cell owner\nrefs, policies, catalog root, outcomes] + cell --> readers[Independent read workers\npinned certified catalog generation] + bulk --> readers + readers --> output[Authorized generated-pack cache] + output --> clients +``` + +### 1. Concurrent preparation, short authoritative publication + +Spooling, decompression, thin-pack completion, indexing, canonical verification, graph extraction and artifact uploads run concurrently in private request workspaces. Node and account admission already exist and must cover all child processes, temporary files and uploads until they finish. Add separate foreground write, read and maintenance budgets; avoid an unbounded queue inside a gateway. Preparation acquires a repository snapshot and operation token, rather than a repository-wide mutex for its lifetime. + +Keep one fenced publisher per repository. Its final command rechecks authorization, branch rules, expected ref OIDs and versions, candidate/check bindings, verified closure and the expected catalog generation. It atomically publishes refs, catalog dependencies and the exact response. Preserve the existing request identity and outcome mechanism. Native success and completed uploads remain insufficient for acknowledgement. + +Simply removing the current mutex is unsafe. Concurrent publishers can share OIDs, overlap native inventories and interleave object sequences. Current ingestion compares storage representation as well as canonical identity; a matching object in another pack can be rejected. Its cache high-water advancement also assumes that the interval since ingestion began contains only covered objects. Replace these assumptions with identity comparison and explicit certified catalog coverage before admitting concurrent preparation. + +Model serialized utilization as `rho = sum(arrival_rate_i × durable_commands_i × mean_serialized_service_seconds_i)`. Include ACL, policy, checks, issue/PR and maintenance writes. The service time includes Cellule's acknowledgement durability gate, not merely SQLite execution. At 35 pushes/s and a 60% push-only utilization budget, one serial command per push allows approximately 17.1 ms mean service time. Six commands per push would allow only 2.86 ms per command. Neither is a promised latency; other traffic further reduces that budget. + +Measure the selected Cellule durability mode. If per-request object-store publication cannot meet the envelope, implement runtime publication grouping or use a qualified existing follower-log durability mode. Do not weaken acknowledgement to local WAL or independently acknowledge commands within an unproven group. Grouped publication retains each request's transaction, outcome, order and recoverable receipt under the same owner fence. Failover must recover every acknowledged member and correctly resolve ambiguous outcomes. + +The implemented preparation protocol adds a durable Begin command before final publication. Its minimum successful push therefore requires two serial durable commands; renewal, claim, aborted attempts, command rejections and runtime handling of replay add measured work. At 35 pushes/s, two equally costly commands and a 60% push-only utilization budget allow approximately 8.57 ms mean durable service per command. If every preparation renews once, the three-command budget becomes 5.71 ms. Include product and maintenance traffic explicitly and measure the distribution, rather than assuming either mean guarantees p99. A fresh lease check is a query, but still consumes SQL-worker, routing and admission capacity. + +Conditional certificate issuance now runs through the privately constructed `PreparedCatalog`, fresh lease/ACL checks and the existing trusted application SQL capability. It reuses the repository secret with a separate BLAKE3 key-derivation domain and binds tenant/application, repository, owner fence, attempt, actor, base, prepared catalog and canonical/input inventories. Its encoded size is at most 1 KiB. The ordinary path must carry this certificate inline into final publication. Durable registration on the existing operation row and independent retention pin is an optional checkpoint; making it mandatory adds a third serialized command and reduces the illustrative budget to 5.71 ms. Final publication must authenticate the certificate and recheck current authority; a recorded checkpoint success is not permission to publish. The issuer, optional checkpoint and typed catalog/ref publisher are implemented primitives. `PreparedCatalog::ref_proof` additionally binds exact PushPlan wire bytes and catalog-verified ancestry evidence. The publisher authenticates that binding and writes catalog, refs and logical outcome atomically while rechecking current policy. CompleteCatalogPush now persists the exact native response, options and optional signed bytes with the catalog/refs in this same transaction, including durable final-policy refusals and native-error outcomes. Its service wrapper, completed-request preflight and receipt-bound exact-response reader exist. Begin does not allocate another namespace for a completed ID, and a restarted gateway can replay through fresh read authority without rebuilding native preparation. Actual HTTP/SSH production wiring remains open. No response/plan/certificate staging commands are required on this inline path. The current typed input envelope is 4 MiB, so very large ref plans still require an immutable plan-root interface. + +Final-generation reconciliation needs its own scheduling contract. Under an illustrative Poisson arrival model, the probability that a base remains current after preparation lasting `t` seconds is `exp(-publication_rate × t)`. At 35 publications/s this is about 3% after 100 ms and 6.3×10⁻¹⁶ after one second. The model is illustrative, but exposes why retrying the entire physical/closure pipeline on catalog CAS failure cannot be the normal path. + +Physical verification is reusable; reconcile only the incoming inventory and affected dependencies against changes since its pinned base. A bounded, fair per-repository coordinator orders ready operations and maintenance, combines compatible deltas where qualified, and binds the final certificate to the selected actual generation. Do not hold that coordinator during uploads or full-history verification. Reconciliation must make progress while publication continues; measure retries and ready-queue residence, and reject overload through bounded admission rather than repeatedly starving the same operation. Changing a token through Claim adds a durable command, so it cannot be an uncounted mechanism on every ordinary push. The executable publisher must define how it binds the reconciled generation without rerunning Begin/Claim per intervening publication. + +#### Retaining a frontier without another durable Claim + +The fresh schema now interprets an independent attempt pin's existing `generation` field as an immutable retention floor. Until that pin is reaped, every generation at or above its floor remains protected, including intervening publications and compactions. Claim and Abort preserve the detached pin. Expiration alone does not make its facts removable. An indexed minimum over the existing generation-pin index bounds eligible facts below the oldest remaining floor; the reaper still deletes at most 512 facts per command and always preserves generation zero and the current root. A schema DELETE guard prevents bypassing this range through another SQL writer. None of these SQL decisions authorizes remote artifact deletion. + +`CheckPreparationFrontier` (query 21) reuses the exact LeaseCheck and returns `PreparationFrontier { lease, current }`. It checks current Write access, repository identity, actor/request/attempt/namespace, exact independent pin binding and expiry, then reads the original floor and current immutable fact in one committed SQL snapshot. It allocates no namespace, does not renew the lease and requires no new durable mutation. The bounded codec rejects a current generation below the floor, foreign repository/format and changed facts at an identical generation. A caller-supplied DTO remains insufficient to construct PreparedCatalog or a publication certificate; final owner fencing and canonical verification remain mandatory. + +`PreparedCatalog::reconcile` now retains the verified incoming directory/source descriptors and the existing admitted closure scratch privately. It checks incoming canonical headers and previously certified external dependency anchors against the queried current catalog in indexed pages of at most 512. An incoming object absent from the new base is supplied by the retained incoming run; an external-only anchor must remain present with its exact canonical body and graph digest. This preserves the verified incoming DAG without decoding packs or walking historical edges again. The output combines selected current roots with the exact incoming delta. The private certificate binds original retention floor/certification and actual selected base separately; final publication checks the unchanged attempt/pin floor and CASes the selected current fact. Optional checkpoints retain their original bytes while comparing exact verified input facts. Root changes still require a fresh certificate, current policy and final ref expectations. + +The primitive currently checks all incoming OIDs and remembered external anchors against the selected catalog. Changed-run-only skipping and its canonical coverage proof still need qualification. The bounded ready-command dispatcher described below now exists. Maintenance scheduling, frontier orchestration, pipelined/grouped root work, production invocation and measured retry/queue residence remain open. Reconciliation alone does not guarantee progress while other writers continually advance the catalog, and a cold root upload cannot become an unmeasured exclusive lane. The existing fixed catalog-file/node caches are reused; no full-history heap inventory is created. Reconciled selections share their original lease deadline and renewal fence rather than gaining another independent lease. Cancellation of one reconciliation preserves the reusable inventory; queued readers retain its file/admission until they drain. + +#### Ready-command admission and outcome ownership + +`PreparedCatalog::ready_push` constructs a private command-19 output using the existing completion factory and Cellule `PreparedCommand`. `PreparedCompaction::ready_compaction` now constructs the corresponding private maintenance command-22 output. Factories retain original verified inputs, mutation identity, wire bytes and owner incarnation. `ReadyPublication`, `PublicationOutcome` and `PublicationError` preserve their types through one coordinator; a maintenance result cannot be read as an HTTP push response. Push inputs remain capped at 4 MiB and maintenance inputs at 4 KiB; larger push plans still need immutable roots. + +The implemented coordinator reserves four of 32 operation slots for maintenance and the remaining 28 for foreground. Encoded-command credits reserve 8 MiB per push and 8 KiB per compaction inside a 256 MiB budget. Per-actor counts are separate by class, and account FIFO rotation runs within each class. Defaults admit eight concurrent durability waits, cap maintenance at two, and start at most three foreground commands before eligible maintenance when a wait slot and maintenance capacity are available. Uncertain work remains charged to its class, preventing maintenance from filling foreground admission. At its concurrency cap, maintenance remains queued while ready foreground progresses. The [shared dispatch contract](design/shared-publication-dispatch.md) defines checked profiles and the new typed APIs. Transaction/network order and overlapping catalog success remain separate from fair starts; uploads and historical verification stay outside dispatch. + +Dropping a ticket/waiter does not cancel admitted execution. A pending outcome, malformed published result or panicking command task stays retained and charged in the service-owned coordinator. `pending` retrieves its ticket internally; `recover` resolves its original Cellule evidence through the fair queue. An authoritative absent result permits execution of the exact retained command, without rebuilding its certificate/response. A committed result is decoded with its original receipt and never reruns the handler. Unknown, expired or unreachable resolution retains the reservation; changed incarnation requires logical outcome recovery before a new attempt and must not be treated as absence. Current ACL/policy/ref/catalog CAS still runs in the authoritative completion transaction. + +A terminal outcome drops the retained wire payload and original prepared inventory before releasing admission. Small resolved push tickets can read their exact HTTP response at that receipt without keeping scratch alive; compaction tickets retain their typed committed/rejected result. Callers that need reconciliation after a durable catalog conflict must keep their own admitted `Arc`, select a current base, create a new final-command mutation identity and reenter the ready queue. Never reuse a rejected final-command identity with changed proof bytes. `stats` exposes admitted/account/queued/in-flight/uncertain counts and reserved command bytes; dispatch/result trace events record queue wait and total admitted residence. + +For graceful shutdown, `close_and_drain` refuses new submissions, waits without canceling dispatched commands, and returns all still-charged unresolved tickets. Recovery remains possible after closing. Keep those tickets until a resolved outcome or an explicit recovery handoff. This dispatcher is not a durable local outbox: process/owner-loss reconstruction, complete retained-root inventory and isolated recovery remain mandatory release work. Its local ownership does not authorize remote deletion. + +This implements fair bounded final-command dispatch and exact uncertainty handling, not fair publication success against a continually moving root. Pipelined/grouped root construction, continuous maintenance preparation, CPU/I/O shares, classified bounded reconciliation retries and end-to-end producer integration remain required. The eight-command default is a starting admission configuration, not a measured throughput claim. Provider durability grouping and every acknowledged member's failover recovery still need qualification. + +Range retention has a measurable cost: at 35 publications/s, a 60-second oldest floor retains about 2,101 generation facts before compaction publications and cleanup lag. An 8,192-fact cap fills in about 234 seconds if the oldest floor never advances; at 70/s it fills in about 117 seconds. Therefore this primitive is a bounded foreground reconciliation mechanism. A four-hour import deadline cannot become a four-hour generation floor or be handled by indefinitely renewing the original floor under traffic. Bulk producer integration must separate long-lived authenticated input retention from short-lived catalog reconciliation leases, reuse the existing attempt/input ownership structures, and advance reconciliation floors only after old readers drain. The stored phase separation now exists: commands 24–26/28 and query 27 reuse operation/pin rows with an unbound NULL generation; a one-way Bind selects the current floor after physical verification without another namespace or pin. The normalized generated binding retains exact composite-FK validation. Bind preserves input expiry rather than invalidating borrowed upload deadlines. A five-minute input renewal can consequently leave a five-minute bound floor; configure and qualify this remaining floor explicitly. Native fixtures reuse pre-bind verified inputs through the existing assembler, and a retention fixture survives 10,240 intervening facts with bounded reaping. The service-owned StagingCoordinator now retains input tasks/results and exact Begin/Renew/Bind commands, automatically renews through fresh queries, drains before binding and preserves uncertainty across observer cancellation. Defaults bound 32 operations/eight per actor and 64 worker/result slots/eight per actor with 8 KiB command reservations. This is process-local ownership, not durable takeover reconstruction; see the [service contract](design/staging-service-lifecycle.md). Production integration, authenticated takeover adoption, larger input limits and admission configuration remain required before selecting the schema; see the [staged input contract](design/staged-input-retention.md). Hitting the cap must fail admission/publication safely; it must never release retained roots to make room. + +Measure coordinator occupancy independently of Cell SQL service: a 100 ms catalog-upload step held exclusively for each ready push caps that lane at 10 pushes/s even if SQL is fast. At 35 pushes/s and a 60% coordinator utilization budget, average exclusive work must fit roughly 17.1 ms per push. Pipeline or group catalog work under a defined generation frontier while preserving each operation's policy decision, ordering and exact durable outcome; adding a second exclusive root-upload lane without measuring it does not solve contention. + +The fresh schema candidate and preparation commands are in `crates/canopy-server/src/packs/publication/`. Begin/claim stamp the runtime's admitted incarnation, epoch and execution sequence. Independent attempt pins retain superseded bases; renewal requires the current owner fence, never shortens an existing pin, and cannot resurrect an expired attempt. The schema enforces exact operation/pin generation and expiry with a deferred composite foreign key and immutable catalog-generation facts. Initial per-repository caps are 1,024 operation rows, 4,096 attempt pins and 8,192 retained generation facts, including generation zero; expired rows count until reaped. These are admission limits, not demonstrated concurrency capacity. Reaping removes at most 512 rows from each class per command using indexed selection. It removes SQL facts only and provides no authority for remote artifact deletion. + +Size these caps against residence time. At 35 new attempts/s, a 60-second retained pin contributes about 2,100 live pins before claims, abandoned attempts and cleanup lag; five-minute pins contribute 10,500 and exceed the initial pin cap. The 1,024-operation cap allows only 29.3 seconds of mean operation residence at that arrival rate before any safety margin. Claim churn consumes additional pin slots. The maintenance scheduler must continuously clear expired rows and expose residence, live/expired counts and quota rejection metrics. At 70 attempts/s, even 60-second pins exceed 4,096. The headroom campaign must report this saturation; population growth requires a qualified cap/lease/admission configuration rather than assuming the 10,000-engineer settings scale unchanged. A worker may release a pin early only after the implementation proves that every associated active or detached read has drained. + +`PreparationBaseResolver` obtains its base through a fresh authorized Cell query, then reuses the admitted catalog loader for batches of at most 512 requested OIDs. It computes its local monotonic deadline from before the query; queue/transport time cannot extend the reported remaining lease. Renewal performs another fresh query after the durable command, including exact command replay, so an old successful reply cannot restart a lease. Failed renewal fences the session. This is the conditional base adapter. The new typed publisher now creates genuine immutable generation facts consumed by this adapter in tests; production orchestration and HTTP/SSH invocation remain missing. Cross-node clock bounds, suspended-reader drain, renewable serving/backup pins and the complete recovery-root inventory must be qualified before remote reclamation. The fresh schema and commands are not yet selected by the production RepositoryModule. + +`CatalogPreparation` now supplies the next local verification boundary. It obtains roots from that adapter, consumes complete isolated physical witnesses, copies exact metadata partitions into closure/directory scratch, uploads those files and derives every new source leaf internally. The incoming directory inventory must match the unique closure inventory; preserved base roots come from the queried generation, rather than caller-provided descriptors. Physical witnesses also bind the actual in-process artifact-store capability, preventing verification in one backend from authorizing references to absent bytes in another. Assembly phases obey the live local deadline, and admitted files retain their private workspace through queued cancellation. The private `PreparedCatalog` result remains conditional on the base and does not acknowledge a push or insert a generation fact. The bounded conditional certificate factory, optional durable checkpoint, target-ref membership/ancestry proof and atomic typed publisher now exist. Current checks/reporters and branch rules are evaluated in the final transaction; ordinary pushes cannot bypass a required-PR rule. The bounded reconciliation and ready-command dispatcher primitives now exist; frontier/maintenance scheduling, reviewed merges and production HTTP/SSH invocation remain required. Assembly verifies its incoming union in one admitted spool, then streams bounded output files into one indexed ingress root. Output partitioning and geometric selection are implemented; full-history input profiles, continuous maintenance scheduling and production integration still need qualification. + +After assembly, PreparedCatalog retains one admitted closure scratch for reconciliation until all selections are released. Metadata, directory and closure builders now start with a 64 KiB main-file cap and 192 KiB charge, then double under the shared disk budget before raising SQLite's cap. They can still reach three times MetadataLimits.max_file_bytes (768 MiB at the 256 MiB default). Sealed metadata/directory files shrink to exact lengths; retained closure keeps its admitted cap until all users and queued workers drain. Release prepared inventory after a resolved outcome and worker drain; uncertain publication still retains inputs. This is local scratch ownership, not a serving or backup pin. See the [admitted growth contract](design/admitted-sqlite-growth.md). + +Local resource bounds need a workload-qualified configuration. Two default construction databases initially charge 384 KiB, and can grow to a combined 1.5 GiB at their 256 MiB main-file ceilings. Physical verification now shares one admitted dependency file per at-most-512-object page, using private exact witness ranges; its dependency file count falls from as many as 512 to one. Full file credit remains held until the last witness/worker drains. See the [edge-spool contract](design/verified-edge-spool.md). Native descendants, pack/index files, SQLite connections and uploaded-file cache costs remain independently admitted; global file/process quotas still require implementation. Growth retries only rolled-back SQLite capacity errors; denied disk credit cannot raise the old page cap. Producers must choose measured class/operation limits and count queued canceled jobs, reservations and actual disk usage. Nontrivial ref ancestry now initially admits 192 KiB and grows before raising the SQLite cap; a 256 MiB main-file limit can still require up to 768 MiB per active walker. Every write reuses the same admitted transaction helper. Exact-catalog pair reuse is exclusive to one walker; error/cancellation fences it and retains queued worker ownership until drain. Its indexed disk queue bounds Rust heap growth but can still traverse substantial history. Native commit-graph acceleration, CPU/file/process admission and end-to-end resource/latency qualification remain open. Current tests cover native 1,200-commit histories and a separate synthetic 100,001-entry scratch queue, not native 100k-history performance. + +The [shared native process guard](design/native-process-ownership.md) now retains supplied cache/input/account owners through Unix inherited-end drain and leader reaping. Cancellation signals the original group and moves ownership to a bounded independent reaper; closed standard streams or a sent signal cannot alone release credits. Decoded-object/history actors reuse the guard and cache fence, but still require explicit preparation native permits. Participating descendants must preserve the descriptor; OS containment and whole-node CPU/RSS/file/process admission remain mandatory. These ownership checks do not qualify workload capacity. + +### 2. Immutable bulk metadata and graph segments + +Remove historical per-object insertion, per-edge closure work and per-object placement updates from the Repository Cell's hot command stream. Store canonical headers and typed edges in immutable metadata segments alongside the native pack/index artifacts. Reuse the existing `objects`, `object_edges` and ordered structural parsing semantics in read-only SQLite sidecars; retain native `.idx`/MIDX and commit-graph formats for Git lookup and traversal. This avoids inventing a delta decoder or a second mutable distributed database. + +Each segment contains repository/format identity, schema version, pack binding, `(oid, kind, size, body_digest)` headers, typed edges and verification/dependency certificates. Seal an exact length, BLAKE3 digest and hashed-part manifest for the sidecar just as for pack/index bytes. A segment is opened read-only and immutable only after whole-artifact verification. Its SQLite page cache is bounded; no serving worker opens every sidecar connection indefinitely. Canonical OIDs remain logical identities; catalog and segment cursors replace a globally growing object sequence for inventory paging. + +The Cell stores the current immutable catalog root and generation, bounded operation records, root-level certificates, refs, policies and durable outcomes. Product tables and identifiers are reused. A catalog uses a bounded-fanout immutable tree of segment descriptors, with incremental path updates; publishing one segment must not rewrite a list of every historical pack. Catalog nodes use the same authenticated artifact transport. Keep the existing proposed per-object SQL DDL as an earlier design validation fixture only; it is not the scale architecture's release schema. + +#### Bounded OID lookup is a separate requirement + +A descriptor tree alone does not satisfy the lookup contract. Git OIDs are distributed across the hash space, so incoming packs have heavily overlapping OID ranges. Looking through every pack or every metadata sidecar remains linear in push history, even when inserting a descriptor costs only a tree path update. The serving cache's current index-handle loop must not become the durable catalog lookup algorithm. + +Use two indexes under the published catalog root: an immutable artifact-descriptor tree and a leveled immutable object directory. Directory runs reuse SQLite `objects` canonical fields and add a preferred metadata/pack descriptor key. Typed edges remain in the source metadata shard; native `.idx` files retain offsets. Compaction may change the descriptor key while preserving the canonical header, edge count and edge digest. No historical object location is updated in the mutable Cell. + +The initial directory contract is: + +- Level 0 has at most 32 overlapping run-set roots. Each root indexes disjoint raw-OID ranges using the same persistent range index as higher levels. A partitioned push consumes one root slot regardless of its file count. Higher levels use a size ratio of 10 and at most 16 levels. A point lookup selects at most one run per root or level, preserving the 48-candidate bound; it never opens every physical pack or downloads a complete descriptor array. +- A normal directory run targets at most 64 MiB. Its records contain fixed-size canonical identity fields and a descriptor key, not object bodies or inline edge lists. Run construction and merging use pages of at most 512 records and admitted disk-backed scratch. The assembler now streams disjoint output runs from its verified incoming spool, indexing and uploading one file at a time. Small spools reuse their original sealed file. Larger spools retain one input page and one output builder, retry bounded prefixes after SQLite page-limit rollback, and check the complete output count/range/canonical inventory before preparation succeeds. Output size classes are independent of the admitted verification-spool class. Wide trees page their source edges separately. This does not remove the verification spool's size limit or implement long-lived bulk input ownership. +- Compare every matching canonical header across the selected runs. Identical overlap is allowed; disagreement in kind, size, body digest or graph inventory is an integrity failure. The highest `location_version` wins physical placement, with a stable source-key tie break. Choosing the newest location cannot hide an identity conflict. An optional Bloom filter accelerates negative lookup but is never a membership, closure or authorization proof. +- Preparation queries incoming OIDs in bounded batches against its pinned directory root. Publication revalidation considers changed runs/ranges since that snapshot and binds the resulting attestation to the actual CAS generation. It does not rescan all historical headers. +- Merge small runs and perform geometric range compaction under separate maintenance admission. If level 0 reaches its bound or maintenance falls behind, reject/defer new preparation through the bounded admission queue. Do not extend the run count or silently accumulate unlimited staged operations. +- A selected root resolves its complete current directory without following a linked list of prior generations. A previous root is an expected CAS/audit value, not an unconditional retained-artifact edge. Compaction carries forward certified canonical and closure facts; it must not make every historic catalog generation a permanent dependency. + +These values are executable initial limits, not measured optimal settings. Qualification must report selected-run count, index-node reads, open SQLite connections, bytes read/written per object, compaction debt and admission rejections over at least 100,000 incremental commits on an imported full history. An empty-history lookup microbenchmark cannot establish the contract. Directory/index/snapshot primitives now exist; complete catalog integration, reader admission/caching and compaction scheduling remain required work. + +The level-0 cap is an active throughput constraint. If every push adds one root, 35 pushes/s fills 32 slots in 0.91 seconds starting from empty; grouping, flush and compaction must free capacity continuously. Partitioning prevents one large push from consuming many root slots, but does not relieve sustained root-arrival pressure. A steady-state test must show compaction service exceeds the measured root/file/byte arrival rate under foreground load. The 48-candidate lookup bound is a complexity bound, not a latency bound: whole-file verification of 48 cold 64 MiB runs could transfer roughly 3 GiB before source metadata. Admit cold hydration separately, keep qualified warm coverage and measure the actual selected files and bytes. Increasing the candidate cap to avoid admission failures defeats the lookup contract. + +The persisted range-index implementation caps directory node fanout at 128, encoded bytes at 64 KiB and height at seven, so one level's cold point lookup traverses at most eight nodes. It caches at most 64 nodes, writes only changed paths, and does not link a new root to every previous generation. Snapshot selection is capped at 48 run candidates. Bounds on descriptor work do not establish cold-read performance: opening a directory run currently verifies its entire file. Preparation workers need qualified warm metadata coverage or a certified native-index/negative-lookup accelerator. A provider download per incoming OID is unacceptable at the target workload. Native MIDX absence can skip canonical lookup only when its complete coverage of the pinned directory generation has been verified; raw index entry counts and structural-only coverage cannot establish that condition. + +Directory runs reuse canonical header folding with seed `BLAKE3("canopy.directory.v1\0" || oid_width:u8)`. Source keys and placement versions are excluded from that canonical inventory, but are bound by the complete file digest. New source records start at placement version one. A compaction's private spool compares the exact expected header/source/version, verifies the replacement metadata header, and increments the version with checked signed-SQL bounds. This local CAS is not the authoritative publish fence: the final Cell command still compares the input catalog generation and admitted operation fence. + +Lookup checks all overlapping canonical headers before selecting the highest placement version; older runs cannot restore an old source merely by being merged later. Source-descriptor lookup and retention must follow the effective selected placement. A collector must distinguish shadowed source fields in old directory runs from effective current source dependencies, and independently retain the complete sources of every pinned older root. Do not implement reclamation by counting raw directory rows or by assuming that a new location immediately releases all old roots. + +Range nodes and directory snapshots use the existing Cellule bounded codec (big-endian integer/length fields) with distinct `canopy.range-index.v1` and `canopy.directory-root.v2` NUL-terminated domains. They use authenticated artifacts under `repos//git-catalogs//nodes/`; directory SQLite files use the same family with `directory/`. Operation-scoped paths preserve incarnation identity across delayed deletion. The source-descriptor tree now shares this bounded path-copy engine with typed 48-byte operation/digest keys, fanout 128 and domain `canopy.source-index.v1` followed by NUL. Source leaves bind metadata and native pack/index descriptors, including authenticated manifests, checksum and physical pack count. Descriptor insertion does not establish artifact existence, canonical body identity or closure. The whole-catalog publication contract, certified verification and production reader orchestration remain unfinished, so these artifacts are not the completed release format. + +The combined catalog root now binds the directory snapshot and source-tree root using the same codec and authenticated catalog-node namespace, with domain `canopy.catalog-root.v1` followed by NUL and a 1 KiB encoding cap. It contains no predecessor-root edge; the authoritative generation and verification facts belong to the final Cell transaction. `CatalogReader` validates root context and resolves a directory entry through its exact source metadata. Read-worker-owned index clients share their bounded caches across immutable snapshots. Root opening verifies at most 48 directory root nodes and one source root node; it does not walk every descendant or replace the trusted publication verifier. SHA-1/SHA-256 golden wire vectors are recorded under `docs/design/catalog-root-v1-*.hex`. + +The local file loader now shares authenticated directory/metadata files across catalog generations. `CatalogFiles` defaults to 32 open/reserved slots and 16 cached files, with a hard maximum of 128 slots; the bound includes borrowed files and detached canceled jobs. Each slot pins the loader's private directory and retains disk admission through cleanup. Fixed 16-way miss locks coalesce equal loads, cache hits compare every descriptor field, and a saturated all-borrowed pool rejects admission. Default SQLite page caches are 8 MiB per file; this is a configured local cache bound, not a whole-process/native RSS guarantee. Batch header reads retain at most 512 input IDs and their at-most-48 directory candidates per ID, group indexed SQL reads by file, compare all overlaps, and verify exact preferred-source headers. They return headers without pinning one source file per result, allowing even a one-slot pool to process a batch. These files and caches do not replace authoritative generation certification, renewable remote reader pins or access checks, and are not yet wired into the server serving/publication path. + +#### Metadata shard implementation boundary + +The current `packs::metadata` implementation reuses canonical object fields and typed edges in sealed, read-only SQLite files. Each file covers an exact contiguous ordinal interval of one checked native pack index. Its descriptor binds repository, creating operation, object format, pack BLAKE3, native checksum, ordinal interval, canonical inventory, exact file length and BLAKE3. Covering a physical pack requires a complete nonoverlapping partition of its index ordinals; equal row counts alone do not suffice. + +The shard inventory seed is `BLAKE3("canopy.segment.v1\0" || oid_width:u8 || pack_digest:32 || first_ordinal:u64le || object_count:u64le)`, followed by the original canonical header folding rule in native raw-OID order with shard-local ordinals starting at zero. The seed's domain string ends in one NUL byte. Edge hashing retains `canopy.edges.v1` and the original typed/deduplicated child ordering. This seed is deliberately distinct from the earlier whole-pack fixture's inventory seed. + +Build batches and read pages are capped at 512 records; the default SQLite cache is 8 MiB, mmap is disabled, and the default main-file limit is 256 MiB with 3× scratch admission for rollback journals. An oversized structural object requires explicitly admitted larger scratch/shard limits; splitting only native object ordinals cannot split that object's edge inventory. None of these defaults is a maximum repository size. + +Upload/download uses the authenticated artifact transport at `...//metadata/`. A downloader validates repository and descriptor bindings, reserves disk before provider reads, checks authenticated parts and complete file bytes, and only then opens SQLite read-only. Queued blocking file jobs retain admission and file ownership through cancellation. Shard membership/shape checks are not canonical body verification or closure attestation: the trusted streaming verifier and directory/publication integration remain required. + +Decoded witnesses now spool dependency occurrences to admitted disk with at most 512 edges per write/read page; blobs require no edge file. Only successful complete canonical inspection constructs a witness. Each growth reserves bytes first, canceled queued blocking jobs retain the file and reservation, and the per-object edge-byte cap is enforced before writing. Metadata import groups at most 512 witnesses in one SQLite transaction, rechecks scratch byte integrity and rejects canonical/typed conflicts. Failed imports roll back and poison the builder against sealing. These bounds cover Rust buffers and admitted scratch, not native Git memory or publication throughput. The service must run metadata assembly on a blocking worker that retains ownership through cancellation, finish the native actor, isolate verified physical pack bytes and certify global typed closure before publication. + +The source-binding implementation validates shard bounds and exact descriptor relationships before catalog insertion. Its constant-space partition checker requires shard intervals in native ordinal order with no gaps, overlaps or duplicates; failure poisons the checker. Verification streams each physical pack once to check whole BLAKE3 and the native header/checksum, checks the exact index BLAKE3/native format, and reuses the checked index for shard endpoint validation. The directory resolver additionally compares the preferred source key, exact metadata descriptor and canonical header in the verified file. None of these checks decodes object bodies or establishes typed closure; the trusted verifier must still perform those checks before publishing a catalog. Production loading must retain admitted immutable files and generation leases, and run blocking verification under bounded worker admission. + +The decoded-object inspector now performs Git OID and body BLAKE3 hashing in 64 KiB chunks alongside a constant-state structural parser. It sends at most 512 typed edge occurrences per sink call, ignores message/signature bytes for graph extraction while hashing them, and retains no whole tree/name/message in Rust. File-backed native index iteration drives the persistent batch reader without a heap OID inventory. Cancellation during a confirmed sink write, malformed frames and sink errors poison the native actor. Sink output is private preparation until inspection completes; the service must discard failed preparations. The disk-backed sink and isolated physical stage below now compose these decoded witnesses; global closure and producer/publication integration remain required. Native Git may allocate independently, so child-process memory admission and qualification remain mandatory. + +The isolated physical verifier now composes these primitives before publication: it admits and downloads the exact authenticated native pair into a fresh fenced cache, verifies complete bytes/checksums and native index relationships, decodes every native ordinal in batches of at most 512, and builds sealed contiguous metadata shards on blocking workers. The workspace contains neither loose objects nor alternates and receives no second pack copy. A successful whole-pack witness binds the exact ordered shard descriptors; failed or canceled assembly cannot finish. The runtime-only `NativePackDescriptor` reuses the pack/index fields already persisted in source records, without changing their wire layout. Tests distinguish delta independence from graph closure: a normalized thin pack includes its appended bases; an unresolved thin pack fails even when an ambient cache has the requested object; a commit-only pack may legitimately reference a tree in another pack. Global typed closure, incremental collision revalidation, producer wiring and authoritative publication remain required. Native index/decode RSS, worst-case hash-order delta locality and whole-operation admission/deadlines still require implementation/qualification; bounded Rust batches alone do not establish those costs. + +The operation-local `packs::closure` stage now reuses the sealed canonical headers, typed edges and directory inventory encoding. It accepts a complete isolated physical witness and checks its exact metadata partition while copying shards one at a time. Only incoming vertices/edges and requested base headers enter an admitted SQLite scratch graph; it neither copies historical graph edges nor retains a repository-wide heap inventory. All incoming OIDs are queried against the bound base, including locally complete overlaps, so a new pack cannot hide a canonical conflict with history. Base lookup batches are ordered and capped at 512, and must name the exact catalog descriptor and generation. A raw catalog lookup cannot establish the required certification status. + +An indexed disk-backed topological queue rejects missing/wrong-kind dependencies and cycles, including cycles in incoming overlaps that also have base headers. Processing batches at most 512 vertex/edge updates per scratch transaction and pages reverse fanout independently; a wide parent set and a deep chain do not require a heap graph. SQLite progress hooks and 64 KiB file-hash checkpoints interrupt canceled blocking work; queued jobs retain file ownership and disk admission. Scratch uses the same bounded cache/main-file limits and 3× journal admission. Its `synchronous=OFF` setting applies only to private disposable scratch, which is never reopened as a recovery root or acknowledged as durable state. Authenticated inputs rebuild it after a crash. Published artifacts and the Cell acknowledgement gate keep their separate durability requirements. + +`ClosureWitness` binds the unique incoming canonical inventory, physical-input digest set and exact base context. Its inventory can be checked directly against a directory run or a bounded sorted header stream without introducing a second canonical encoding. This remains a conditional in-process witness: a trusted production base adapter, renewable base-generation lease, catalog source coverage, changed-generation overlap revalidation and admitted owner-fenced publication are still missing. Caller-supplied certification flags must never cross the service boundary as authority. The new stage is not evidence that the publication path or workload gates pass. + +The trusted verifier independently checks canonical identity and graph semantics, including typed children, ordered commit parents, gitlinks and missing dependencies. Closure attestation names the exact input catalog root, operation attempt and output artifact digests. Every new non-leaf dependency must be in the verified operation or in a certified dependency generation. Pack membership and native index cardinality never establish closure or read authorization. + +Before publication against generation G, the coordinator compares incoming OIDs with G's canonical metadata. Identical canonical headers may overlap packs; conflicting kind, size or body digest is an integrity failure. If G changes, check intersections with the intervening segments and rebuild the attestation for the new generation before retrying CAS. This includes races involving different bytes with the same Git OID; do not rely on a SHA-1 collision being unlikely. Bind the final command to the revalidated root and operation fence. Grouping compatible prepared operations can amortize catalog publication, but each push retains an independent policy decision and outcome. + +Physical placement is resolved by the catalog and native indexes, rather than updated on each historical object row. Compaction preserves canonical header and graph inventory digests, atomically replaces selected catalog segments, and leaves pinned old roots readable. Permanent commit ancestry and closure facts can be stored in certified segments; small SQL caches may accelerate product checks but cannot become the only proof or grow without a retention policy. + +### 3. Independent read workers and authorized fetch reuse + +A Repository Cell's single writer is not the execution location for all Git reads. Read workers consume a certified catalog/ref snapshot, hydrate packs into pinned local cache generations, and perform negotiation and packing locally. Native MIDX, bitmaps and commit graphs accelerate complete inventories. Keep separate full and structural-only coverage certificates for partial clones. Cold-reader routing should prefer an already warm worker while independent readers remain horizontally expandable. + +Generated-pack caching must key repository identity, object format, canonicalized native negotiation inputs (including wants/haves, shallow boundaries and filters), pack-affecting options, certified inventory and a protocol/implementation version. Reuse pack bytes, not an entire HTTP response with session-specific negotiation or sideband framing. Recheck current read access and want reachability on every request before serving a hit. Revocation and hidden refs cannot be bypassed by a cache key. Duplicate misses share bounded generation work; canceled waiters do not release the producer's cache/process pin. + +Bind cached bytes to the certified requested object inventory/closure, rather than unconditionally keying them by the whole repository's newest catalog generation. At 35 pushes/s, unrelated branch publications would otherwise invalidate reuse continuously. Cross-generation reuse requires checking the cached canonical bindings against the current certified state and using the actual negotiated common objects; neither a stale generation nor raw client-supplied haves establish that condition. Physical repacking may change source descriptors while leaving these canonical bindings unchanged. + +Do not add a distributed object-store round trip per small object or spawn `cat-file` per blob. Use native batch processes with bounded queues and supervised process lifetimes. Serve pack artifacts through bounded parallel transfers and account for OS page cache, native process memory, pinned old generations, scratch, output-cache bytes and egress. A process RSS chart excluding Git descendants is insufficient. + +GitLab documents a [pack-objects cache](https://docs.gitlab.com/administration/gitaly/configure_gitaly/) to reduce repeated pack generation. GitHub documents [geometric repacking](https://github.blog/open-source/git/scaling-monorepo-maintenance/) to reduce monorepo maintenance work. Canopy adopts these mechanisms with its own publication and authorization boundaries; those services' results are not Canopy capacity evidence. SQLite permits [one writer per database](https://www.sqlite.org/isolation.html), so read replicas alone do not remove the publication limit. + +### 4. Maintenance and retention without global downtime + +The implemented selected-ingress compactor can merge 2–32 authoritative roots into one, freeing up to 31 level-zero slots. Its defaults bound input to 128 runs/256 MiB and use a 256 MiB merge-spool ceiling with 192 KiB initial charge and up to 768 MiB reservation, with output files capped at 64 MiB. It reuses the existing canonical directory and range-index structures, preserves refs and sources, and verifies the complete unique inventory before issuing a purpose-bound v2 certificate. The admin-only catalog publisher atomically records the outcome. Reconciliation keeps concurrent ingress but rejects a selection already replaced by another compactor. Native fixtures verify both object formats and old-reader access; they do not establish continuous service at 35 pushes/s. Higher-level geometric range compaction, physical pack rewriting, fair maintenance dispatch and retained-root collection remain necessary. + +Adjacent-level range maintenance now supports bounded windows over a source run spanning more target files than one job can admit. StoredRun retains complete authenticated physical facts plus a logical coverage with the existing canonical inventory fold. A bounded consecutive target prefix merges with a verified source prefix; the exact verified suffix reuses the original physical file. The cache shares that file across projections and physical byte admission deduplicates it. Every window scans and folds its complete parent projection before publication; this preserves exact proof meanings but requires amplification measurement and geometric scheduling. Disjoint promotion reuses the original artifact, and exact path replacements retain unrelated files and reject changed overlapping inputs. Source and first overlapping target files still must fit the job's physical budget. The native two-input-window fixture checks repeated progress and complete inventory preservation; it does not establish a stable backlog at the target workload. + +Geometric advisory selection is now implemented with bounded per-level count targets and fixed-size pressure diagnostics. Defaults use a 262,144-object first-level target and ratio four, with urgent ingress at eight roots and at most three urgent jobs before a rotated higher-level turn. Urgent ingress preserves the higher-level rotation position, preventing a continuously pressured first level from starving deeper eligible levels. Per-level indexed cursors revisit retained suffixes and wrap after exhaustion. Native fixtures drain both ingress and geometric debt with exact inventory/ref preservation; these are small correctness checks. A production service must still qualify maintenance/foreground resource shares, uncertain-outcome recovery, owner-loss reconstruction, repeated-parent scan amplification and stable backlog under arrivals. + +Use geometric/incremental compaction under separate resource admission. Do not repack all historical packs after every push. Measure write amplification, live pack count, lookup work and the backlog over at least 100,000 incremental commits. Catalog compaction and cache repacking are different operations; a cache rewrite does not reclaim durable bytes. + +Replace the initial globally drained deletion gate with retention accounting scoped to repository/catalog generations. Roots include the live catalog, active operation inputs/outputs, backups, retained Cellule recovery roots, replica snapshots and reader pins. Persist the retention decision with an owner fence. A reader registers a renewable generation pin before acquiring artifact access, stays within its lease, and stops/reacquires before expiration; publishers retain old generations throughout that window. A collector cannot infer safety from object-store listing or a guessed delay. + +After proving no retained root or valid pin refers to an artifact incarnation, mark it deleting and perform idempotent deletion. The preparation implementation now allocates creating artifact identities from a persistent per-repository watermark: the existing 16-byte descriptor operation field contains eight bytes `CANOPY01` followed by a positive big-endian 64-bit allocation number, bounded by SQLite’s signed integer range. Begin/Claim advance the watermark in the existing admitted transaction; duplicate admission and exact outcome replay do not allocate. Allocation cannot reset when operations/pins are pruned or the owner changes, and exhaustion fails without replacing a live attempt. The logical push ID remains the durable outcome key. Independent pins retain the logical ID, owner fence, creating namespace, base generation and optional bounded certificate after Claim/Abort. Schema constraints bind the active operation to that exact pin; pin identities and registered certificates are immutable. Assembly uses the granted namespace for all incoming artifacts. This closes the namespace-reuse boundary within the authoritative deployment lineage; an isolated rollback restore must use a fresh provider namespace. Remote collection and complete recovery/reader retention remain unimplemented. Test delayed deletes after the same logical push ID is retried, reader pause beyond lease expiry, owner takeover, collector crash and concurrent backup. An offline drain remains a recovery tool, not the normal high-scale maintenance strategy. + +### 5. Shared branch landings + +Independent branch pushes can prepare concurrently. Updates to the same branch still require ordered CAS decisions. Add a merge queue for protected mainline landings that binds candidate inputs, parent order, policy/check versions and the expected base generation. Use CI batching where the team's workflow supports it; a stale tested candidate must be rebuilt or retested before landing. + +100,000 feature-branch commits/day does not imply 100,000 main-branch landing operations/day. Record both distributions and actual CI duration. Storage throughput cannot remove the need to validate overlapping changes. Queue throughput, stale candidate rates and time to land have their own workload gates. + +## Implementation deliverables and dependencies + +These requirements amend packages A–L in the [implementation plan](large-repository-implementation-plan.md). They are mandatory for the large-team claim, not post-release optimizations. + +| Order | Deliverable | Dependency and concrete acceptance | +| --- | --- | --- | +| 1 | Authenticated artifacts and file-backed native indexes | SHA-1/SHA-256, corruption, bounded buffers, immutable namespaces; integrated with all producers/consumers | +| 2 | Sealed metadata segments and incremental catalog | Reuse metadata/edge schema and parser; deterministic inventory; conflicting OID rejection; no repository-wide heap set or Cell SQL insert per historical object | +| 3 | Fenced concurrent preparation and publication | Requires catalog CAS and explicit coverage; independent pushes progress while another is paused; duplicate ID replay, overlapping objects, ref ABA and policy races pass | +| 4 | Cellule publication qualification/grouping if needed | Measure complete durability gate on selected provider; crash/takeover recovers every acknowledged request; queue stays bounded at mixed target rate | +| 5 | Read workers, native batch pool and fetch cache | Requires certified generations; authorized hits, bounded duplicate misses, cold eviction, SHA-256 and partial-clone coverage pass | +| 6 | Online compaction, retention and isolated restore | Requires complete retained-root inventory and reader fencing; no referenced artifact deleted; byte amplification and backlog remain bounded | +| 7 | Protected branch merge queue | Tested candidate bindings survive contention, cancellation and policy changes | +| 8 | Full-history and large-team qualification | Target workload and faults below; retain source/provider/hardware/driver provenance and every failure | + +Do not implement the superseded per-object preferred-location schema and then design a second migration. The fresh-format marker, dependency pin, executable release DDL and persisted contract must describe the selected catalog/segment architecture together. Intermediate development commits may be unreleasable; no legacy reader or durable inline fallback is added. + +## Qualification gates + +Provisional latency gates below apply to ordinary incremental pushes up to 1 MiB of newly transferred pack bytes and fetches up to 10 MiB, measured on a declared low-latency network. Full clones and large imports are evaluated by throughput and time to first byte separately. Publish payload distributions alongside percentiles; never discard larger requests to make a claim about the full workload. + +| Campaign | Offered load and duration | Required result | +| --- | --- | --- | +| Working-day soak | 4 pushes/s plus 70 fetches/s for 8 hours, background compaction and product traffic | Stable queues/resources and no growing maintenance backlog | +| Peak | 35 pushes/s plus 700 fetches/s for 30 minutes | Proposed p99: push ≤5s, warm fetch first byte ≤1s; infrastructure error rate <0.1%; count admission rejection and driver drops | +| Headroom | 70 pushes/s plus 1,400 fetches/s, staged ramp | Identify saturation; bounded rejection/recovery; no corruption or acknowledgement loss | +| Single hot repository | At least 90% of peak traffic targets one imported full-history repository | Meets same applicable gates; no substitution with a many-repository aggregate | +| Correctness/failure | Owner and read-worker loss, lost replies, corrupted parts/indexes, exhausted disk, reader/collector races during load | Zero lost acknowledged updates, false success, unauthorized reads or unsafe deletion | +| Persistence | Isolated restore into fresh nodes after deleting all local caches | Every acknowledged ref/object and required product state verifies independently | + +Use pinned full-history Kubernetes, Linux mainline (stable histories separately) and `chromium/src`, plus adversarial SHA-1/SHA-256 synthetic histories. Record whether the corpus is full, shallow or filtered. A Chrome-sized checkout includes dependencies and is a separate multi-repository campaign. + +Reuse `scripts/benchmark_repositories.py` for scheduled independent-branch `push_commit` traffic and its digest-bound acknowledged-write verifier. Select `--active-repositories 1` for the hot repository and run against a manifest whose tip actually names the imported full-history corpus. Its tiny seed fixture is insufficient. Concurrent push/fetch drivers, multiple bounded client hosts and a synchronized mixed-load campaign are still required; the current 256-client driver limit can itself saturate. Report driver saturation rather than counting unsent arrivals as successes. + +Add server phase metrics for admission, native decode/index/verification, upload, catalog revalidation, Cell queue/SQL/durable publication, response replay, fetch negotiation/generation, cache hits and retention/compaction. Capture mean and p99 serialized service time, completions inside the offered window, queue depth, retry amplification, object-store operations/bytes, CPU, RSS including descendants, disk and network. Verify writes after each owner-loss event using the saved original outcomes, not a blind retry with new identities. + +No capacity statement is approved until these campaigns pass for an identified revision, provider, durability mode and resource envelope. The architecture is intended to reach the workload; current code and measurements do not establish that it does. + +Native workers now use the [shared four-dimension admission pool](design/native-resource-admission.md), configured once per node and retained through process/descendant drain. Separate maintenance and foreground caps protect both classes. CPU units and memory/descriptor claims remain estimates; they do not establish kernel isolation or full-history bounds. Native account shares, fair preparation waiting and the declared mixed-load campaign remain qualification work. diff --git a/docs/operations.md b/docs/operations.md index 7cb59ef..747dac0 100644 --- a/docs/operations.md +++ b/docs/operations.md @@ -187,3 +187,5 @@ Incomplete uploads, decode failures and unavailable or uncertain response publication can still return transport errors. Response bodies are limited to 64 MiB and serialized response headers to 64 KiB. Completed records and abandoned staging chunks currently have no expiry or collector and consume repository storage. + +Native Git requires an explicit native_limits configuration. See the [native resource admission contract](design/native-resource-admission.md) for vector profiles, reserved maintenance shares, HTTP 503 on exhaustion and drain ownership. Choose measured claims within OS limits and server/filesystem headroom; the default production and small evaluation profiles are estimates. The container checker includes these claims in memory and descriptor headroom checks. Missing configuration rejects startup; each foreground/maintenance share must fit a read/pack pipeline. Shutdown permanently closes native admission and waits for both class owners before Cell shutdown, workspace release and lease withdrawal. Poisoned accounting or quarantined native work can retain this drain until process restart; an elapsed shutdown deadline is not a successful drain acknowledgement. diff --git a/scripts/benchmark_container.py b/scripts/benchmark_container.py index d94e470..89e94dc 100644 --- a/scripts/benchmark_container.py +++ b/scripts/benchmark_container.py @@ -8,6 +8,7 @@ import argparse import hashlib import json +from native_limits import fixture_native_limits import os from pathlib import Path import random @@ -90,7 +91,7 @@ def sample(): "node_id": str(uuid.uuid4()), "fleet_digest": "11" * 32, "image_digest": "22" * 32, "owner": "canopy", "listen": "0.0.0.0:8080", "public_url": url, "peer_endpoint": "https://density.example.invalid", "data_dir": "/var/lib/canopy", - "local_disk_limit_bytes": 1536 * 1024**2, + "native_limits": fixture_native_limits(), "local_disk_limit_bytes": 1536 * 1024**2, "max_active_repositories": args.active_repositories} (directory / "config.json").write_text(json.dumps(config)) profile_file = directory / "profile.json" diff --git a/scripts/benchmark_large_repository.py b/scripts/benchmark_large_repository.py index 43f8279..bab54e4 100644 --- a/scripts/benchmark_large_repository.py +++ b/scripts/benchmark_large_repository.py @@ -93,6 +93,7 @@ def main(): "provider_image": metadata["provider_image"], "git_version": local_eval.run("git", "--version"), "host_platform": platform.platform(), "host_cpu_count": os.cpu_count(), "disk_limit_bytes": config["local_disk_limit_bytes"], + "native_limits": config["native_limits"], "active_repository_limit": config["max_active_repositories"], "node_process_tree_peak_rss_bytes": 0, "started_at_utc": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), diff --git a/scripts/check_container.py b/scripts/check_container.py index 55d9fe8..aded3ed 100644 --- a/scripts/check_container.py +++ b/scripts/check_container.py @@ -58,6 +58,27 @@ def verify(container): raise ValueError("container data directory or listener differs from the profile") if not 0 < config["local_disk_limit_bytes"] <= scratch: raise ValueError("application disk admission exceeds the filesystem capacity") + native = config["native_limits"] + total = native["total"] + reserved = native["maintenance_reserved"] + read, pack = native["read"], native["pack"] + keys = {"processes", "cpu_units", "memory_bytes", "descriptors"} + for profile in (total, reserved, read, pack): + if set(profile) != keys or any(type(value) is not int or value <= 0 for value in profile.values()): + raise ValueError("native admission vectors must contain positive integer dimensions") + if read["processes"] != 1 or pack["processes"] != 1: + raise ValueError("native profiles must charge one process slot per guard") + if (read["cpu_units"] < 1 or read["memory_bytes"] < 128 * 1024**2 or read["descriptors"] < 32 + or pack["cpu_units"] < 4 or pack["memory_bytes"] < 512 * 1024**2 or pack["descriptors"] < 64): + raise ValueError("native profiles fall below minimum admission estimates") + for key in keys: + pipeline = read[key] + pack[key] + if pipeline > reserved[key] or pipeline > total[key] - reserved[key]: + raise ValueError("each native resource share must fit its read/pack pipeline") + # tmpfs consumes this same cgroup's memory. Leave at least 256 MiB for the + # server/runtime; these estimates still require measured headroom. + if total["memory_bytes"] + config["local_disk_limit_bytes"] + 256 * 1024**2 > int(memory): + raise ValueError("native claims and admitted tmpfs leave insufficient server memory headroom") descriptor_line = next(line for line in output("docker", "exec", container, "cat", "/proc/1/limits").splitlines() if line.startswith("Max open files")) soft, hard = descriptor_line.split()[3:5] @@ -65,13 +86,13 @@ def verify(container): raise ValueError("soft and hard open-file limits must both be 16384") # The pinned runtime reserves eight descriptors per active Cell. Directory # also consumes one slot; keep 1024 descriptors for sockets, Git and I/O. - required_descriptors = 8 * (config["max_active_repositories"] + 1) + 1024 + required_descriptors = 8 * (config["max_active_repositories"] + 1) + total["descriptors"] + 1024 if required_descriptors > int(soft): raise ValueError("active Cell admission leaves insufficient descriptor headroom") return {"memory_bytes": int(memory), "swap_bytes": 0, "cpu_quota": int(quota), "cpu_period": int(period), "processes_and_threads": int(tasks), "scratch_bytes": scratch, - "file_descriptors_per_process": int(soft)} + "file_descriptors_per_process": int(soft), "native_admission": native} def main(): diff --git a/scripts/local_eval.py b/scripts/local_eval.py index 3fcab71..0ab88a7 100644 --- a/scripts/local_eval.py +++ b/scripts/local_eval.py @@ -7,6 +7,7 @@ import argparse import hashlib import json +from native_limits import fixture_native_limits import os from pathlib import Path import secrets @@ -100,7 +101,7 @@ def initialize(args): "public_url": metadata["url"], "listen": f"127.0.0.1:{args.port}", "peer_endpoint": "https://local-evaluation.example.invalid", "data_dir": str(state / "node"), - "local_disk_limit_bytes": args.disk_gib * 1024**3, + "native_limits": fixture_native_limits(), "local_disk_limit_bytes": args.disk_gib * 1024**3, "max_active_repositories": args.active_repositories, } save(state / "secrets.json", credentials, private=True) diff --git a/scripts/native_limits.py b/scripts/native_limits.py new file mode 100644 index 0000000..8f653d0 --- /dev/null +++ b/scripts/native_limits.py @@ -0,0 +1,17 @@ +"""Explicit native claims for small evaluation fixtures; no capacity qualification.""" + + +def fixture_native_limits(): + # Two GiB of native claims leaves space for the 1.5-GiB tmpfs and server + # within the four-GiB container profile. CPU units are admission weights, + # not a kernel CPU quota; cgroups still enforce the two-CPU fixture quota. + return { + "total": {"processes": 16, "cpu_units": 16, + "memory_bytes": 2 * 1024**3, "descriptors": 1024}, + "maintenance_reserved": {"processes": 2, "cpu_units": 5, + "memory_bytes": 768 * 1024**2, "descriptors": 128}, + "read": {"processes": 1, "cpu_units": 1, + "memory_bytes": 128 * 1024**2, "descriptors": 32}, + "pack": {"processes": 1, "cpu_units": 4, + "memory_bytes": 512 * 1024**2, "descriptors": 64}, + } diff --git a/scripts/serve_three_gateways.py b/scripts/serve_three_gateways.py index dd9a751..26f604c 100644 --- a/scripts/serve_three_gateways.py +++ b/scripts/serve_three_gateways.py @@ -11,6 +11,7 @@ from contextlib import ExitStack import hashlib import json +from native_limits import fixture_native_limits import os from pathlib import Path import secrets @@ -37,7 +38,7 @@ def serve(args): settings = {"storage_url": args.storage_url.rstrip("/") + "/" + uuid.uuid4().hex, "tenant_id": str(uuid.uuid4()), "application_id": str(uuid.uuid4()), "fleet_digest": "11" * 32, "image_digest": "22" * 32, "owner": "canopy", - "local_disk_limit_bytes": 1536 * 1024**2, + "native_limits": fixture_native_limits(), "local_disk_limit_bytes": 1536 * 1024**2, "max_active_repositories": 100} if args.max_active_repositories is not None: settings["max_active_repositories"] = args.max_active_repositories @@ -79,6 +80,7 @@ def serve(args): binary_sha256=hashlib.sha256(args.binary.read_bytes()).hexdigest(), public_url=front.url, work_dir=str(args.work_dir), max_active_repositories_per_node=settings["max_active_repositories"], + native_limits_per_node=settings["native_limits"], topology="three host processes, loopback TCP proxy, TLS peer proxies; external S3 fixture", balance="round-robin TCP connections; keep-alive retains its backend", proxy_limits={"connections": front.max_connections, "buffer_bytes": front.buffer_bytes}, diff --git a/scripts/serve_two_gateways.py b/scripts/serve_two_gateways.py index 2e5bf08..a832d08 100644 --- a/scripts/serve_two_gateways.py +++ b/scripts/serve_two_gateways.py @@ -72,6 +72,7 @@ def main(): print(json.dumps({"ingresses": ingresses, "corpus_repositories": len(manifest["repositories"]), "max_active_repositories_per_node": settings["max_active_repositories"], + "native_limits_per_node": settings["native_limits"], "work_dir": str(args.work_dir)}), flush=True) while not stop.wait(0.5): for index, process in enumerate(processes): diff --git a/scripts/smoke_container.py b/scripts/smoke_container.py index 96ca067..8ebab3f 100644 --- a/scripts/smoke_container.py +++ b/scripts/smoke_container.py @@ -2,6 +2,7 @@ """Qualify the bounded container against a caller-provided disposable S3 prefix.""" import argparse import json +from native_limits import fixture_native_limits import os from pathlib import Path import subprocess @@ -71,7 +72,7 @@ def qualify(args): "tenant_id": str(uuid.uuid4()), "application_id": str(uuid.uuid4()), "node_id": str(uuid.uuid4()), "fleet_digest": "11" * 32, "image_digest": "22" * 32, "owner": "canopy", "listen": "0.0.0.0:8080", "public_url": url, "peer_endpoint": "https://container.example.invalid", - "data_dir": "/var/lib/canopy", "local_disk_limit_bytes": 1536 * 1024**2, + "data_dir": "/var/lib/canopy", "native_limits": fixture_native_limits(), "local_disk_limit_bytes": 1536 * 1024**2, "max_active_repositories": 3, } config_file = directory / "config.json" diff --git a/scripts/smoke_s3_activation.py b/scripts/smoke_s3_activation.py index aeae8f4..761a643 100644 --- a/scripts/smoke_s3_activation.py +++ b/scripts/smoke_s3_activation.py @@ -5,6 +5,7 @@ import hashlib import http.client import json +from native_limits import fixture_native_limits import os from pathlib import Path import signal @@ -22,7 +23,7 @@ def qualify(args): "tenant_id": str(uuid.uuid4()), "application_id": str(uuid.uuid4()), "node_id": str(uuid.uuid4()), "fleet_digest": "11" * 32, "image_digest": "22" * 32, "owner": "canopy", "peer_endpoint": "https://activation.example.invalid", - "local_disk_limit_bytes": 512 * 1024**2, "max_active_repositories": 64} + "native_limits": fixture_native_limits(), "local_disk_limit_bytes": 512 * 1024**2, "max_active_repositories": 64} report = {"binary_sha256": hashlib.sha256(args.binary.read_bytes()).hexdigest(), "repositories": 64, "concurrency": args.concurrency, "cold_recovery_passed": False, "git_recovery_passed": False, "shutdown_passed": False} diff --git a/scripts/smoke_s3_cache.py b/scripts/smoke_s3_cache.py index 0670a8d..05f5add 100644 --- a/scripts/smoke_s3_cache.py +++ b/scripts/smoke_s3_cache.py @@ -3,6 +3,7 @@ import argparse import hashlib import json +from native_limits import fixture_native_limits import os from pathlib import Path import re @@ -128,7 +129,7 @@ def qualify(args): "tenant_id": str(uuid.uuid4()), "application_id": str(uuid.uuid4()), "node_id": str(uuid.uuid4()), "fleet_digest": "11" * 32, "image_digest": "22" * 32, "owner": "canopy", "peer_endpoint": "https://cache.example.invalid", - "local_disk_limit_bytes": 512 * 1024**2, "max_active_repositories": 3} + "native_limits": fixture_native_limits(), "local_disk_limit_bytes": 512 * 1024**2, "max_active_repositories": 3} report = {"binary_sha256": hashlib.sha256(args.binary.read_bytes()).hexdigest(), "storage_url": settings["storage_url"], "cache_reuse_passed": False, "recovery_passed": False, "shutdown_passed": False, diff --git a/scripts/smoke_s3_logging.py b/scripts/smoke_s3_logging.py index 092d8fc..5a8e219 100644 --- a/scripts/smoke_s3_logging.py +++ b/scripts/smoke_s3_logging.py @@ -7,6 +7,7 @@ """ import argparse import json +from native_limits import fixture_native_limits import os from pathlib import Path import re @@ -31,7 +32,7 @@ def qualify(args): "owner": "canopy", "listen": address, "public_url": "http://" + address, "peer_endpoint": "https://logging.example.invalid", "data_dir": str(args.work_dir / "node"), - "local_disk_limit_bytes": 256 * 1024**2, "max_active_repositories": 3, + "native_limits": fixture_native_limits(), "local_disk_limit_bytes": 256 * 1024**2, "max_active_repositories": 3, } config_path = args.work_dir / "config.json" config_path.write_text(json.dumps(config)) diff --git a/scripts/smoke_s3_process.py b/scripts/smoke_s3_process.py index f336a25..c388fb0 100644 --- a/scripts/smoke_s3_process.py +++ b/scripts/smoke_s3_process.py @@ -10,6 +10,7 @@ from http.server import BaseHTTPRequestHandler, HTTPServer import hashlib import json +from native_limits import fixture_native_limits import os from pathlib import Path import secrets @@ -571,7 +572,7 @@ def main(): "image_digest": "22" * 32, "owner": "canopy", "peer_endpoint": "https://smoke.example.invalid", - "local_disk_limit_bytes": 1 << 30, + "native_limits": fixture_native_limits(), "local_disk_limit_bytes": 1 << 30, "max_active_repositories": 3, } workspace = (nullcontext(tempfile.mkdtemp(prefix="canopy-process-", dir=args.work_parent)) diff --git a/scripts/smoke_s3_transfer_fairness.py b/scripts/smoke_s3_transfer_fairness.py index 94f1257..455929e 100644 --- a/scripts/smoke_s3_transfer_fairness.py +++ b/scripts/smoke_s3_transfer_fairness.py @@ -3,6 +3,7 @@ import argparse import hashlib import json +from native_limits import fixture_native_limits import os from pathlib import Path import signal @@ -62,7 +63,7 @@ def qualify(args): "tenant_id": str(uuid.uuid4()), "application_id": str(uuid.uuid4()), "node_id": str(uuid.uuid4()), "fleet_digest": "11" * 32, "image_digest": "22" * 32, "owner": "canopy", "peer_endpoint": "https://fairness.example.invalid", - "local_disk_limit_bytes": 512 * 1024**2, "max_active_repositories": 3} + "native_limits": fixture_native_limits(), "local_disk_limit_bytes": 512 * 1024**2, "max_active_repositories": 3} report = {"binary_sha256": hashlib.sha256(args.binary.read_bytes()).hexdigest(), "storage_url": settings["storage_url"], "objects": 16, "object_bytes": 1024**2, "git_lfs_version": git("lfs", "version").decode(), "passed": False}